pi-crew 0.9.59 → 0.9.61

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -2,8 +2,10 @@ import * as fs from "node:fs";
2
2
  import * as os from "node:os";
3
3
  import * as path from "node:path";
4
4
  import { errors } from "../../errors.ts";
5
+ import { logInternalError } from "../../utils/internal-error.ts";
5
6
  import { fuzzyResolveModelId } from "./model-resolver.ts";
6
7
  import { checkModelScope } from "./model-scope.ts";
8
+ import { providerRankFromQuota, deprioritizedProviders as quotaDeprioritizedProviders } from "./provider-quota.ts";
7
9
 
8
10
  export interface AvailableModelInfo {
9
11
  provider: string;
@@ -43,6 +45,21 @@ interface PiProviderConfigLike {
43
45
  }
44
46
 
45
47
  function modelInfoFromUnknown(value: unknown): AvailableModelInfo | undefined {
48
+ // A plain `"provider/id"` (or bare `"id"`) string is a valid model reference.
49
+ // The child-process path receives pi's `Model` object, but the live-session
50
+ // path and the background path (manifest-persisted) carry strings; both must
51
+ // resolve identically or parent-model inheritance silently disappears.
52
+ if (typeof value === "string") {
53
+ const raw = value.trim();
54
+ if (!raw) return undefined;
55
+ const slashIdx = raw.indexOf("/");
56
+ if (slashIdx <= 0) return { provider: "", id: raw, fullId: raw };
57
+ return {
58
+ provider: raw.slice(0, slashIdx),
59
+ id: raw.slice(slashIdx + 1),
60
+ fullId: raw,
61
+ };
62
+ }
46
63
  if (!value || typeof value !== "object" || Array.isArray(value)) return undefined;
47
64
  const record = value as ModelLike;
48
65
  if (typeof record.provider !== "string" || typeof record.id !== "string") return undefined;
@@ -70,6 +87,21 @@ export function modelStringFromUnknown(model: unknown): string | undefined {
70
87
  return modelInfoFromUnknown(model)?.fullId;
71
88
  }
72
89
 
90
+ /**
91
+ * Normalize any model reference (pi `Model` object, `"provider/id"`, bare id)
92
+ * to its canonical string form. Returns undefined for unrecognized input.
93
+ */
94
+ export function modelRefToString(model: unknown): string | undefined {
95
+ return modelInfoFromUnknown(model)?.fullId;
96
+ }
97
+
98
+ /** Provider segment of a `"provider/id"` reference, or undefined for bare ids. */
99
+ export function providerOfModelRef(model: string | undefined): string | undefined {
100
+ if (!model) return undefined;
101
+ const slashIdx = model.indexOf("/");
102
+ return slashIdx > 0 ? model.slice(0, slashIdx) : undefined;
103
+ }
104
+
73
105
  function uniqueModelInfos(models: AvailableModelInfo[]): AvailableModelInfo[] {
74
106
  const seen = new Set<string>();
75
107
  return models.filter((model) => {
@@ -130,19 +162,103 @@ function modelsJsonInfos(modelsJson: PiModelsJsonLike | undefined): AvailableMod
130
162
  return infos;
131
163
  }
132
164
 
133
- export function configuredModelInfosFromPiConfig(cwd?: string): AvailableModelInfo[] {
165
+ /**
166
+ * Providers that have a discoverable credential, so a model from them is
167
+ * plausibly runnable. Mirrors (loosely) pi's own `configuredProviders` set,
168
+ * which is what `ModelRegistry.getAvailable()` filters on — the raw-JSON
169
+ * fallback below has no registry, so without this check a background run would
170
+ * happily queue models the user has no key for and burn a child spawn per one.
171
+ *
172
+ * Detection channels (existence only — no credential value is ever read into
173
+ * a return value or a log):
174
+ * • a top-level provider key in `~/.pi/agent/auth.json`
175
+ * • `apiKey` / `baseUrl` set on the provider in `models.json` (local
176
+ * providers such as ollama are keyless but carry a baseUrl)
177
+ * • an `<PROVIDER>_API_KEY` environment variable
178
+ */
179
+ export function providersWithCredentials(modelsJson: PiModelsJsonLike | undefined, env: NodeJS.ProcessEnv = process.env): Set<string> {
180
+ const providers = new Set<string>();
181
+ const auth = readJsonObject(path.join(piAgentDir(), "auth.json"));
182
+ for (const key of Object.keys(auth ?? {})) providers.add(key);
183
+ if (modelsJson?.providers && typeof modelsJson.providers === "object" && !Array.isArray(modelsJson.providers)) {
184
+ for (const [provider, rawConfig] of Object.entries(modelsJson.providers as Record<string, unknown>)) {
185
+ if (!rawConfig || typeof rawConfig !== "object" || Array.isArray(rawConfig)) continue;
186
+ const config = rawConfig as { apiKey?: unknown; baseUrl?: unknown };
187
+ if (typeof config.apiKey === "string" && config.apiKey.trim()) providers.add(provider);
188
+ if (typeof config.baseUrl === "string" && config.baseUrl.trim()) providers.add(provider);
189
+ }
190
+ }
191
+ for (const key of Object.keys(env)) {
192
+ const match = /^([A-Z0-9]+(?:_[A-Z0-9]+)*)_API_KEY$/.exec(key);
193
+ if (match && env[key]?.trim()) providers.add(match[1]!.toLowerCase().replace(/_/g, "-"));
194
+ }
195
+ return providers;
196
+ }
197
+
198
+ interface ConfiguredModelCacheEntry {
199
+ signature: string;
200
+ all: AvailableModelInfo[];
201
+ credentialed: AvailableModelInfo[];
202
+ }
203
+
204
+ /** US-012: avoid re-reading 3 JSON files on every routing build (and every retry). */
205
+ const configuredModelCache = new Map<string, ConfiguredModelCacheEntry>();
206
+
207
+ function fileSignature(filePath: string): string {
208
+ try {
209
+ const stat = fs.statSync(filePath);
210
+ return `${stat.mtimeMs}:${stat.size}`;
211
+ } catch {
212
+ return "-";
213
+ }
214
+ }
215
+
216
+ export interface ConfiguredModelOptions {
217
+ /** Drop models whose provider has no discoverable credential. */
218
+ requireCredentials?: boolean;
219
+ }
220
+
221
+ export function configuredModelInfosFromPiConfig(cwd?: string, options?: ConfiguredModelOptions): AvailableModelInfo[] {
134
222
  const agentDir = piAgentDir();
135
- const globalSettings = readJsonObject(path.join(agentDir, "settings.json")) as PiSettingsLike | undefined;
136
- const projectSettings = cwd ? (readJsonObject(path.join(cwd, ".pi", "settings.json")) as PiSettingsLike | undefined) : undefined;
223
+ const globalSettingsPath = path.join(agentDir, "settings.json");
224
+ const modelsJsonPath = path.join(agentDir, "models.json");
225
+ const authPath = path.join(agentDir, "auth.json");
226
+ const projectSettingsPath = cwd ? path.join(cwd, ".pi", "settings.json") : undefined;
227
+ const cacheKey = `${agentDir}\u0000${cwd ?? ""}`;
228
+ const signature = [globalSettingsPath, modelsJsonPath, authPath, ...(projectSettingsPath ? [projectSettingsPath] : [])]
229
+ .map(fileSignature)
230
+ .join("|");
231
+ const cached = configuredModelCache.get(cacheKey);
232
+ if (cached?.signature === signature) return options?.requireCredentials ? cached.credentialed : cached.all;
233
+
234
+ const globalSettings = readJsonObject(globalSettingsPath) as PiSettingsLike | undefined;
235
+ const projectSettings = projectSettingsPath ? (readJsonObject(projectSettingsPath) as PiSettingsLike | undefined) : undefined;
137
236
  const effectiveSettings = {
138
237
  ...(globalSettings ?? {}),
139
238
  ...(projectSettings ?? {}),
140
239
  };
141
240
  const defaultModel = settingsModelInfo(effectiveSettings);
142
- return uniqueModelInfos([
143
- ...(defaultModel ? [defaultModel] : []),
144
- ...modelsJsonInfos(readJsonObject(path.join(agentDir, "models.json")) as PiModelsJsonLike | undefined),
145
- ]);
241
+ const modelsJson = readJsonObject(modelsJsonPath) as PiModelsJsonLike | undefined;
242
+ const all = uniqueModelInfos([...(defaultModel ? [defaultModel] : []), ...modelsJsonInfos(modelsJson)]);
243
+ const credentialedProviders = providersWithCredentials(modelsJson);
244
+ // The settings.json default model is kept regardless: it is the user's
245
+ // explicit choice and pi resolves its auth through channels we do not model
246
+ // here (OAuth, keychain, provider extensions).
247
+ const credentialed = all.filter(
248
+ (info) => info.fullId === defaultModel?.fullId || credentialedProviders.has(info.provider.toLowerCase()),
249
+ );
250
+ configuredModelCache.set(cacheKey, { signature, all, credentialed });
251
+ // LRU cap: prevent unbounded growth across many distinct agent-dir/cwd combos.
252
+ if (configuredModelCache.size > 32) {
253
+ const oldestKey = configuredModelCache.keys().next().value;
254
+ if (oldestKey !== undefined) configuredModelCache.delete(oldestKey);
255
+ }
256
+ return options?.requireCredentials ? credentialed : all;
257
+ }
258
+
259
+ /** @internal Test seam — clear the mtime-keyed configured-model cache. */
260
+ export function __test_resetConfiguredModelCache(): void {
261
+ configuredModelCache.clear();
146
262
  }
147
263
 
148
264
  export function splitThinkingSuffix(model: string): {
@@ -304,17 +420,182 @@ function isAvailableModel(model: string, availableModels: AvailableModelInfo[] |
304
420
  return fuzzy !== undefined;
305
421
  }
306
422
 
423
+ /**
424
+ * Ordering + budget policy for the AUTO portion of the fallback chain (the
425
+ * models appended from the registry / pi config that nobody declared).
426
+ *
427
+ * Explicit declarations (tool override, step, team role, agent model and the
428
+ * declared `fallbackModels`) are never reordered and never truncated — only the
429
+ * auto tail is governed here.
430
+ */
431
+ export interface ModelFallbackPolicy {
432
+ /**
433
+ * How many auto-appended models to keep. `undefined` = keep all (legacy).
434
+ * Each extra candidate multiplies the worst-case child-spawn budget by
435
+ * `maxAttempts + 1`, so an unbounded tail on a large catalogue is the main
436
+ * cost amplifier.
437
+ */
438
+ maxAutoFallbacks?: number;
439
+ /**
440
+ * `"parentFirst"` keeps the auto tail on the same provider as the model
441
+ * actually in use before crossing to another provider when a policy is
442
+ * configured or quota data enriches it — same auth, similar cost/latency
443
+ * profile. `"asIs"` preserves raw catalogue order. Without explicit
444
+ * configuration, auto tail stays catalogue order.
445
+ */
446
+ order?: "parentFirst" | "asIs";
447
+ /** Lower rank = try earlier. Populated from provider quota when available. */
448
+ providerRank?: Record<string, number>;
449
+ /** Providers at/near their quota limit — pushed to the back of the tail. */
450
+ deprioritizedProviders?: string[];
451
+ /** Drop pi-config models whose provider has no discoverable credential. */
452
+ requireCredentials?: boolean;
453
+ /**
454
+ * When true (default), populate `providerRank` and `deprioritizedProviders`
455
+ * from the provider-quota cache before ordering the auto tail. Set to false
456
+ * to disable quota-aware ordering entirely.
457
+ */
458
+ quotaAwareOrdering?: boolean;
459
+ }
460
+
461
+ /**
462
+ * Order the auto tail. Stable: equal-priority entries keep catalogue order.
463
+ * Priority, most significant first:
464
+ * 1. not quota-exhausted
465
+ * 2. same provider as the anchor (the model we are actually going to run)
466
+ * 3. explicit provider rank (quota-derived), unknown providers last
467
+ */
468
+ export function orderAutoFallbacks(candidates: string[], policy: ModelFallbackPolicy | undefined, anchorProvider?: string): string[] {
469
+ if (!policy || policy.order === "asIs") return candidates;
470
+ const deprioritized = new Set((policy.deprioritizedProviders ?? []).map((p) => p.toLowerCase()));
471
+ const rank = policy.providerRank ?? {};
472
+ const rankFor = (provider: string | undefined): number => {
473
+ if (!provider) return Number.MAX_SAFE_INTEGER;
474
+ const value = rank[provider] ?? rank[provider.toLowerCase()];
475
+ return typeof value === "number" ? value : Number.MAX_SAFE_INTEGER;
476
+ };
477
+ return candidates
478
+ .map((model, index) => {
479
+ const provider = providerOfModelRef(model);
480
+ return {
481
+ model,
482
+ index,
483
+ exhausted: provider && deprioritized.has(provider.toLowerCase()) ? 1 : 0,
484
+ anchored: anchorProvider && provider === anchorProvider ? 0 : 1,
485
+ rank: rankFor(provider),
486
+ };
487
+ })
488
+ .sort((a, b) => a.exhausted - b.exhausted || a.anchored - b.anchored || a.rank - b.rank || a.index - b.index)
489
+ .map((entry) => entry.model);
490
+ }
491
+
492
+ /**
493
+ * Build a {@link ModelFallbackPolicy} from crew config + environment. Env vars
494
+ * override config (lowest friction for a quick experiment without editing JSON):
495
+ * PI_CREW_MAX_AUTO_FALLBACKS — cap the auto tail
496
+ * PI_CREW_MODEL_FALLBACK_ORDER — "parentFirst" | "asIs"
497
+ * PI_CREW_MODEL_REQUIRE_CREDENTIALS — "1" to drop uncredentialed providers
498
+ *
499
+ * `parentFirst` ordering applies only when a policy is explicitly configured or
500
+ * quota data enriches it; otherwise the auto tail stays catalogue order.
501
+ *
502
+ * Returns undefined when nothing is configured, so callers that pass it to
503
+ * `buildConfiguredModelRouting` get legacy (unbounded, unordered) behaviour.
504
+ */
505
+ export function resolveModelFallbackPolicy(
506
+ config:
507
+ | {
508
+ maxAutoFallbacks?: number;
509
+ order?: "parentFirst" | "asIs";
510
+ requireCredentials?: boolean;
511
+ quotaAwareOrdering?: boolean;
512
+ }
513
+ | undefined,
514
+ env: NodeJS.ProcessEnv = process.env,
515
+ ): ModelFallbackPolicy | undefined {
516
+ const envMaxAuto = env.PI_CREW_MAX_AUTO_FALLBACKS;
517
+ let maxAutoFallbacks: number | undefined;
518
+ if (envMaxAuto) {
519
+ const parsed = Number.parseInt(envMaxAuto, 10);
520
+ if (Number.isFinite(parsed) && parsed >= 0) {
521
+ maxAutoFallbacks = parsed;
522
+ } else if (Number.isFinite(parsed)) {
523
+ // Negative — clamp to 0 with warning.
524
+ logInternalError(
525
+ "model-fallback.max-auto-fallbacks-invalid",
526
+ undefined,
527
+ `PI_CREW_MAX_AUTO_FALLBACKS="${envMaxAuto}" is negative, clamped to 0`,
528
+ "warn",
529
+ );
530
+ maxAutoFallbacks = 0;
531
+ } else {
532
+ // NaN or non-finite — fall back to config.
533
+ logInternalError(
534
+ "model-fallback.max-auto-fallbacks-invalid",
535
+ undefined,
536
+ `PI_CREW_MAX_AUTO_FALLBACKS="${envMaxAuto}" is invalid, falling back to config`,
537
+ "warn",
538
+ );
539
+ maxAutoFallbacks = config?.maxAutoFallbacks;
540
+ }
541
+ } else {
542
+ maxAutoFallbacks = config?.maxAutoFallbacks;
543
+ }
544
+ const order = (env.PI_CREW_MODEL_FALLBACK_ORDER as "parentFirst" | "asIs" | undefined) ?? config?.order;
545
+ const requireCredentials =
546
+ env.PI_CREW_MODEL_REQUIRE_CREDENTIALS === "1"
547
+ ? true
548
+ : env.PI_CREW_MODEL_REQUIRE_CREDENTIALS === "0"
549
+ ? false
550
+ : config?.requireCredentials;
551
+ // quotaAwareOrdering defaults to true (user preference: default-on with cache).
552
+ // When explicitly disabled, no deprioritization/rank data is attached.
553
+ const quotaAware = config?.quotaAwareOrdering !== false;
554
+ if (maxAutoFallbacks === undefined && !order && requireCredentials === undefined && config?.quotaAwareOrdering === undefined)
555
+ return undefined;
556
+ return {
557
+ ...(maxAutoFallbacks !== undefined && Number.isFinite(maxAutoFallbacks) ? { maxAutoFallbacks } : {}),
558
+ ...(order ? { order } : {}),
559
+ ...(requireCredentials !== undefined ? { requireCredentials } : {}),
560
+ quotaAwareOrdering: quotaAware,
561
+ };
562
+ }
563
+
564
+ /**
565
+ * Resolve the default subagent model from config or env. Precedence:
566
+ * PI_CREW_MODEL env > config.runtime.modelFallback.defaultSubagentModel
567
+ * Returns undefined when neither is set.
568
+ */
569
+ export function resolveDefaultSubagentModel(
570
+ config: { defaultSubagentModel?: string } | undefined,
571
+ env: NodeJS.ProcessEnv = process.env,
572
+ ): string | undefined {
573
+ return env.PI_CREW_MODEL?.trim() || config?.defaultSubagentModel?.trim() || undefined;
574
+ }
575
+
307
576
  export interface ConfiguredModelRouting {
308
577
  requested?: string;
309
578
  candidates: string[];
310
579
  reason?: string;
580
+ /**
581
+ * Set when the caller asked for a model that is not resolvable against the
582
+ * available catalogue, so the chain silently runs something else. Callers
583
+ * surface this as a warning instead of dropping it on the floor.
584
+ */
585
+ droppedRequested?: string;
586
+ /** How many candidates came from the auto tail (diagnostics). */
587
+ autoFallbackCount?: number;
311
588
  /**
312
589
  * F7 scope gate verdict. Populated when the caller passed `scopeModelsPatterns`.
313
590
  * - `inScope: true` → the resolved model is inside the allowlist (or no allowlist).
314
- * - `inScope: false, source: "caller"` → caller override is out-of-scope; the
315
- * function throws `errors.modelOutOfScope` (hard error before spawn) UNLESS
316
- * the caller marked it as a frontmatter override (`isFrontmatterOverride: true`),
317
- * in which case the verdict is returned for the caller to log as a warning.
591
+ * - `inScope: false, source: "caller"` → caller override (override/step/team role)
592
+ * is out-of-scope; the function throws `errors.modelOutOfScope` (hard error
593
+ * before spawn) UNLESS the caller marked it as a frontmatter override
594
+ * (`isFrontmatterOverride: true`), in which case the verdict is returned for
595
+ * the caller to log as a warning.
596
+ * - `inScope: false, source: "frontmatter" | "resolved"` → frontmatter-pinned,
597
+ * defaultSubagentModel, or parentModel-inherited model is out-of-scope;
598
+ * soft warn + run anyway (no throw).
318
599
  */
319
600
  scopeVerdict?: import("./model-scope.ts").ModelScopeCheck;
320
601
  }
@@ -323,11 +604,22 @@ export function buildConfiguredModelRouting(input: {
323
604
  overrideModel?: string;
324
605
  stepModel?: string;
325
606
  teamRoleModel?: string;
607
+ /** Team-role declared fallbacks (`fallbackModels=a,b` on the role line). */
608
+ teamRoleFallbackModels?: string[];
326
609
  agentModel?: string;
610
+ /**
611
+ * Config-level default model for subagents. Sits between the agent model
612
+ * and the inherited parent model in precedence: when the agent has
613
+ * `model: false` (all builtins), this takes over before falling back to
614
+ * the session's model. Ignored when the agent or caller already set one.
615
+ */
616
+ defaultSubagentModel?: string;
327
617
  fallbackModels?: string[];
328
618
  parentModel?: unknown;
329
619
  modelRegistry?: unknown;
330
620
  cwd?: string;
621
+ /** Ordering + budget policy for the auto tail. */
622
+ policy?: ModelFallbackPolicy;
331
623
  /**
332
624
  * F7: when set, enforce the enabledModels allowlist. Caller-supplied out-of-
333
625
  * scope models throw `errors.modelOutOfScope`; frontmatter-pinned out-of-scope
@@ -343,14 +635,20 @@ export function buildConfiguredModelRouting(input: {
343
635
  isFrontmatterOverride?: boolean;
344
636
  }): ConfiguredModelRouting {
345
637
  const registryModels = availableModelInfosFromRegistry(input.modelRegistry);
346
- const configModels = configuredModelInfosFromPiConfig(input.cwd);
638
+ const configModels = configuredModelInfosFromPiConfig(input.cwd, {
639
+ requireCredentials: input.policy?.requireCredentials,
640
+ });
347
641
  const availableModels =
348
642
  registryModels && registryModels.length > 0 ? registryModels : configModels.length > 0 ? configModels : registryModels;
349
643
  const parentModel = modelStringFromUnknown(input.parentModel);
350
- const preferredProvider = parentModel?.split("/")[0] ?? availableModels?.[0]?.provider;
644
+ const preferredProvider = providerOfModelRef(parentModel) ?? availableModels?.[0]?.provider;
351
645
  // B3: Parent model inheritance — when agent has no model specified,
352
646
  // inherit from parent session model before falling back to defaults.
353
- const effectiveAgentModel = input.agentModel?.trim() ? input.agentModel : parentModel;
647
+ // defaultSubagentModel (config/env) sits between agent and parent: when
648
+ // the agent has `model: false` but a default is configured, the default
649
+ // takes over and the parent model becomes the first fallback.
650
+ const defaultSubagent = input.defaultSubagentModel?.trim() || undefined;
651
+ const effectiveAgentModel = input.agentModel?.trim() ? input.agentModel : (defaultSubagent ?? parentModel);
354
652
  const requested = [input.overrideModel, input.stepModel, input.teamRoleModel, effectiveAgentModel].find((model): model is string =>
355
653
  Boolean(model?.trim()),
356
654
  );
@@ -360,52 +658,130 @@ export function buildConfiguredModelRouting(input: {
360
658
  candidates: [],
361
659
  reason: "no configured Pi models available",
362
660
  };
363
- const rawModels = availableModels
364
- ? [
365
- input.overrideModel,
366
- input.stepModel,
367
- input.teamRoleModel,
368
- effectiveAgentModel,
369
- ...(input.fallbackModels ?? []),
370
- ...availableModels.map((model) => model.fullId),
371
- ]
372
- : [input.overrideModel, input.stepModel, input.teamRoleModel, effectiveAgentModel, ...(input.fallbackModels ?? []), parentModel];
661
+ // Explicit declarations, highest precedence first. These are authoritative:
662
+ // never reordered, never truncated by the auto-tail budget.
663
+ const declaredRaw = [
664
+ input.overrideModel,
665
+ input.stepModel,
666
+ input.teamRoleModel,
667
+ effectiveAgentModel,
668
+ // When defaultSubagentModel replaced parentModel as the effective agent
669
+ // model, keep parentModel as an explicit fallback before the auto tail.
670
+ ...(defaultSubagent && !input.agentModel?.trim() ? [parentModel] : []),
671
+ ...(input.teamRoleFallbackModels ?? []),
672
+ ...(input.fallbackModels ?? []),
673
+ ];
373
674
  // Fix (Round 18): when an agent has `model: false` (frontmatter) the
374
675
  // inherited `parentModel` (= session chính's model, e.g. minimax-M3) IS the
375
676
  // desired primary. It must NOT be filtered out by isAvailableModel — which
376
677
  // only knows about models from models.json / registry, NOT builtin Pi models.
377
678
  // Pin the inherited parentModel at index 0 regardless of availability.
378
679
  const parentModelRaw = effectiveAgentModel?.trim() || undefined;
379
- const configuredModels = rawModels
680
+ // When defaultSubagentModel replaced parentModel as the effective agent
681
+ // model, the parent model is still a valid fallback (it IS the session's
682
+ // live model) — pin it too so isAvailableModel doesn't filter it out.
683
+ const parentModelFallback = defaultSubagent && !input.agentModel?.trim() ? parentModel : undefined;
684
+ const declaredModels = declaredRaw
380
685
  .filter((model): model is string => Boolean(model?.trim()))
381
686
  .filter((model, idx) => {
382
687
  if (parentModelRaw && idx === 0 && model.trim() === parentModelRaw) return true;
688
+ if (parentModelFallback && model.trim() === parentModelFallback) return true;
383
689
  return isAvailableModel(model.trim(), availableModels);
384
690
  });
385
- const candidates = buildModelCandidates(configuredModels[0], configuredModels.slice(1), availableModels, preferredProvider);
386
- const reason =
387
- requested && candidates[0] && resolveModelCandidate(requested, availableModels, preferredProvider) !== candidates[0]
388
- ? "requested model unavailable; selected configured Pi fallback"
389
- : candidates.length > 1
390
- ? "configured Pi fallback chain"
391
- : undefined;
691
+ const declaredCandidates = buildModelCandidates(declaredModels[0], declaredModels.slice(1), availableModels, preferredProvider);
692
+ // Auto tail: everything the user did NOT declare. Without a registry the
693
+ // only auto candidate is the inherited parent model.
694
+ const autoRaw = availableModels ? availableModels.map((model) => model.fullId) : parentModel ? [parentModel] : [];
695
+ const declaredSet = new Set(declaredCandidates);
696
+ const autoResolved = buildModelCandidates(undefined, autoRaw, availableModels, preferredProvider).filter(
697
+ (candidate) => !declaredSet.has(candidate),
698
+ );
699
+ const anchorProvider = providerOfModelRef(declaredCandidates[0]) ?? providerOfModelRef(parentModel);
700
+ // Quota-aware ordering: when the policy allows it, enrich with live quota
701
+ // data so exhausted providers sink to the back of the auto tail.
702
+ let effectivePolicy = input.policy;
703
+ if (effectivePolicy?.quotaAwareOrdering !== false && autoResolved.length > 0) {
704
+ const tailProviders = [...new Set(autoResolved.map((m) => providerOfModelRef(m)).filter((p): p is string => Boolean(p)))];
705
+ const deprioritized = quotaDeprioritizedProviders(tailProviders);
706
+ const rank = providerRankFromQuota(tailProviders);
707
+ if (deprioritized.length > 0 || Object.keys(rank).length > 0) {
708
+ effectivePolicy = {
709
+ ...effectivePolicy,
710
+ deprioritizedProviders: [...(effectivePolicy?.deprioritizedProviders ?? []), ...deprioritized],
711
+ providerRank: { ...rank, ...(effectivePolicy?.providerRank ?? {}) },
712
+ };
713
+ }
714
+ }
715
+ const autoOrdered = orderAutoFallbacks(autoResolved, effectivePolicy, anchorProvider);
716
+ const autoCandidates =
717
+ effectivePolicy?.maxAutoFallbacks === undefined ? autoOrdered : autoOrdered.slice(0, Math.max(0, effectivePolicy.maxAutoFallbacks));
718
+ const candidates = [...declaredCandidates, ...autoCandidates];
719
+ const resolvedRequested = requested ? resolveModelCandidate(requested, availableModels, preferredProvider) : undefined;
720
+ const droppedRequested = requested && candidates[0] && resolvedRequested !== candidates[0] ? requested : undefined;
721
+ const reason = droppedRequested
722
+ ? "requested model unavailable; selected configured Pi fallback"
723
+ : candidates.length > 1
724
+ ? "configured Pi fallback chain"
725
+ : undefined;
392
726
  // F7 scope gate: when `scopeModelsPatterns` is configured, check the
393
727
  // resolved model. Caller-supplied (override/step/team role) out-of-scope
394
728
  // is a HARD ERROR (we surface it via the verdict AND throw, so spawn aborts
395
- // before any cost is incurred). Frontmatter-pinned out-of-scope is a
396
- // WARNING returned on the verdict for the caller to log.
729
+ // before any cost is incurred). Frontmatter-pinned, defaultSubagentModel,
730
+ // and parentModel-inherited out-of-scope is a WARNING returned on the verdict
731
+ // for the caller to log.
397
732
  let scopeVerdict: ConfiguredModelRouting["scopeVerdict"];
398
733
  if (input.scopeModelsPatterns && input.scopeModelsPatterns.length > 0) {
399
734
  const resolved = candidates[0] ?? requested;
400
- const source = input.overrideModel ? "caller" : input.agentModel ? "frontmatter" : "resolved";
735
+ // Attribution by REAL precedence: override/step/team role are caller-level
736
+ // (hard-error when out-of-scope); agentModel is frontmatter (soft warn);
737
+ // defaultSubagentModel + parentModel are resolved (soft warn).
738
+ // F5: isFrontmatterOverride means the caller override equals the agent's
739
+ // frontmatter model (author authority) → treat as "frontmatter" so the
740
+ // soft warning surfaces instead of being silent (throw is also skipped).
741
+ const source = input.isFrontmatterOverride
742
+ ? "frontmatter"
743
+ : input.overrideModel
744
+ ? "caller"
745
+ : input.stepModel
746
+ ? "caller"
747
+ : input.teamRoleModel
748
+ ? "caller"
749
+ : input.agentModel?.trim()
750
+ ? "frontmatter"
751
+ : "resolved";
401
752
  scopeVerdict = checkModelScope(resolved, input.scopeModelsPatterns, source);
402
753
  if (!scopeVerdict.inScope && source === "caller" && !input.isFrontmatterOverride) {
403
754
  throw errors.modelOutOfScope(resolved ?? "", input.scopeModelsPatterns);
404
755
  }
405
756
  }
406
- return { requested, candidates, reason, scopeVerdict };
757
+ return {
758
+ requested,
759
+ candidates,
760
+ reason,
761
+ droppedRequested,
762
+ autoFallbackCount: autoCandidates.length,
763
+ scopeVerdict,
764
+ };
407
765
  }
408
766
 
409
767
  export function buildConfiguredModelCandidates(input: Parameters<typeof buildConfiguredModelRouting>[0]): string[] {
410
768
  return buildConfiguredModelRouting(input).candidates;
411
769
  }
770
+
771
+ /**
772
+ * Surface a non-silent warning when a soft-sourced (non-caller) model is
773
+ * out-of-scope and runs anyway. Centralises the `"warn"` severity so call
774
+ * sites cannot accidentally omit it — Round-2 F1 found two sites forgot it,
775
+ * making the warning debug-gated/silent and defeating the entire Sec-M1 fix.
776
+ * Caller-sourced out-of-scope already throws inside `buildConfiguredModelRouting`,
777
+ * so this is a no-op for `source === "caller"`.
778
+ */
779
+ export function warnOutOfScopeSoft(verdict: import("./model-scope.ts").ModelScopeCheck | undefined, scope: string, prefix = "Model"): void {
780
+ if (!verdict || verdict.inScope || verdict.source === "caller") return;
781
+ logInternalError(
782
+ scope,
783
+ undefined,
784
+ `${prefix} "${verdict.model}" from source "${verdict.source}" is outside enabledModels scope: ${verdict.reason ?? "unknown"}. Running anyway (soft warn).`,
785
+ "warn",
786
+ );
787
+ }
@@ -8,6 +8,8 @@
8
8
  * - Frontmatter-pinned (AgentConfig.model) out-of-scope
9
9
  * → WARNING + runs anyway (frontmatter is authoritative; the agent
10
10
  * author made a deliberate choice).
11
+ * - defaultSubagentModel / parentModel-inherited out-of-scope
12
+ * → WARNING + runs anyway (soft warn, same as frontmatter).
11
13
  *
12
14
  * Pattern semantics match pi's `--models` CLI / `enabledModels` allowlist:
13
15
  * - `"anthropic/claude-opus-4-5"` — exact match (case-insensitive).
@@ -45,6 +45,8 @@ export interface BuildPiWorkerArgsInput {
45
45
  env?: NodeJS.ProcessEnv;
46
46
  /** Role for tool restrictions (uses role-tools.ts config) */
47
47
  role?: string;
48
+ /** Per-role thinking override (teamRole.thinking). Takes precedence over agent.thinking. */
49
+ thinkingOverride?: string;
48
50
  }
49
51
 
50
52
  export interface BuildPiWorkerArgsResult {
@@ -265,13 +267,15 @@ export function buildPiWorkerArgs(input: BuildPiWorkerArgsInput): BuildPiWorkerA
265
267
  if (input.sessionEnabled === false) args.push("--no-session");
266
268
 
267
269
  const resolvedModel = input.model ?? input.agent.model;
270
+ // H1.a: teamRole.thinking (passed as thinkingOverride) takes precedence over agent.thinking.
271
+ const effectiveThinking = input.thinkingOverride ?? input.agent.thinking;
268
272
  if (resolvedModel) {
269
- const modelWithThinking = applyThinkingSuffix(resolvedModel, input.agent.thinking);
273
+ const modelWithThinking = applyThinkingSuffix(resolvedModel, effectiveThinking);
270
274
  if (modelWithThinking) args.push("--model", modelWithThinking);
271
275
  }
272
276
  // When no model resolved, pass thinking separately so Pi can apply it to the inherited parent model.
273
- if (!resolvedModel && input.agent.thinking && input.agent.thinking !== "off" && isValidThinkingLevel(input.agent.thinking)) {
274
- args.push("--thinking", input.agent.thinking);
277
+ if (!resolvedModel && effectiveThinking && effectiveThinking !== "off" && isValidThinkingLevel(effectiveThinking)) {
278
+ args.push("--thinking", effectiveThinking);
275
279
  }
276
280
 
277
281
  // Apply role-based tool restrictions (from role-tools.ts)