@centerforagenticai/pi-multi-account 0.1.4 → 0.1.6

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (35) hide show
  1. package/README.md +17 -3
  2. package/package.json +5 -5
  3. package/packages/pi-anthropic-oauth/package.json +2 -2
  4. package/packages/pi-anthropic-oauth/src/stream.ts +26 -8
  5. package/packages/pi-anthropic-oauth/src/transport-activity.ts +59 -0
  6. package/packages/pi-antigravity/package.json +2 -2
  7. package/packages/pi-antigravity/src/models/discovery.ts +2 -1
  8. package/packages/pi-antigravity/src/models/grouping.ts +12 -10
  9. package/packages/pi-antigravity/src/models/models.ts +11 -4
  10. package/src/account-group-failure.ts +149 -0
  11. package/src/account-group-members.ts +139 -0
  12. package/src/anthropic-adaptive-stream.ts +18 -9
  13. package/src/anthropic-alias-stream.ts +13 -68
  14. package/src/codex-adapter.ts +78 -50
  15. package/src/commands.ts +3 -3
  16. package/src/config.ts +197 -2
  17. package/src/diagnostic-store.ts +17 -15
  18. package/src/diagnostics.ts +97 -35
  19. package/src/host-final-stop-message.ts +20 -118
  20. package/src/index.ts +452 -87
  21. package/src/logical-dispatch.ts +12 -13
  22. package/src/logical-provider.ts +1282 -355
  23. package/src/model-fallback-policy.ts +384 -0
  24. package/src/models-declaration.ts +15 -8
  25. package/src/public-assistant-projection.ts +163 -0
  26. package/src/recovery-engine.ts +648 -110
  27. package/src/recovery-plan.ts +7 -1
  28. package/src/recovery-send-evidence.ts +29 -0
  29. package/src/refusal-advice.ts +139 -0
  30. package/src/routing.ts +4 -8
  31. package/src/runtime-state.ts +7 -0
  32. package/src/shared-usage.ts +21 -4
  33. package/src/upstream-anthropic.ts +12 -4
  34. package/src/upstream-antigravity.ts +2 -38
  35. package/src/usage-fetch.ts +42 -59
@@ -35,6 +35,7 @@ import {
35
35
  type Api,
36
36
  type AssistantMessage,
37
37
  type AssistantMessageEventStream,
38
+ type JsonObject,
38
39
  calculateCost,
39
40
  type Context,
40
41
  createAssistantMessageEventStream,
@@ -53,7 +54,7 @@ type IndexedBlock =
53
54
  type: "toolCall";
54
55
  id: string;
55
56
  name: string;
56
- arguments: Record<string, unknown>;
57
+ arguments: JsonObject;
57
58
  partialJson: string;
58
59
  } & { index: number });
59
60
  type UpstreamHelpers = {
@@ -67,15 +68,20 @@ type UpstreamHelpers = {
67
68
  convertPiToolsToAnthropic: (tools: NonNullable<Context["tools"]>, isOAuth: boolean) => unknown;
68
69
  fromClaudeCodeToolName: (name: string, tools?: NonNullable<Context["tools"]>) => string;
69
70
  buildAnthropicSystemPrompt: (systemPrompt: string | undefined, isOAuth: boolean) => unknown;
71
+ transportActivityListener: (options: unknown) => (() => void) | undefined;
72
+ createTransportActivityFetch: (onActivity: () => void) => typeof fetch;
70
73
  };
71
74
 
72
75
  const upstreamAuthPath = "../packages/pi-anthropic-oauth/src/auth.ts";
73
76
  const upstreamConvertPath = "../packages/pi-anthropic-oauth/src/convert.ts";
74
77
  const upstreamPromptPath = "../packages/pi-anthropic-oauth/src/prompt.ts";
78
+ const upstreamTransportActivityPath =
79
+ "../packages/pi-anthropic-oauth/src/transport-activity.ts";
75
80
  const upstreamHelpers = {
76
81
  ...(await import(upstreamAuthPath)),
77
82
  ...(await import(upstreamConvertPath)),
78
83
  ...(await import(upstreamPromptPath)),
84
+ ...(await import(upstreamTransportActivityPath)),
79
85
  } as unknown as UpstreamHelpers;
80
86
  const {
81
87
  isClaudeOAuthAccessToken,
@@ -84,6 +90,8 @@ const {
84
90
  convertPiToolsToAnthropic,
85
91
  fromClaudeCodeToolName,
86
92
  buildAnthropicSystemPrompt,
93
+ transportActivityListener,
94
+ createTransportActivityFetch,
87
95
  } = upstreamHelpers;
88
96
 
89
97
  const REQUIRED_BETAS = [
@@ -230,12 +238,19 @@ export function streamAnthropicAdaptive(
230
238
 
231
239
  if (isOAuth) defaultHeaders.authorization = `Bearer ${apiKey}`;
232
240
 
241
+ // A caller that supplies onTransportActivity hears about every response
242
+ // body chunk, including the pings the SDK drops; otherwise the SDK keeps
243
+ // its default fetch.
244
+ const onTransportActivity = transportActivityListener(options);
233
245
  const client = new Anthropic({
234
246
  baseURL: model.baseUrl,
235
247
  apiKey: isOAuth ? null : apiKey,
236
248
  authToken: isOAuth ? apiKey : null,
237
249
  defaultHeaders,
238
250
  dangerouslyAllowBrowser: true,
251
+ ...(onTransportActivity === undefined
252
+ ? {}
253
+ : { fetch: createTransportActivityFetch(onTransportActivity) }),
239
254
  });
240
255
 
241
256
  const maxTokens =
@@ -468,10 +483,7 @@ export function streamAnthropicAdaptive(
468
483
  ) {
469
484
  block.partialJson += event.delta.partial_json;
470
485
  try {
471
- block.arguments = JSON.parse(block.partialJson) as Record<
472
- string,
473
- unknown
474
- >;
486
+ block.arguments = JSON.parse(block.partialJson) as JsonObject;
475
487
  } catch {}
476
488
  stream.push({
477
489
  type: "toolcall_delta",
@@ -507,10 +519,7 @@ export function streamAnthropicAdaptive(
507
519
  });
508
520
  } else if (block.type === "toolCall") {
509
521
  try {
510
- block.arguments = JSON.parse(block.partialJson) as Record<
511
- string,
512
- unknown
513
- >;
522
+ block.arguments = JSON.parse(block.partialJson) as JsonObject;
514
523
  } catch {}
515
524
  delete (block as { partialJson?: string }).partialJson;
516
525
  stream.push({
@@ -1,23 +1,26 @@
1
1
  import {
2
2
  createAssistantMessageEventStream,
3
3
  type Api,
4
- type AssistantMessage,
5
- type AssistantMessageEvent,
6
4
  type AssistantMessageEventStream,
7
- type Context,
8
5
  type Model,
6
+ type Context,
7
+ type TranscriptContext,
9
8
  type ProviderResponse,
10
9
  type SimpleStreamOptions,
11
10
  } from "@earendil-works/pi-ai";
12
- import {
13
- sanitizeDiagnosticText,
14
- sanitizeHeaderValue,
15
- } from "./diagnostics.js";
16
- import { hostFinalStopMessage } from "./host-final-stop-message.js";
11
+ import { projectResponseHeaders } from "./diagnostics.js";
12
+ import { projectAliasAssistantEvent } from "./public-assistant-projection.js";
17
13
 
18
14
  export const ANTHROPIC_ALIAS_API = "hypha-anthropic-oauth" as const;
19
15
 
20
16
  export type AnthropicUpstreamStream = (
17
+ model: Model<Api>,
18
+ context: Context | TranscriptContext,
19
+ options?: SimpleStreamOptions,
20
+ ) => AssistantMessageEventStream;
21
+
22
+ /** Legacy Pi 0.84 context-shaped stream consumed by the exact-provenance adaptive adapter. */
23
+ export type AnthropicLegacyStream = (
21
24
  model: Model<Api>,
22
25
  context: Context,
23
26
  options?: SimpleStreamOptions,
@@ -33,68 +36,10 @@ export function sanitizeAnthropicProviderResponse(
33
36
  ): ProviderResponse {
34
37
  return {
35
38
  status: response.status,
36
- headers: Object.fromEntries(
37
- Object.entries(response.headers).map(([name, value]) => [
38
- sanitizeDiagnosticText(name),
39
- sanitizeHeaderValue(name, value),
40
- ]),
41
- ),
42
- };
43
- }
44
-
45
- function withAliasAttribution(
46
- message: AssistantMessage,
47
- aliasModel: Model<Api>,
48
- ): AssistantMessage {
49
- return {
50
- ...message,
51
- api: aliasModel.api,
52
- provider: aliasModel.provider,
53
- model: aliasModel.id,
39
+ headers: projectResponseHeaders(response.headers),
54
40
  };
55
41
  }
56
42
 
57
- /**
58
- * Bounds an upstream error and, for a structured refusal or unknown stop,
59
- * publishes the shared host-final message. A direct alias turn reaches the
60
- * host's retry and compaction predicates without the unified provider, so
61
- * provider-authored stop wording must not make the host resend the request.
62
- */
63
- function sanitizeUpstreamError(message: AssistantMessage): AssistantMessage {
64
- if (message.errorMessage === undefined) return message;
65
- const sanitized: AssistantMessage = {
66
- ...message,
67
- errorMessage: sanitizeDiagnosticText(message.errorMessage),
68
- };
69
- return { ...sanitized, ...hostFinalStopMessage(sanitized) };
70
- }
71
-
72
- function withAliasEvent(
73
- event: AssistantMessageEvent,
74
- aliasModel: Model<Api>,
75
- ): AssistantMessageEvent {
76
- switch (event.type) {
77
- case "done":
78
- return {
79
- ...event,
80
- message: withAliasAttribution(event.message, aliasModel),
81
- };
82
- case "error":
83
- return {
84
- ...event,
85
- error: withAliasAttribution(
86
- sanitizeUpstreamError(event.error),
87
- aliasModel,
88
- ),
89
- };
90
- default:
91
- return {
92
- ...event,
93
- partial: withAliasAttribution(event.partial, aliasModel),
94
- };
95
- }
96
- }
97
-
98
43
  function reattributeStream(
99
44
  upstream: AssistantMessageEventStream,
100
45
  aliasModel: Model<Api>,
@@ -103,7 +48,7 @@ function reattributeStream(
103
48
 
104
49
  void (async () => {
105
50
  for await (const event of upstream) {
106
- attributed.push(withAliasEvent(event, aliasModel));
51
+ attributed.push(projectAliasAssistantEvent(event, aliasModel));
107
52
  }
108
53
  })();
109
54
 
@@ -28,9 +28,11 @@ import {
28
28
  import { cloneProviderModelCatalog } from "./catalog-rebinding.js";
29
29
  import {
30
30
  sanitizeDiagnosticText,
31
- sanitizeHeaderValue,
31
+ projectResponseHeaders,
32
32
  } from "./diagnostics.js";
33
33
 
34
+ import { projectAliasAssistantEvent } from "./public-assistant-projection.js";
35
+
34
36
  const CODEX_BASE_API = "openai-codex-responses" as const;
35
37
  /**
36
38
  * Keep alias models on Pi's host-known API id. The provider-scoped
@@ -141,66 +143,85 @@ type CodexUpstreamStream = (
141
143
  options?: SimpleStreamOptions,
142
144
  ) => AssistantMessageEventStream;
143
145
 
144
- function withAliasAttribution(
145
- message: AssistantMessage,
146
+ function sanitizeProviderResponse(response: ProviderResponse): ProviderResponse {
147
+ return { status: response.status, headers: projectResponseHeaders(response.headers) };
148
+ }
149
+
150
+ /**
151
+ * One attributed, sanitized error terminal for an upstream that failed before
152
+ * producing a stream (a synchronous throw, a rejected setup, or a throwing
153
+ * iterator). The shape matches the host's own setup-error terminal: no content,
154
+ * zero usage, no diagnostics. Only the bounded, redacted error text survives.
155
+ * A failure after the caller's signal fired is a cancellation: it ends as
156
+ * `aborted`, matching the maintained stream, so it never cools the account.
157
+ */
158
+ function aliasSetupErrorMessage(
159
+ error: unknown,
146
160
  aliasModel: Model<Api>,
147
- ): AssistantMessage {
161
+ aborted: boolean,
162
+ ): AssistantMessage & { stopReason: "error" | "aborted" } {
163
+ let detail: string;
164
+ try {
165
+ detail = error instanceof Error ? error.message : String(error);
166
+ } catch {
167
+ detail = "Codex alias stream setup failed";
168
+ }
148
169
  return {
149
- ...message,
170
+ role: "assistant",
171
+ content: [],
150
172
  api: aliasModel.api,
151
173
  provider: aliasModel.provider,
152
174
  model: aliasModel.id,
175
+ usage: {
176
+ input: 0,
177
+ output: 0,
178
+ cacheRead: 0,
179
+ cacheWrite: 0,
180
+ totalTokens: 0,
181
+ cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
182
+ },
183
+ stopReason: aborted ? "aborted" : "error",
184
+ errorMessage: sanitizeDiagnosticText(detail),
185
+ timestamp: Date.now(),
153
186
  };
154
187
  }
155
188
 
156
- function sanitizeUpstreamError(message: AssistantMessage): AssistantMessage {
157
- if (message.errorMessage === undefined) return message;
158
- return {
159
- ...message,
160
- errorMessage: sanitizeDiagnosticText(message.errorMessage),
161
- };
162
- }
163
-
164
- function sanitizeProviderResponse(response: ProviderResponse): ProviderResponse {
165
- return {
166
- status: response.status,
167
- headers: Object.fromEntries(
168
- Object.entries(response.headers).map(([name, value]) => [
169
- sanitizeDiagnosticText(name),
170
- sanitizeHeaderValue(name, value),
171
- ]),
172
- ),
173
- };
174
- }
175
-
176
- function withAliasEvent(
177
- event: AssistantMessageEvent,
178
- aliasModel: Model<Api>,
179
- ): AssistantMessageEvent {
180
- switch (event.type) {
181
- case "done":
182
- return { ...event, message: withAliasAttribution(event.message, aliasModel) };
183
- case "error":
184
- return {
185
- ...event,
186
- error: withAliasAttribution(
187
- sanitizeUpstreamError(event.error),
188
- aliasModel,
189
- ),
190
- };
191
- default:
192
- return { ...event, partial: withAliasAttribution(event.partial, aliasModel) };
193
- }
189
+ function isTerminalEvent(event: AssistantMessageEvent): boolean {
190
+ return event.type === "done" || event.type === "error";
194
191
  }
195
192
 
193
+ /**
194
+ * Forwards upstream events with alias attribution. Every failure mode ends in
195
+ * exactly one terminal: a rejected setup or a throwing iterator before any
196
+ * terminal becomes one attributed error event, and the attributed stream always
197
+ * ends, so no failure escapes as an unhandled rejection or a hang.
198
+ */
196
199
  function reattributeStream(
197
- upstream: AssistantMessageEventStream,
200
+ upstream: AssistantMessageEventStream | PromiseLike<AssistantMessageEventStream>,
198
201
  aliasModel: Model<Api>,
202
+ signal: AbortSignal | undefined,
199
203
  ): AssistantMessageEventStream {
200
204
  const attributed = createAssistantMessageEventStream();
201
205
  void (async () => {
202
- for await (const event of upstream) {
203
- attributed.push(withAliasEvent(event, aliasModel));
206
+ let sawTerminal = false;
207
+ try {
208
+ for await (const event of await upstream) {
209
+ if (sawTerminal) continue;
210
+ if (isTerminalEvent(event)) sawTerminal = true;
211
+ attributed.push(projectAliasAssistantEvent(event, aliasModel));
212
+ }
213
+ } catch (error) {
214
+ if (!sawTerminal) {
215
+ sawTerminal = true;
216
+ const message = aliasSetupErrorMessage(
217
+ error,
218
+ aliasModel,
219
+ signal?.aborted === true,
220
+ );
221
+ attributed.push({ type: "error", reason: message.stopReason, error: message });
222
+ }
223
+ } finally {
224
+ attributed.end();
204
225
  }
205
226
  })();
206
227
  return attributed;
@@ -296,10 +317,17 @@ export function createCodexAliasStream(
296
317
  };
297
318
  }
298
319
 
299
- return reattributeStream(
300
- upstream(upstreamModel, upstreamContext, upstreamOptions),
301
- aliasModel,
302
- );
320
+ // Defensive only under the live host: its `lazyApi` stream catches a setup
321
+ // throw first and returns its own setup-error terminal. A non-lazy upstream
322
+ // (plain Node) can still throw synchronously; convert that into the same
323
+ // single attributed, sanitized terminal as any other failure (UPSTREAM.md).
324
+ let upstreamStream: ReturnType<CodexUpstreamStream>;
325
+ try {
326
+ upstreamStream = upstream(upstreamModel, upstreamContext, upstreamOptions);
327
+ } catch (error) {
328
+ upstreamStream = Promise.reject(error) as unknown as ReturnType<CodexUpstreamStream>;
329
+ }
330
+ return reattributeStream(upstreamStream, aliasModel, options?.signal);
303
331
  };
304
332
  return aliasStream as unknown as NonNullable<ProviderConfig["streamSimple"]>;
305
333
  }
package/src/commands.ts CHANGED
@@ -14,6 +14,7 @@ import {
14
14
  import { basename, dirname, join } from "node:path";
15
15
  import { randomUUID } from "node:crypto";
16
16
  import type { Api, Model } from "@earendil-works/pi-ai";
17
+ import type { AccountGroupMemberAvailability } from "./account-group-members.js";
17
18
  import { acquireMachineLease } from "./machine-lease.js";
18
19
  import {
19
20
  DECLARATION_BASE_URL,
@@ -290,9 +291,8 @@ export function declarationNoticeMessage(notice: DeclarationNotice): string {
290
291
  : `LOGICAL ROUTING OFF: The managed model declaration is unreadable. Run ${notice.remedy}.`;
291
292
  }
292
293
 
293
- export interface AccountGroupCommandMemberStatus {
294
- readonly providerId: string;
295
- readonly eligible: boolean;
294
+ export interface AccountGroupCommandMemberStatus extends Pick<AccountGroupMemberAvailability, "providerId" | "eligible"> {
295
+ /** Membership availability or an additional managed-routing/metered block. */
296
296
  readonly reason: string;
297
297
  }
298
298
 
package/src/config.ts CHANGED
@@ -29,6 +29,8 @@ import {
29
29
  } from "node:fs";
30
30
  import { randomUUID } from "node:crypto";
31
31
  import { basename, dirname, isAbsolute, join } from "node:path";
32
+ import { sanitizeDiagnosticText } from "./diagnostics.js";
33
+ import { isAccountGroupMemberReference } from "./account-group-members.js";
32
34
  import { PROJECT_KEY_PATTERN } from "./project-identity.js";
33
35
  import {
34
36
  AccountRateHistoryError,
@@ -167,6 +169,18 @@ export interface CrossFamilyChain {
167
169
  readonly to: AllowedFamily;
168
170
  }
169
171
 
172
+ /** One exact, directional model-substitution egress authorization. */
173
+ export interface ModelFallbackEgressAuthorization {
174
+ readonly sourceModelId: string;
175
+ readonly destinationModelId: string;
176
+ }
177
+
178
+ /** Exact source unified model id -> ordered exact fallback model ids. */
179
+ export type ModelFallbackMap = Readonly<Record<string, readonly string[]>>;
180
+
181
+ /** Upper bound for `recoveryStallTimeoutMs`: one stalled attempt never waits longer. */
182
+ export const MAX_RECOVERY_STALL_TIMEOUT_MS = 30 * 60_000;
183
+
170
184
  export interface MultiAccountConfig {
171
185
  readonly accountLimit: number;
172
186
  readonly sameFamilyFailover: boolean;
@@ -178,6 +192,14 @@ export interface MultiAccountConfig {
178
192
  readonly recoveryIdleTimeoutMs: number;
179
193
  /** Total elapsed time allowed for one complete recovery invocation. */
180
194
  readonly recoveryAbsoluteTimeoutMs: number;
195
+ /**
196
+ * Longest wait for the next event of one unified physical attempt, opening
197
+ * included. A stall before any content ends that attempt as a pre-start
198
+ * transient failure that may recover once on another account; a stall after
199
+ * content ends the call with no retry. Must be less than
200
+ * `recoveryIdleTimeoutMs`, so the stall fires before the invocation idles out.
201
+ */
202
+ readonly recoveryStallTimeoutMs: number;
181
203
  /**
182
204
  * Operator-chosen display labels keyed by canonical provider id, so managed
183
205
  * accounts are distinguishable in Pi's login list and the status view.
@@ -238,6 +260,10 @@ export interface MultiAccountConfig {
238
260
  */
239
261
  readonly preferredModels: Readonly<Record<string, readonly string[]>>;
240
262
  readonly tierModelMap: TierModelMap;
263
+ /** Explicit ordered fallback policy. Absent or empty disables model substitution. */
264
+ readonly modelFallbacks?: ModelFallbackMap;
265
+ /** Directional authorization required in addition to policy for cross-vendor edges. */
266
+ readonly modelFallbackEgress?: readonly ModelFallbackEgressAuthorization[];
241
267
  /**
242
268
  * How close to expiry a credential may get before routing prefers a fresher
243
269
  * same-family account, in milliseconds. Pre-emption avoids spending a turn to
@@ -275,6 +301,7 @@ export const DEFAULT_CONFIG: MultiAccountConfig = {
275
301
  cooldownMaxMs: 300_000,
276
302
  recoveryIdleTimeoutMs: 5 * 60_000,
277
303
  recoveryAbsoluteTimeoutMs: 30 * 60_000,
304
+ recoveryStallTimeoutMs: 3 * 60_000,
278
305
  accountLabels: {},
279
306
  projectLabels: {},
280
307
  accountGroups: {},
@@ -284,6 +311,8 @@ export const DEFAULT_CONFIG: MultiAccountConfig = {
284
311
  accountRateHistory: {},
285
312
  preferredModels: {},
286
313
  tierModelMap: Object.freeze({}),
314
+ modelFallbacks: Object.freeze({}),
315
+ modelFallbackEgress: Object.freeze([]),
287
316
  // Comfortably longer than a turn, short enough that accounts are not retired
288
317
  // while they still have useful life.
289
318
  preemptiveExpiryWindowMs: 120_000,
@@ -299,6 +328,7 @@ const CONFIG_KEYS = new Set<keyof MultiAccountConfig>([
299
328
  "cooldownMaxMs",
300
329
  "recoveryIdleTimeoutMs",
301
330
  "recoveryAbsoluteTimeoutMs",
331
+ "recoveryStallTimeoutMs",
302
332
  "accountLabels",
303
333
  "projectLabels",
304
334
  "accountGroups",
@@ -309,6 +339,8 @@ const CONFIG_KEYS = new Set<keyof MultiAccountConfig>([
309
339
  "accountRateHistory",
310
340
  "preferredModels",
311
341
  "tierModelMap",
342
+ "modelFallbacks",
343
+ "modelFallbackEgress",
312
344
  "preemptiveExpiryWindowMs",
313
345
  "usageFetchEnabled",
314
346
  ]);
@@ -689,6 +721,135 @@ function isValidTierModelId(value: unknown): value is string {
689
721
  );
690
722
  }
691
723
 
724
+ export const MAX_MODEL_FALLBACK_SOURCES = 128;
725
+ export const MAX_MODEL_FALLBACK_DESTINATIONS = 16;
726
+ export const MAX_MODEL_FALLBACK_EDGES = 256;
727
+ /**
728
+ * Exact catalog ids: managed wire ids, OpenRouter `vendor/model[:variant]`
729
+ * slugs, and the `~vendor/...` and `@cf/...` forms in the pinned catalog.
730
+ */
731
+ const MODEL_FALLBACK_ID_PATTERN = /^[A-Za-z0-9~@][A-Za-z0-9._:/@~-]{0,255}$/;
732
+
733
+ /**
734
+ * A fallback id may later appear in routing diagnostics, so it is accepted only
735
+ * when the shared diagnostic sanitizer would leave it byte-for-byte unchanged.
736
+ * Anything the sanitizer would redact anywhere in the string (credential
737
+ * prefixes, JWTs, token/canary shapes, long opaque runs) is rejected here, and
738
+ * the rejection message never echoes the value.
739
+ */
740
+ function isValidModelFallbackId(value: unknown): value is string {
741
+ return (
742
+ typeof value === "string" &&
743
+ MODEL_FALLBACK_ID_PATTERN.test(value) &&
744
+ sanitizeDiagnosticText(value) === value
745
+ );
746
+ }
747
+
748
+ function invalidModelFallbackId(field: string): ConfigValidationError {
749
+ return new ConfigValidationError(
750
+ `${field} must use an exact, non-credential model id of at most 256 characters.`,
751
+ );
752
+ }
753
+
754
+ function parseModelFallbacks(value: unknown): ModelFallbackMap {
755
+ if (value === undefined) return DEFAULT_CONFIG.modelFallbacks ?? Object.freeze({});
756
+ if (!isPlainObject(value)) {
757
+ throw new ConfigValidationError("modelFallbacks must be a plain JSON object.");
758
+ }
759
+ const entries = Object.entries(value);
760
+ if (entries.length > MAX_MODEL_FALLBACK_SOURCES) {
761
+ throw new ConfigValidationError(
762
+ `modelFallbacks must contain at most ${MAX_MODEL_FALLBACK_SOURCES} sources.`,
763
+ );
764
+ }
765
+ const parsed = Object.create(null) as Record<string, readonly string[]>;
766
+ for (const [sourceModelId, rawDestinations] of entries) {
767
+ if (!isValidModelFallbackId(sourceModelId)) {
768
+ throw invalidModelFallbackId("modelFallbacks source ids");
769
+ }
770
+ if (!Array.isArray(rawDestinations) || rawDestinations.length === 0) {
771
+ throw new ConfigValidationError(
772
+ "modelFallbacks values must be non-empty arrays of model ids.",
773
+ );
774
+ }
775
+ if (rawDestinations.length > MAX_MODEL_FALLBACK_DESTINATIONS) {
776
+ throw new ConfigValidationError(
777
+ `modelFallbacks may contain at most ${MAX_MODEL_FALLBACK_DESTINATIONS} destinations per source.`,
778
+ );
779
+ }
780
+ const destinations: string[] = [];
781
+ const seen = new Set<string>();
782
+ for (const destinationModelId of rawDestinations) {
783
+ if (!isValidModelFallbackId(destinationModelId)) {
784
+ throw invalidModelFallbackId("modelFallbacks destination ids");
785
+ }
786
+ if (destinationModelId === sourceModelId) {
787
+ throw new ConfigValidationError(
788
+ "modelFallbacks cannot list the source model as its own destination.",
789
+ );
790
+ }
791
+ if (seen.has(destinationModelId)) {
792
+ throw new ConfigValidationError(
793
+ "modelFallbacks cannot contain duplicate destinations for one source.",
794
+ );
795
+ }
796
+ seen.add(destinationModelId);
797
+ destinations.push(destinationModelId);
798
+ }
799
+ parsed[sourceModelId] = Object.freeze(destinations);
800
+ }
801
+ return Object.freeze(parsed);
802
+ }
803
+
804
+ function parseModelFallbackEgress(
805
+ value: unknown,
806
+ ): readonly ModelFallbackEgressAuthorization[] {
807
+ if (value === undefined) return DEFAULT_CONFIG.modelFallbackEgress ?? Object.freeze([]);
808
+ if (!Array.isArray(value)) {
809
+ throw new ConfigValidationError("modelFallbackEgress must be an array.");
810
+ }
811
+ if (value.length > MAX_MODEL_FALLBACK_EDGES) {
812
+ throw new ConfigValidationError(
813
+ `modelFallbackEgress must contain at most ${MAX_MODEL_FALLBACK_EDGES} edges.`,
814
+ );
815
+ }
816
+ const parsed: ModelFallbackEgressAuthorization[] = [];
817
+ const seen = new Set<string>();
818
+ for (const candidate of value) {
819
+ if (!isPlainObject(candidate)) {
820
+ throw new ConfigValidationError("modelFallbackEgress entries must be plain objects.");
821
+ }
822
+ const unknownKeys = Object.keys(candidate).filter(
823
+ (key) => key !== "sourceModelId" && key !== "destinationModelId",
824
+ );
825
+ if (unknownKeys.length > 0) {
826
+ throw new ConfigValidationError(
827
+ "modelFallbackEgress entries contain an unsupported field.",
828
+ );
829
+ }
830
+ const sourceModelId = candidate["sourceModelId"];
831
+ const destinationModelId = candidate["destinationModelId"];
832
+ if (!isValidModelFallbackId(sourceModelId)) {
833
+ throw invalidModelFallbackId("modelFallbackEgress sourceModelId");
834
+ }
835
+ if (!isValidModelFallbackId(destinationModelId)) {
836
+ throw invalidModelFallbackId("modelFallbackEgress destinationModelId");
837
+ }
838
+ if (sourceModelId === destinationModelId) {
839
+ throw new ConfigValidationError(
840
+ "modelFallbackEgress cannot authorize a model to itself.",
841
+ );
842
+ }
843
+ const key = `${sourceModelId}\u0000${destinationModelId}`;
844
+ if (seen.has(key)) {
845
+ throw new ConfigValidationError("modelFallbackEgress cannot contain duplicate edges.");
846
+ }
847
+ seen.add(key);
848
+ parsed.push(Object.freeze({ sourceModelId, destinationModelId }));
849
+ }
850
+ return Object.freeze(parsed);
851
+ }
852
+
692
853
  export function parseTierModelMap(value: unknown): TierModelMap {
693
854
  if (value === undefined) return DEFAULT_CONFIG.tierModelMap;
694
855
  if (!isRecord(value)) {
@@ -814,10 +975,10 @@ function parseAccountGroups(
814
975
  const providerIds = members.map((member, index) => {
815
976
  if (
816
977
  typeof member !== "string" ||
817
- !isCanonicalSubscriptionAccountId(member, accountLimit)
978
+ !isAccountGroupMemberReference(member, accountLimit, MANAGED_FAMILIES)
818
979
  ) {
819
980
  throw new ConfigValidationError(
820
- `accountGroups.${groupId}[${index}] must be a canonical managed subscription provider id within accountLimit.`,
981
+ `accountGroups.${groupId}[${index}] must be a safe provider reference with canonical managed slots within accountLimit.`,
821
982
  );
822
983
  }
823
984
  return member;
@@ -1119,6 +1280,8 @@ export function parseConfig(value: unknown): MultiAccountConfig {
1119
1280
  const accountRateHistory = parseAccountRateHistory(value["accountRateHistory"]);
1120
1281
  const preferredModels = parsePreferredModels(value["preferredModels"]);
1121
1282
  const tierModelMap = parseTierModelMap(value["tierModelMap"]);
1283
+ const modelFallbacks = parseModelFallbacks(value["modelFallbacks"]);
1284
+ const modelFallbackEgress = parseModelFallbackEgress(value["modelFallbackEgress"]);
1122
1285
  const preemptiveExpiryWindowMs =
1123
1286
  value["preemptiveExpiryWindowMs"] ??
1124
1287
  DEFAULT_CONFIG.preemptiveExpiryWindowMs;
@@ -1178,6 +1341,35 @@ export function parseConfig(value: unknown): MultiAccountConfig {
1178
1341
  );
1179
1342
  }
1180
1343
  }
1344
+ // An omitted stall limit defaults below the effective idle limit, so a
1345
+ // config that sets only a short `recoveryIdleTimeoutMs` stays valid and its
1346
+ // stall can still fire first.
1347
+ const recoveryStallTimeoutMs =
1348
+ value["recoveryStallTimeoutMs"] ??
1349
+ Math.min(
1350
+ DEFAULT_CONFIG.recoveryStallTimeoutMs,
1351
+ (recoveryIdleTimeoutMs as number) - 1,
1352
+ );
1353
+ if (
1354
+ typeof recoveryStallTimeoutMs !== "number" ||
1355
+ !Number.isFinite(recoveryStallTimeoutMs) ||
1356
+ recoveryStallTimeoutMs < 1_000 ||
1357
+ recoveryStallTimeoutMs > MAX_RECOVERY_STALL_TIMEOUT_MS
1358
+ ) {
1359
+ throw new ConfigValidationError(
1360
+ value["recoveryStallTimeoutMs"] === undefined
1361
+ ? "recoveryIdleTimeoutMs must be more than 1000 ms, so the 1000 ms minimum recoveryStallTimeoutMs can end first."
1362
+ : `recoveryStallTimeoutMs must be a finite number from 1000 through ${MAX_RECOVERY_STALL_TIMEOUT_MS} ms.`,
1363
+ );
1364
+ }
1365
+ // The engine's idle timer covers the whole invocation. A stall limit that is
1366
+ // not shorter would let it abort the call before a stalled attempt could
1367
+ // move to another account.
1368
+ if (recoveryStallTimeoutMs >= (recoveryIdleTimeoutMs as number)) {
1369
+ throw new ConfigValidationError(
1370
+ "recoveryStallTimeoutMs must be less than recoveryIdleTimeoutMs.",
1371
+ );
1372
+ }
1181
1373
  if (
1182
1374
  typeof preemptiveExpiryWindowMs !== "number" ||
1183
1375
  !Number.isFinite(preemptiveExpiryWindowMs) ||
@@ -1197,6 +1389,7 @@ export function parseConfig(value: unknown): MultiAccountConfig {
1197
1389
  cooldownMaxMs,
1198
1390
  recoveryIdleTimeoutMs: recoveryIdleTimeoutMs as number,
1199
1391
  recoveryAbsoluteTimeoutMs: recoveryAbsoluteTimeoutMs as number,
1392
+ recoveryStallTimeoutMs,
1200
1393
  accountLabels,
1201
1394
  projectLabels,
1202
1395
  accountGroups,
@@ -1206,6 +1399,8 @@ export function parseConfig(value: unknown): MultiAccountConfig {
1206
1399
  accountRateHistory,
1207
1400
  preferredModels,
1208
1401
  tierModelMap,
1402
+ modelFallbacks,
1403
+ modelFallbackEgress,
1209
1404
  preemptiveExpiryWindowMs,
1210
1405
  usageFetchEnabled,
1211
1406
  };