@centerforagenticai/pi-multi-account 0.1.2 → 0.1.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,5 +1,5 @@
1
1
  {
2
- "version": "0.1.2",
2
+ "version": "0.1.5",
3
3
  "description": "Global Anthropic and OpenAI Codex multi-account OAuth routing for Pi.",
4
4
  "type": "module",
5
5
  "bin": {
@@ -221,10 +221,18 @@ export function streamAnthropicOAuth(
221
221
  // which throws under fine-grained-tool-streaming (input may be invalid
222
222
  // mid-flight) and aborts the turn. The raw stream yields the same
223
223
  // RawMessageStreamEvents; tool args are already parsed leniently below.
224
+ // A finite non-negative integer request-local maxRetries bounds the SDK's
225
+ // own retries (0 sends exactly once); any other value keeps the SDK default.
226
+ const maxRetries = options?.maxRetries;
224
227
  const { data: anthropicStream, response: httpResponse } =
225
228
  await client.messages
226
229
  .create(params, {
227
230
  signal: options?.signal,
231
+ ...(Number.isFinite(maxRetries) &&
232
+ Number.isInteger(maxRetries) &&
233
+ (maxRetries ?? -1) >= 0
234
+ ? { maxRetries }
235
+ : {}),
228
236
  })
229
237
  .withResponse();
230
238
 
@@ -24,8 +24,9 @@
24
24
  * Provenance: pi-anthropic-oauth@0.2.4-intel.2, private pin
25
25
  * 99ac00f290efbcb93e7f23e6b7482d8727367aa7, upstream baseline
26
26
  * 53266ecb51b6d1890ef3f7251a64cb1d71d96099, source src/stream.ts.
27
- * Local delta: adaptive thinking emits type=adaptive and mapped output effort,
28
- * without budget_tokens; all other request and response behavior is unchanged.
27
+ * Local delta: adaptive thinking emits type=adaptive and mapped output effort
28
+ * without budget_tokens; request-local retry/timeout options reach the SDK; and
29
+ * refusal or unknown stop reasons retain bounded, structured error details.
29
30
  */
30
31
 
31
32
  import { Anthropic } from "@anthropic-ai/sdk";
@@ -41,6 +42,8 @@ import {
41
42
  type SimpleStreamOptions,
42
43
  type StopReason,
43
44
  } from "@earendil-works/pi-ai";
45
+ import { sanitizeDiagnosticText } from "./diagnostics.js";
46
+ import type { ProviderErrorCode } from "./error-classification.js";
44
47
  type IndexedBlock =
45
48
  | ({ type: "text"; text: string } & { index: number })
46
49
  | ({ type: "thinking"; thinking: string; thinkingSignature?: string } & {
@@ -95,18 +98,51 @@ const REQUIRED_BETAS = [
95
98
  "interleaved-thinking-2025-05-14",
96
99
  ] as const;
97
100
 
98
- function mapStopReason(reason: string | null | undefined): StopReason {
101
+ type StopReasonResult = Readonly<{
102
+ stopReason: StopReason;
103
+ errorMessage?: string;
104
+ code?: Extract<ProviderErrorCode, "refusal" | "unknown_stop">;
105
+ }>;
106
+
107
+ function mapStopReason(
108
+ reason: string | null | undefined,
109
+ stopDetails?: unknown,
110
+ ): StopReasonResult {
99
111
  switch (reason) {
100
112
  case "end_turn":
101
113
  case "pause_turn":
102
114
  case "stop_sequence":
103
- return "stop";
115
+ return { stopReason: "stop" };
104
116
  case "max_tokens":
105
- return "length";
117
+ return { stopReason: "length" };
106
118
  case "tool_use":
107
- return "toolUse";
108
- default:
109
- return "error";
119
+ return { stopReason: "toolUse" };
120
+ case "refusal": {
121
+ const explanation =
122
+ typeof stopDetails === "object" &&
123
+ stopDetails !== null &&
124
+ typeof (stopDetails as { explanation?: unknown }).explanation === "string"
125
+ ? sanitizeDiagnosticText(
126
+ (stopDetails as { explanation: string }).explanation,
127
+ )
128
+ : "";
129
+ return {
130
+ stopReason: "error",
131
+ errorMessage:
132
+ explanation || "The model refused to complete the request",
133
+ code: "refusal",
134
+ };
135
+ }
136
+ default: {
137
+ const boundedReason = sanitizeDiagnosticText(reason ?? "unknown");
138
+ return {
139
+ stopReason: "error",
140
+ errorMessage: sanitizeDiagnosticText(
141
+ `Provider stopped with: ${boundedReason}`,
142
+ ),
143
+ code: "unknown_stop",
144
+ };
145
+ }
110
146
  }
111
147
  }
112
148
 
@@ -163,7 +199,7 @@ export function streamAnthropicAdaptive(
163
199
  const stream = createAssistantMessageEventStream();
164
200
 
165
201
  void (async () => {
166
- const output: AssistantMessage = {
202
+ const output: AssistantMessage & { code?: ProviderErrorCode } = {
167
203
  role: "assistant",
168
204
  content: [],
169
205
  api: model.api,
@@ -276,10 +312,24 @@ export function streamAnthropicAdaptive(
276
312
  // which throws under fine-grained-tool-streaming (input may be invalid
277
313
  // mid-flight) and aborts the turn. The raw stream yields the same
278
314
  // RawMessageStreamEvents; tool args are already parsed leniently below.
315
+ const maxRetries = options?.maxRetries;
316
+ const timeoutMs = options?.timeoutMs;
279
317
  const { data: anthropicStream, response: httpResponse } =
280
318
  await client.messages
281
319
  .create(params, {
282
320
  signal: options?.signal,
321
+ ...(Number.isFinite(maxRetries) &&
322
+ Number.isInteger(maxRetries) &&
323
+ (maxRetries ?? -1) >= 0
324
+ ? { maxRetries }
325
+ : {}),
326
+ // Any finite positive timeout is honored; the SDK timer takes whole
327
+ // milliseconds, so a fractional value is floored to at least 1.
328
+ ...(typeof timeoutMs === "number" &&
329
+ Number.isFinite(timeoutMs) &&
330
+ timeoutMs > 0
331
+ ? { timeout: Math.max(1, Math.floor(timeoutMs)) }
332
+ : {}),
283
333
  })
284
334
  .withResponse();
285
335
 
@@ -474,7 +524,19 @@ export function streamAnthropicAdaptive(
474
524
  }
475
525
 
476
526
  if (event.type === "message_delta") {
477
- output.stopReason = mapStopReason(event.delta.stop_reason);
527
+ const rawStopReason = event.delta.stop_reason;
528
+ const mapped = mapStopReason(
529
+ rawStopReason,
530
+ (event.delta as { stop_details?: unknown }).stop_details,
531
+ );
532
+ output.stopReason = mapped.stopReason;
533
+ if (typeof rawStopReason === "string") {
534
+ output.rawStopReason = sanitizeDiagnosticText(rawStopReason);
535
+ }
536
+ if (mapped.errorMessage !== undefined) {
537
+ output.errorMessage = mapped.errorMessage;
538
+ }
539
+ if (mapped.code !== undefined) output.code = mapped.code;
478
540
  output.usage.input =
479
541
  (event.usage as { input_tokens?: number }).input_tokens ||
480
542
  output.usage.input;
@@ -505,11 +567,15 @@ export function streamAnthropicAdaptive(
505
567
  }
506
568
 
507
569
  if (options?.signal?.aborted) throw new Error("Request aborted");
508
- stream.push({
509
- type: "done",
510
- reason: output.stopReason as "stop" | "length" | "toolUse",
511
- message: output,
512
- });
570
+ if (output.stopReason === "error") {
571
+ stream.push({ type: "error", reason: "error", error: output });
572
+ } else {
573
+ stream.push({
574
+ type: "done",
575
+ reason: output.stopReason as "stop" | "length" | "toolUse",
576
+ message: output,
577
+ });
578
+ }
513
579
  stream.end();
514
580
  } catch (error) {
515
581
  for (const block of output.content as Array<{
@@ -13,6 +13,7 @@ import {
13
13
  sanitizeDiagnosticText,
14
14
  sanitizeHeaderValue,
15
15
  } from "./diagnostics.js";
16
+ import { hostFinalStopMessage } from "./host-final-stop-message.js";
16
17
 
17
18
  export const ANTHROPIC_ALIAS_API = "hypha-anthropic-oauth" as const;
18
19
 
@@ -53,12 +54,19 @@ function withAliasAttribution(
53
54
  };
54
55
  }
55
56
 
57
+ /**
58
+ * Bounds an upstream error and, for a structured refusal or unknown stop,
59
+ * publishes the shared host-final message. A direct alias turn reaches the
60
+ * host's retry and compaction predicates without the unified provider, so
61
+ * provider-authored stop wording must not make the host resend the request.
62
+ */
56
63
  function sanitizeUpstreamError(message: AssistantMessage): AssistantMessage {
57
64
  if (message.errorMessage === undefined) return message;
58
- return {
65
+ const sanitized: AssistantMessage = {
59
66
  ...message,
60
67
  errorMessage: sanitizeDiagnosticText(message.errorMessage),
61
68
  };
69
+ return { ...sanitized, ...hostFinalStopMessage(sanitized) };
62
70
  }
63
71
 
64
72
  function withAliasEvent(
@@ -193,14 +193,81 @@ function withAliasEvent(
193
193
  }
194
194
  }
195
195
 
196
+ /**
197
+ * One attributed, sanitized error terminal for an upstream that failed before
198
+ * producing a stream (a synchronous throw, a rejected setup, or a throwing
199
+ * iterator). The shape matches the host's own setup-error terminal: no content,
200
+ * zero usage, no diagnostics. Only the bounded, redacted error text survives.
201
+ * A failure after the caller's signal fired is a cancellation: it ends as
202
+ * `aborted`, matching the maintained stream, so it never cools the account.
203
+ */
204
+ function aliasSetupErrorMessage(
205
+ error: unknown,
206
+ aliasModel: Model<Api>,
207
+ aborted: boolean,
208
+ ): AssistantMessage & { stopReason: "error" | "aborted" } {
209
+ let detail: string;
210
+ try {
211
+ detail = error instanceof Error ? error.message : String(error);
212
+ } catch {
213
+ detail = "Codex alias stream setup failed";
214
+ }
215
+ return {
216
+ role: "assistant",
217
+ content: [],
218
+ api: aliasModel.api,
219
+ provider: aliasModel.provider,
220
+ model: aliasModel.id,
221
+ usage: {
222
+ input: 0,
223
+ output: 0,
224
+ cacheRead: 0,
225
+ cacheWrite: 0,
226
+ totalTokens: 0,
227
+ cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
228
+ },
229
+ stopReason: aborted ? "aborted" : "error",
230
+ errorMessage: sanitizeDiagnosticText(detail),
231
+ timestamp: Date.now(),
232
+ };
233
+ }
234
+
235
+ function isTerminalEvent(event: AssistantMessageEvent): boolean {
236
+ return event.type === "done" || event.type === "error";
237
+ }
238
+
239
+ /**
240
+ * Forwards upstream events with alias attribution. Every failure mode ends in
241
+ * exactly one terminal: a rejected setup or a throwing iterator before any
242
+ * terminal becomes one attributed error event, and the attributed stream always
243
+ * ends, so no failure escapes as an unhandled rejection or a hang.
244
+ */
196
245
  function reattributeStream(
197
- upstream: AssistantMessageEventStream,
246
+ upstream: AssistantMessageEventStream | PromiseLike<AssistantMessageEventStream>,
198
247
  aliasModel: Model<Api>,
248
+ signal: AbortSignal | undefined,
199
249
  ): AssistantMessageEventStream {
200
250
  const attributed = createAssistantMessageEventStream();
201
251
  void (async () => {
202
- for await (const event of upstream) {
203
- attributed.push(withAliasEvent(event, aliasModel));
252
+ let sawTerminal = false;
253
+ try {
254
+ for await (const event of await upstream) {
255
+ if (sawTerminal) continue;
256
+ if (isTerminalEvent(event)) sawTerminal = true;
257
+ attributed.push(withAliasEvent(event, aliasModel));
258
+ }
259
+ } catch (error) {
260
+ if (!sawTerminal) {
261
+ sawTerminal = true;
262
+ const message = aliasSetupErrorMessage(
263
+ error,
264
+ aliasModel,
265
+ signal?.aborted === true,
266
+ );
267
+ attributed.push({ type: "error", reason: message.stopReason, error: message });
268
+ }
269
+ } finally {
270
+ attributed.end();
204
271
  }
205
272
  })();
206
273
  return attributed;
@@ -235,9 +302,33 @@ function normalizeAliasContext(
235
302
  }
236
303
 
237
304
  /**
238
- * Wraps the exact maintained stream without replacing transport or OAuth. Pi's
305
+ * Temporary containment for the Pi 0.99 Codex WebSocket failure: a WebSocket
306
+ * error leaves the session in a state where the next Codex turn crashes with
307
+ * "Cannot read properties of undefined (reading 'length')". Routes this
308
+ * extension owns always request SSE. The base `openai-codex` provider is never
309
+ * touched; only options on calls we already route are adjusted. Remove once the
310
+ * WebSocket path is fixed upstream.
311
+ */
312
+ export const CODEX_FORCED_TRANSPORT = "sse" as const;
313
+
314
+ /** Returns options with Codex transport pinned to SSE, and whether a value was overridden. */
315
+ export function forceCodexSseOptions<T extends SimpleStreamOptions>(
316
+ options: T | undefined,
317
+ ): { options: T; overridden: boolean } {
318
+ if (options?.transport === CODEX_FORCED_TRANSPORT) {
319
+ return { options, overridden: false };
320
+ }
321
+ return {
322
+ options: { ...(options ?? {}), transport: CODEX_FORCED_TRANSPORT } as T,
323
+ overridden: true,
324
+ };
325
+ }
326
+
327
+ /**
328
+ * Wraps the exact maintained stream without replacing OAuth. Pi's
239
329
  * alias-resolved apiKey and every other option field are forwarded unchanged;
240
- * only model/context identity and callback attribution are adapted.
330
+ * model/context identity and callback attribution are adapted, and transport
331
+ * is pinned to SSE (see {@link CODEX_FORCED_TRANSPORT}).
241
332
  */
242
333
  export function createCodexAliasStream(
243
334
  maintainedStream: NonNullable<ProviderConfig["streamSimple"]>,
@@ -251,12 +342,12 @@ export function createCodexAliasStream(
251
342
  };
252
343
  const upstreamContext = normalizeAliasContext(context, aliasModel);
253
344
 
254
- let upstreamOptions = options;
345
+ let upstreamOptions = forceCodexSseOptions(options).options;
255
346
  if (options?.onPayload || options?.onResponse) {
256
347
  const aliasOnPayload = options.onPayload;
257
348
  const aliasOnResponse = options.onResponse;
258
349
  upstreamOptions = {
259
- ...options,
350
+ ...upstreamOptions,
260
351
  ...(aliasOnPayload
261
352
  ? {
262
353
  onPayload: (payload: unknown) =>
@@ -272,10 +363,17 @@ export function createCodexAliasStream(
272
363
  };
273
364
  }
274
365
 
275
- return reattributeStream(
276
- upstream(upstreamModel, upstreamContext, upstreamOptions),
277
- aliasModel,
278
- );
366
+ // Defensive only under the live host: its `lazyApi` stream catches a setup
367
+ // throw first and returns its own setup-error terminal. A non-lazy upstream
368
+ // (plain Node) can still throw synchronously; convert that into the same
369
+ // single attributed, sanitized terminal as any other failure (UPSTREAM.md).
370
+ let upstreamStream: ReturnType<CodexUpstreamStream>;
371
+ try {
372
+ upstreamStream = upstream(upstreamModel, upstreamContext, upstreamOptions);
373
+ } catch (error) {
374
+ upstreamStream = Promise.reject(error) as unknown as ReturnType<CodexUpstreamStream>;
375
+ }
376
+ return reattributeStream(upstreamStream, aliasModel, options?.signal);
279
377
  };
280
378
  return aliasStream as unknown as NonNullable<ProviderConfig["streamSimple"]>;
281
379
  }
package/src/config.ts CHANGED
@@ -29,6 +29,7 @@ import {
29
29
  } from "node:fs";
30
30
  import { randomUUID } from "node:crypto";
31
31
  import { basename, dirname, isAbsolute, join } from "node:path";
32
+ import { sanitizeDiagnosticText } from "./diagnostics.js";
32
33
  import { PROJECT_KEY_PATTERN } from "./project-identity.js";
33
34
  import {
34
35
  AccountRateHistoryError,
@@ -167,6 +168,15 @@ export interface CrossFamilyChain {
167
168
  readonly to: AllowedFamily;
168
169
  }
169
170
 
171
+ /** One exact, directional model-substitution egress authorization. */
172
+ export interface ModelFallbackEgressAuthorization {
173
+ readonly sourceModelId: string;
174
+ readonly destinationModelId: string;
175
+ }
176
+
177
+ /** Exact source unified model id -> ordered exact fallback model ids. */
178
+ export type ModelFallbackMap = Readonly<Record<string, readonly string[]>>;
179
+
170
180
  export interface MultiAccountConfig {
171
181
  readonly accountLimit: number;
172
182
  readonly sameFamilyFailover: boolean;
@@ -238,6 +248,10 @@ export interface MultiAccountConfig {
238
248
  */
239
249
  readonly preferredModels: Readonly<Record<string, readonly string[]>>;
240
250
  readonly tierModelMap: TierModelMap;
251
+ /** Explicit ordered fallback policy. Absent or empty disables model substitution. */
252
+ readonly modelFallbacks?: ModelFallbackMap;
253
+ /** Directional authorization required in addition to policy for cross-vendor edges. */
254
+ readonly modelFallbackEgress?: readonly ModelFallbackEgressAuthorization[];
241
255
  /**
242
256
  * How close to expiry a credential may get before routing prefers a fresher
243
257
  * same-family account, in milliseconds. Pre-emption avoids spending a turn to
@@ -284,6 +298,8 @@ export const DEFAULT_CONFIG: MultiAccountConfig = {
284
298
  accountRateHistory: {},
285
299
  preferredModels: {},
286
300
  tierModelMap: Object.freeze({}),
301
+ modelFallbacks: Object.freeze({}),
302
+ modelFallbackEgress: Object.freeze([]),
287
303
  // Comfortably longer than a turn, short enough that accounts are not retired
288
304
  // while they still have useful life.
289
305
  preemptiveExpiryWindowMs: 120_000,
@@ -309,6 +325,8 @@ const CONFIG_KEYS = new Set<keyof MultiAccountConfig>([
309
325
  "accountRateHistory",
310
326
  "preferredModels",
311
327
  "tierModelMap",
328
+ "modelFallbacks",
329
+ "modelFallbackEgress",
312
330
  "preemptiveExpiryWindowMs",
313
331
  "usageFetchEnabled",
314
332
  ]);
@@ -689,6 +707,135 @@ function isValidTierModelId(value: unknown): value is string {
689
707
  );
690
708
  }
691
709
 
710
+ export const MAX_MODEL_FALLBACK_SOURCES = 128;
711
+ export const MAX_MODEL_FALLBACK_DESTINATIONS = 16;
712
+ export const MAX_MODEL_FALLBACK_EDGES = 256;
713
+ /**
714
+ * Exact catalog ids: managed wire ids, OpenRouter `vendor/model[:variant]`
715
+ * slugs, and the `~vendor/...` and `@cf/...` forms in the pinned catalog.
716
+ */
717
+ const MODEL_FALLBACK_ID_PATTERN = /^[A-Za-z0-9~@][A-Za-z0-9._:/@~-]{0,255}$/;
718
+
719
+ /**
720
+ * A fallback id may later appear in routing diagnostics, so it is accepted only
721
+ * when the shared diagnostic sanitizer would leave it byte-for-byte unchanged.
722
+ * Anything the sanitizer would redact anywhere in the string (credential
723
+ * prefixes, JWTs, token/canary shapes, long opaque runs) is rejected here, and
724
+ * the rejection message never echoes the value.
725
+ */
726
+ function isValidModelFallbackId(value: unknown): value is string {
727
+ return (
728
+ typeof value === "string" &&
729
+ MODEL_FALLBACK_ID_PATTERN.test(value) &&
730
+ sanitizeDiagnosticText(value) === value
731
+ );
732
+ }
733
+
734
+ function invalidModelFallbackId(field: string): ConfigValidationError {
735
+ return new ConfigValidationError(
736
+ `${field} must use an exact, non-credential model id of at most 256 characters.`,
737
+ );
738
+ }
739
+
740
+ function parseModelFallbacks(value: unknown): ModelFallbackMap {
741
+ if (value === undefined) return DEFAULT_CONFIG.modelFallbacks ?? Object.freeze({});
742
+ if (!isPlainObject(value)) {
743
+ throw new ConfigValidationError("modelFallbacks must be a plain JSON object.");
744
+ }
745
+ const entries = Object.entries(value);
746
+ if (entries.length > MAX_MODEL_FALLBACK_SOURCES) {
747
+ throw new ConfigValidationError(
748
+ `modelFallbacks must contain at most ${MAX_MODEL_FALLBACK_SOURCES} sources.`,
749
+ );
750
+ }
751
+ const parsed = Object.create(null) as Record<string, readonly string[]>;
752
+ for (const [sourceModelId, rawDestinations] of entries) {
753
+ if (!isValidModelFallbackId(sourceModelId)) {
754
+ throw invalidModelFallbackId("modelFallbacks source ids");
755
+ }
756
+ if (!Array.isArray(rawDestinations) || rawDestinations.length === 0) {
757
+ throw new ConfigValidationError(
758
+ "modelFallbacks values must be non-empty arrays of model ids.",
759
+ );
760
+ }
761
+ if (rawDestinations.length > MAX_MODEL_FALLBACK_DESTINATIONS) {
762
+ throw new ConfigValidationError(
763
+ `modelFallbacks may contain at most ${MAX_MODEL_FALLBACK_DESTINATIONS} destinations per source.`,
764
+ );
765
+ }
766
+ const destinations: string[] = [];
767
+ const seen = new Set<string>();
768
+ for (const destinationModelId of rawDestinations) {
769
+ if (!isValidModelFallbackId(destinationModelId)) {
770
+ throw invalidModelFallbackId("modelFallbacks destination ids");
771
+ }
772
+ if (destinationModelId === sourceModelId) {
773
+ throw new ConfigValidationError(
774
+ "modelFallbacks cannot list the source model as its own destination.",
775
+ );
776
+ }
777
+ if (seen.has(destinationModelId)) {
778
+ throw new ConfigValidationError(
779
+ "modelFallbacks cannot contain duplicate destinations for one source.",
780
+ );
781
+ }
782
+ seen.add(destinationModelId);
783
+ destinations.push(destinationModelId);
784
+ }
785
+ parsed[sourceModelId] = Object.freeze(destinations);
786
+ }
787
+ return Object.freeze(parsed);
788
+ }
789
+
790
+ function parseModelFallbackEgress(
791
+ value: unknown,
792
+ ): readonly ModelFallbackEgressAuthorization[] {
793
+ if (value === undefined) return DEFAULT_CONFIG.modelFallbackEgress ?? Object.freeze([]);
794
+ if (!Array.isArray(value)) {
795
+ throw new ConfigValidationError("modelFallbackEgress must be an array.");
796
+ }
797
+ if (value.length > MAX_MODEL_FALLBACK_EDGES) {
798
+ throw new ConfigValidationError(
799
+ `modelFallbackEgress must contain at most ${MAX_MODEL_FALLBACK_EDGES} edges.`,
800
+ );
801
+ }
802
+ const parsed: ModelFallbackEgressAuthorization[] = [];
803
+ const seen = new Set<string>();
804
+ for (const candidate of value) {
805
+ if (!isPlainObject(candidate)) {
806
+ throw new ConfigValidationError("modelFallbackEgress entries must be plain objects.");
807
+ }
808
+ const unknownKeys = Object.keys(candidate).filter(
809
+ (key) => key !== "sourceModelId" && key !== "destinationModelId",
810
+ );
811
+ if (unknownKeys.length > 0) {
812
+ throw new ConfigValidationError(
813
+ "modelFallbackEgress entries contain an unsupported field.",
814
+ );
815
+ }
816
+ const sourceModelId = candidate["sourceModelId"];
817
+ const destinationModelId = candidate["destinationModelId"];
818
+ if (!isValidModelFallbackId(sourceModelId)) {
819
+ throw invalidModelFallbackId("modelFallbackEgress sourceModelId");
820
+ }
821
+ if (!isValidModelFallbackId(destinationModelId)) {
822
+ throw invalidModelFallbackId("modelFallbackEgress destinationModelId");
823
+ }
824
+ if (sourceModelId === destinationModelId) {
825
+ throw new ConfigValidationError(
826
+ "modelFallbackEgress cannot authorize a model to itself.",
827
+ );
828
+ }
829
+ const key = `${sourceModelId}\u0000${destinationModelId}`;
830
+ if (seen.has(key)) {
831
+ throw new ConfigValidationError("modelFallbackEgress cannot contain duplicate edges.");
832
+ }
833
+ seen.add(key);
834
+ parsed.push(Object.freeze({ sourceModelId, destinationModelId }));
835
+ }
836
+ return Object.freeze(parsed);
837
+ }
838
+
692
839
  export function parseTierModelMap(value: unknown): TierModelMap {
693
840
  if (value === undefined) return DEFAULT_CONFIG.tierModelMap;
694
841
  if (!isRecord(value)) {
@@ -1119,6 +1266,8 @@ export function parseConfig(value: unknown): MultiAccountConfig {
1119
1266
  const accountRateHistory = parseAccountRateHistory(value["accountRateHistory"]);
1120
1267
  const preferredModels = parsePreferredModels(value["preferredModels"]);
1121
1268
  const tierModelMap = parseTierModelMap(value["tierModelMap"]);
1269
+ const modelFallbacks = parseModelFallbacks(value["modelFallbacks"]);
1270
+ const modelFallbackEgress = parseModelFallbackEgress(value["modelFallbackEgress"]);
1122
1271
  const preemptiveExpiryWindowMs =
1123
1272
  value["preemptiveExpiryWindowMs"] ??
1124
1273
  DEFAULT_CONFIG.preemptiveExpiryWindowMs;
@@ -1206,6 +1355,8 @@ export function parseConfig(value: unknown): MultiAccountConfig {
1206
1355
  accountRateHistory,
1207
1356
  preferredModels,
1208
1357
  tierModelMap,
1358
+ modelFallbacks,
1359
+ modelFallbackEgress,
1209
1360
  preemptiveExpiryWindowMs,
1210
1361
  usageFetchEnabled,
1211
1362
  };
@@ -13,6 +13,8 @@ export const PROVIDER_ERROR_CODES = [
13
13
  "model_not_found",
14
14
  "unsupported_api_version",
15
15
  "invalid_request",
16
+ "refusal",
17
+ "unknown_stop",
16
18
  ] as const;
17
19
  export type ProviderErrorCode = (typeof PROVIDER_ERROR_CODES)[number];
18
20
 
@@ -50,10 +52,12 @@ export type FailureCategory =
50
52
 
51
53
  export interface FailureClassification {
52
54
  readonly category: FailureCategory;
55
+ readonly kind?: "refusal" | "unknown-stop";
53
56
  readonly accountAction:
54
57
  | "cooldown-and-route"
55
58
  | "invalidate-and-route"
56
- | "route-without-retry";
59
+ | "route-without-retry"
60
+ | "retain-account";
57
61
  readonly cooldownReason?: CooldownReason;
58
62
  readonly serverHint?: NumericServerHint;
59
63
  }
@@ -248,6 +252,14 @@ export function classifyFailure(
248
252
  return { category: "config", accountAction: "route-without-retry" };
249
253
  }
250
254
 
255
+ if (signal.code === "refusal" || signal.code === "unknown_stop") {
256
+ return {
257
+ category: "unknown",
258
+ kind: signal.code === "refusal" ? "refusal" : "unknown-stop",
259
+ accountAction: "retain-account",
260
+ };
261
+ }
262
+
251
263
  if (isTransportFailure(signal)) {
252
264
  return cooldown("transport", "transport", signal);
253
265
  }