@centerforagenticai/pi-multi-account 0.1.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (124) hide show
  1. package/LICENSE +21 -0
  2. package/NOTICE +29 -0
  3. package/README.md +999 -0
  4. package/config/models/pi-multi-account.v1.json +32 -0
  5. package/config/subscription-plans.v1.json +122 -0
  6. package/package.json +76 -0
  7. package/packages/pi-anthropic-oauth/LICENSE +21 -0
  8. package/packages/pi-anthropic-oauth/package.json +54 -0
  9. package/packages/pi-anthropic-oauth/src/auth.ts +396 -0
  10. package/packages/pi-anthropic-oauth/src/context.ts +116 -0
  11. package/packages/pi-anthropic-oauth/src/convert.ts +303 -0
  12. package/packages/pi-anthropic-oauth/src/index.ts +37 -0
  13. package/packages/pi-anthropic-oauth/src/prompt.ts +137 -0
  14. package/packages/pi-anthropic-oauth/src/stream.ts +476 -0
  15. package/packages/pi-antigravity/LICENSE +21 -0
  16. package/packages/pi-antigravity/package.json +77 -0
  17. package/packages/pi-antigravity/src/auth/index.ts +14 -0
  18. package/packages/pi-antigravity/src/auth/oauth.ts +442 -0
  19. package/packages/pi-antigravity/src/client/client.ts +561 -0
  20. package/packages/pi-antigravity/src/client/index.ts +1 -0
  21. package/packages/pi-antigravity/src/context.ts +110 -0
  22. package/packages/pi-antigravity/src/diagnostics/diagnostics.ts +96 -0
  23. package/packages/pi-antigravity/src/diagnostics/index.ts +1 -0
  24. package/packages/pi-antigravity/src/image/image.ts +336 -0
  25. package/packages/pi-antigravity/src/image/index.ts +1 -0
  26. package/packages/pi-antigravity/src/index.ts +280 -0
  27. package/packages/pi-antigravity/src/models/discovery.ts +154 -0
  28. package/packages/pi-antigravity/src/models/grouping.ts +424 -0
  29. package/packages/pi-antigravity/src/models/index.ts +3 -0
  30. package/packages/pi-antigravity/src/models/models.ts +500 -0
  31. package/packages/pi-antigravity/src/stream/index.ts +1 -0
  32. package/packages/pi-antigravity/src/stream/stream.ts +1478 -0
  33. package/packages/pi-antigravity/src/types/enums.ts +42 -0
  34. package/packages/pi-antigravity/src/types/index.ts +2 -0
  35. package/packages/pi-antigravity/src/types/types.ts +292 -0
  36. package/packages/pi-antigravity/src/usage/index.ts +1 -0
  37. package/packages/pi-antigravity/src/usage/usage.ts +416 -0
  38. package/packages/pi-antigravity/src/utils/http.ts +91 -0
  39. package/packages/pi-antigravity/src/utils/index.ts +3 -0
  40. package/packages/pi-antigravity/src/utils/security.ts +73 -0
  41. package/packages/pi-antigravity/src/utils/util.ts +132 -0
  42. package/scripts/multi-account.mjs +44 -0
  43. package/src/account-labels.ts +223 -0
  44. package/src/account-plan-assignment.ts +340 -0
  45. package/src/account-rate-history.ts +372 -0
  46. package/src/anthropic-adaptive-stream.ts +531 -0
  47. package/src/anthropic-alias-stream.ts +140 -0
  48. package/src/anthropic-context-compat.ts +80 -0
  49. package/src/api-pricing.ts +579 -0
  50. package/src/bounded-file-lines.ts +97 -0
  51. package/src/catalog-rebinding.ts +177 -0
  52. package/src/catalog-registration-probe.ts +111 -0
  53. package/src/codex-adapter.ts +345 -0
  54. package/src/codex-model-defaults.ts +785 -0
  55. package/src/command-completions.ts +404 -0
  56. package/src/commands.ts +2000 -0
  57. package/src/compaction.ts +14 -0
  58. package/src/config.ts +1317 -0
  59. package/src/continuation.ts +569 -0
  60. package/src/cooldowns.ts +110 -0
  61. package/src/cost-digest-store.ts +332 -0
  62. package/src/cost-digest.ts +1044 -0
  63. package/src/cost-history.ts +251 -0
  64. package/src/cost-period-closer.ts +160 -0
  65. package/src/cost-report-json.ts +318 -0
  66. package/src/cost-report-reader.ts +368 -0
  67. package/src/cost-report-render.ts +207 -0
  68. package/src/cost-report.ts +1104 -0
  69. package/src/coverage-attestation.ts +397 -0
  70. package/src/credential-lifecycle.ts +169 -0
  71. package/src/credential-refresh.ts +248 -0
  72. package/src/declaration-notice-marker.ts +238 -0
  73. package/src/diagnostic-store.ts +276 -0
  74. package/src/diagnostics.ts +309 -0
  75. package/src/discovery.ts +471 -0
  76. package/src/duration.ts +13 -0
  77. package/src/error-classification.ts +256 -0
  78. package/src/fuzzy.ts +15 -0
  79. package/src/group-policy.ts +81 -0
  80. package/src/history-store.ts +897 -0
  81. package/src/index.ts +5572 -0
  82. package/src/lifecycle.ts +378 -0
  83. package/src/logical-dispatch.ts +279 -0
  84. package/src/logical-model-selector.ts +254 -0
  85. package/src/logical-model-switcher.ts +430 -0
  86. package/src/logical-provider-attribution.ts +544 -0
  87. package/src/logical-provider.ts +1237 -0
  88. package/src/logical-route-indicator.ts +215 -0
  89. package/src/machine-lease.ts +445 -0
  90. package/src/model-support.ts +66 -0
  91. package/src/models-declaration.ts +1091 -0
  92. package/src/openai-adapter.ts +117 -0
  93. package/src/openrouter-budget.ts +304 -0
  94. package/src/openrouter-fallback.ts +146 -0
  95. package/src/period-boundaries.ts +376 -0
  96. package/src/pi-anthropic-oauth.d.ts +6 -0
  97. package/src/preflight.ts +253 -0
  98. package/src/pricing-cache.ts +235 -0
  99. package/src/project-identity.ts +100 -0
  100. package/src/provider-registration.ts +942 -0
  101. package/src/rate-formula.ts +163 -0
  102. package/src/recovery-engine.ts +853 -0
  103. package/src/recovery-output.ts +837 -0
  104. package/src/recovery-plan.ts +239 -0
  105. package/src/report-range.ts +203 -0
  106. package/src/route-resolver.ts +789 -0
  107. package/src/routing-config-transaction.ts +232 -0
  108. package/src/routing.ts +1163 -0
  109. package/src/runtime-state.ts +630 -0
  110. package/src/session-account-groups.ts +284 -0
  111. package/src/session-restore.ts +287 -0
  112. package/src/shared-usage.ts +1392 -0
  113. package/src/standalone-cli.ts +720 -0
  114. package/src/status-view.ts +578 -0
  115. package/src/subscription-plan-catalog.ts +346 -0
  116. package/src/tier-model-resolver.ts +46 -0
  117. package/src/upstream-anthropic.ts +315 -0
  118. package/src/upstream-antigravity.ts +327 -0
  119. package/src/usage-fetch.ts +1634 -0
  120. package/src/usage.ts +1026 -0
  121. package/src/vendor.ts +87 -0
  122. package/src/warmer.ts +231 -0
  123. package/src/watchdog.ts +219 -0
  124. package/src/window-history.ts +270 -0
@@ -0,0 +1,1237 @@
1
+ /**
2
+ * The `unified` logical provider.
3
+ *
4
+ * One virtual provider whose models are the exact wire model ids the managed
5
+ * physical accounts serve. A request to a logical model is dispatched to a
6
+ * physical account that actually serves that exact id, so the model the
7
+ * operator picked is the model that runs.
8
+ *
9
+ * Construction is deliberately separate from registration. This module builds
10
+ * the provider and knows nothing about the host; `registerLogicalProvider` in
11
+ * `provider-registration.ts` puts it in front of one. That split is what lets a
12
+ * request be driven with no host, and a registration be inspected with no
13
+ * request.
14
+ */
15
+
16
+ import { logicalAccountEligible, recordFailureCooldown } from "./routing.js";
17
+ import type { ManagedAccount } from "./routing.js";
18
+ import {
19
+ isRetryableAssistantError,
20
+ type AssistantMessage,
21
+ type ProviderResponse,
22
+ type SimpleStreamOptions,
23
+ } from "@earendil-works/pi-ai";
24
+ import { RuntimeState, type LogicalRoutePin } from "./runtime-state.js";
25
+ import { DEFAULT_CONFIG } from "./config.js";
26
+ import type {
27
+ AllowedFamily,
28
+ CrossFamilyChain,
29
+ ManagedFamily,
30
+ MultiAccountConfig,
31
+ } from "./config.js";
32
+ import {
33
+ PROVIDER_ERROR_CODES,
34
+ TRANSPORT_FAILURE_KINDS,
35
+ classifyFailure,
36
+ providerErrorCodeFromMessage,
37
+ } from "./error-classification.js";
38
+ import type {
39
+ ProviderErrorCode,
40
+ ProviderFailureSignal,
41
+ TransportFailureKind,
42
+ } from "./error-classification.js";
43
+ import { resolveTierModel } from "./tier-model-resolver.js";
44
+ import type { TierModelMap } from "./tier-model-resolver.js";
45
+ import { providerTypeFor, tierRank, vendorForFamily } from "./vendor.js";
46
+ import type { ProviderType, Vendor } from "./vendor.js";
47
+
48
+ export { LOGICAL_PROVIDER_ID } from "./models-declaration.js";
49
+ import { LOGICAL_PROVIDER_ID } from "./models-declaration.js";
50
+
51
+ /** One physical account the logical provider may dispatch to. */
52
+ export interface LogicalPhysicalAccount {
53
+ /** Registered physical provider id, for example `anthropic-account-2`. */
54
+ providerId: string;
55
+ /** Physical provider-family token. OpenAI subscription and API tokens share one vendor. */
56
+ family: ManagedFamily;
57
+ /** Routing tier. Omitted legacy callers default from the physical family. */
58
+ providerType?: Exclude<ProviderType, "openrouter">;
59
+ /** Exact physical wire model ids this account serves. */
60
+ modelIds: string[];
61
+ /**
62
+ * Set when the PROVIDER reported this account exhausted from a usage
63
+ * snapshot. A request that merely failed and was retried elsewhere records a
64
+ * cooldown in {@link LogicalProviderDeps.state} instead; writing it here
65
+ * would make an ordinary cooldown look permanent.
66
+ */
67
+ exhausted?: boolean;
68
+ healthy?: boolean;
69
+ /** False when the credential is present but provably unusable. */
70
+ authenticated?: boolean;
71
+ modelSupported?: boolean;
72
+ /** Bounded, non-identifying. Never a human account name. */
73
+ accountFingerprint?: string;
74
+ /**
75
+ * Observed health of this account.
76
+ *
77
+ * Reported by preflight exactly as supplied, and omitted entirely when it
78
+ * was never observed. Defaulting a liveness window here would hand the
79
+ * operator an expiry nobody measured, presented as though it had been.
80
+ */
81
+ health?: {
82
+ healthy: boolean;
83
+ live: boolean;
84
+ /** Seconds until the observed credential window closes. */
85
+ expiresAt: number;
86
+ };
87
+ }
88
+
89
+ /** One physical attempt the logical provider makes. */
90
+ export interface LogicalDispatchCall {
91
+ providerId: string;
92
+ modelId: string;
93
+ context: unknown;
94
+ options?: unknown;
95
+ }
96
+
97
+ export interface LogicalObservation {
98
+ providerId: string;
99
+ modelId: string;
100
+ family: ManagedFamily;
101
+ providerType: Exclude<ProviderType, "openrouter">;
102
+ /**
103
+ * The physical account this observation belongs to. Present so a reader can
104
+ * attribute it without re-deriving the route from the provider id, which is
105
+ * how an observation ends up filed against the logical name instead.
106
+ */
107
+ route?: LogicalRouteFact;
108
+ kind?: string;
109
+ }
110
+
111
+ /** The physical account an observation or preflight result belongs to. */
112
+ export interface LogicalRouteFact {
113
+ readonly providerId: string;
114
+ readonly family: ManagedFamily;
115
+ readonly providerType: Exclude<ProviderType, "openrouter">;
116
+ readonly accountFingerprint: string;
117
+ readonly [key: string]: unknown;
118
+ }
119
+
120
+ /** Response shape accepted at the fail-soft attempt boundary. */
121
+ export type LogicalAttributionResponse = Readonly<{
122
+ status?: number;
123
+ headers: Record<string, string>;
124
+ }>;
125
+
126
+ export type ManagedAssistantRecordOutcome =
127
+ | { readonly status: "retained" }
128
+ | { readonly status: "failed" };
129
+
130
+ export type LogicalTerminalOutcome =
131
+ | ManagedAssistantRecordOutcome
132
+ | { readonly status: "not-attempted" }
133
+ | { readonly status: "not-attempted-capacity" };
134
+
135
+ /** One immutable, physical-route-bound attribution handle. */
136
+ export interface LogicalTerminalFailureFact {
137
+ readonly alreadyCooled: boolean;
138
+ readonly dispatchedModelId: string;
139
+ readonly failure?: ProviderFailureSignal;
140
+ /** Transient rollback for an association that later proves stale or ambiguous. */
141
+ readonly rollbackCooldown?: () => void;
142
+ }
143
+
144
+ export interface LogicalAttributionAttempt {
145
+ readonly onResponse: (response: LogicalAttributionResponse) => void;
146
+ readonly onAuthentication: (success: boolean) => void;
147
+ readonly onModelSupport: (supported: boolean) => void;
148
+ readonly onHealth: (health: unknown) => void;
149
+ readonly onPayload: (payload: unknown) => void;
150
+ readonly finish: (message: AssistantMessage) => void;
151
+ readonly fail: (
152
+ message: AssistantMessage,
153
+ failure?: LogicalTerminalFailureFact,
154
+ ) => boolean | void;
155
+ readonly abort: (message: AssistantMessage) => void;
156
+ readonly waitForTerminal: () => Promise<LogicalTerminalOutcome>;
157
+ /** Link an emitted public terminal to this attempt’s private physical terminal. */
158
+ readonly bindPublicTerminal?: (physical: AssistantMessage, publicMessage: AssistantMessage) => void;
159
+ }
160
+
161
+ /** Compatibility shape for callers that predate the terminal barrier. */
162
+ export type LogicalAttributionAttemptLike = Omit<LogicalAttributionAttempt, "waitForTerminal"> & {
163
+ readonly waitForTerminal?: LogicalAttributionAttempt["waitForTerminal"];
164
+ };
165
+
166
+ const ignoreAttribution = (): void => {};
167
+ const noTerminalOutcome = (): Promise<LogicalTerminalOutcome> =>
168
+ Promise.resolve({ status: "not-attempted" });
169
+ const acceptUnattributedFailure = (): true => true;
170
+
171
+ /** Shared inert handle for absent, failed, invalid, or released attribution. */
172
+ export const NOOP_LOGICAL_ATTRIBUTION_ATTEMPT: LogicalAttributionAttempt =
173
+ Object.freeze({
174
+ onResponse: ignoreAttribution,
175
+ onAuthentication: ignoreAttribution,
176
+ onModelSupport: ignoreAttribution,
177
+ onHealth: ignoreAttribution,
178
+ onPayload: ignoreAttribution,
179
+ finish: ignoreAttribution,
180
+ // An intentionally absent/released observer cannot reject host-retry
181
+ // bookkeeping. Real association stores return false for stale attempts.
182
+ fail: acceptUnattributedFailure,
183
+ abort: ignoreAttribution,
184
+ waitForTerminal: noTerminalOutcome,
185
+ });
186
+
187
+ /** Synchronous attribution lifecycle owned by one logical-provider session. */
188
+ export interface LogicalAttributionLifecycle {
189
+ beginAttempt(route: LogicalRouteFact): LogicalAttributionAttemptLike;
190
+ settle(): void;
191
+ shutdown(): void;
192
+ }
193
+
194
+ export interface LogicalRoutePinAccess {
195
+ readonly get: () => LogicalRoutePin | undefined;
196
+ readonly consume: (
197
+ expectedGeneration: number,
198
+ requestedModelId: string,
199
+ ) => LogicalRoutePin | undefined;
200
+ readonly clear: () => void;
201
+ }
202
+
203
+ export interface LogicalProviderDeps {
204
+ accounts: LogicalPhysicalAccount[];
205
+ dispatch: (
206
+ call: LogicalDispatchCall,
207
+ ) => Promise<AsyncIterable<unknown>> | AsyncIterable<unknown>;
208
+ logicalModelId?: string;
209
+ originFamily?: AllowedFamily;
210
+ /** Subscription-catalog owner of a declared logical model. */
211
+ modelVendor?: (modelId: string) => Vendor | undefined;
212
+ tierModelMap?: TierModelMap;
213
+ crossFamilyChains?: readonly CrossFamilyChain[];
214
+ onObservation?: (observation: LogicalObservation) => void;
215
+ attribution?: LogicalAttributionLifecycle;
216
+ /** Private session correlation for the public terminal and physical route. */
217
+ onPublicTerminal?: (physical: AssistantMessage, publicMessage: AssistantMessage) => void;
218
+ onDiagnostic?: (message: string) => void;
219
+ onShutdownAbort?: () => void;
220
+ scheduleContinuation?: (run: () => void) => void;
221
+ state?: RuntimeState;
222
+ routePin?: LogicalRoutePinAccess;
223
+ }
224
+
225
+ export interface LogicalProvider {
226
+ api: string;
227
+ streamSimple: (
228
+ model: unknown,
229
+ context: unknown,
230
+ options?: unknown,
231
+ ) => AsyncIterable<unknown> | Promise<AsyncIterable<unknown>>;
232
+ preflight?: (model: unknown) => unknown;
233
+ shutdown?: () => unknown;
234
+ }
235
+
236
+ export function safeAttributionCall(call: () => void): void {
237
+ try {
238
+ call();
239
+ } catch {
240
+ // Attribution must never replace a provider result.
241
+ }
242
+ }
243
+
244
+ export function safeBeginAttributionAttempt(
245
+ attribution: LogicalAttributionLifecycle | undefined,
246
+ route: LogicalRouteFact,
247
+ ): LogicalAttributionAttempt {
248
+ if (attribution === undefined) return NOOP_LOGICAL_ATTRIBUTION_ATTEMPT;
249
+ try {
250
+ const attempt = attribution.beginAttempt(route);
251
+ if (typeof attempt.waitForTerminal === "function") {
252
+ return attempt as LogicalAttributionAttempt;
253
+ }
254
+ return Object.freeze({
255
+ onResponse: attempt.onResponse,
256
+ onAuthentication: attempt.onAuthentication,
257
+ onModelSupport: attempt.onModelSupport,
258
+ onHealth: attempt.onHealth,
259
+ onPayload: attempt.onPayload,
260
+ finish: attempt.finish,
261
+ fail: attempt.fail,
262
+ abort: attempt.abort,
263
+ waitForTerminal: noTerminalOutcome,
264
+ ...(attempt.bindPublicTerminal === undefined ? {} : { bindPublicTerminal: attempt.bindPublicTerminal }),
265
+ });
266
+ } catch {
267
+ return NOOP_LOGICAL_ATTRIBUTION_ATTEMPT;
268
+ }
269
+ }
270
+
271
+ /**
272
+ * The wire model id a request names.
273
+ *
274
+ * The host describes a selected model in more than one shape depending on where
275
+ * the selection came from, so read `id` and ignore the rest. Only the exact id
276
+ * matters: it is matched byte-for-byte against what an account serves.
277
+ */
278
+ function requestedModelId(model: unknown): string | undefined {
279
+ if (typeof model === "string") return model;
280
+ if (typeof model !== "object" || model === null) return undefined;
281
+ const id = (model as { id?: unknown }).id;
282
+ return typeof id === "string" && id.length > 0 ? id : undefined;
283
+ }
284
+
285
+ function logicalProviderType(
286
+ account: LogicalPhysicalAccount,
287
+ ): Exclude<ProviderType, "openrouter"> {
288
+ // The distinct OpenAI families determine their tier regardless of a caller's
289
+ // optional hint. Anthropic alone needs the hint because both tiers share one
290
+ // physical family token.
291
+ return providerTypeFor(
292
+ account.family,
293
+ account.providerType === "owning-vendor-api" ? "api_key" : "unknown",
294
+ );
295
+ }
296
+
297
+ interface LogicalServingAccount {
298
+ readonly account: LogicalPhysicalAccount;
299
+ readonly resolvedModelId: string;
300
+ }
301
+
302
+ function resolveLogicalServingAccount(
303
+ account: LogicalPhysicalAccount,
304
+ modelId: string,
305
+ tierModelMap: TierModelMap,
306
+ ): LogicalServingAccount | undefined {
307
+ const providerType = logicalProviderType(account);
308
+ if (providerType === "subscription") {
309
+ return account.modelIds.includes(modelId)
310
+ ? { account, resolvedModelId: modelId }
311
+ : undefined;
312
+ }
313
+ // Google has no supported metered destination. Reject it before consulting
314
+ // tier mappings so a hostile or inconsistent account shape cannot select or
315
+ // dispatch a guessed Google model through a paid tier.
316
+ if (account.family === "google-antigravity") return undefined;
317
+ const resolvedModelId = resolveTierModel(
318
+ modelId,
319
+ vendorForFamily(account.family),
320
+ account.modelIds,
321
+ tierModelMap,
322
+ );
323
+ return resolvedModelId === undefined ? undefined : { account, resolvedModelId };
324
+ }
325
+
326
+ const PROVIDER_ERROR_CODE_SET: ReadonlySet<string> = new Set(
327
+ PROVIDER_ERROR_CODES,
328
+ );
329
+ const TRANSPORT_FAILURE_KIND_SET: ReadonlySet<string> = new Set(
330
+ TRANSPORT_FAILURE_KINDS,
331
+ );
332
+
333
+ function finiteNumber(value: unknown): number | undefined {
334
+ return typeof value === "number" && Number.isFinite(value)
335
+ ? value
336
+ : undefined;
337
+ }
338
+
339
+ const EXHAUSTION_LENGTH_MAX_OUTPUT_TOKENS = 1;
340
+ const EXHAUSTION_LENGTH_MIN_ALLOWANCE_TOKENS = 1_024;
341
+ const EXHAUSTION_LENGTH_ALLOWANCE_MULTIPLIER = 8;
342
+ const EXHAUSTION_LENGTH_MAX_CONTEXT_FRACTION = 0.8;
343
+ const EXHAUSTION_LENGTH_ERROR_MESSAGE = "provider returned error (usage-limit)";
344
+
345
+ function finiteNonNegative(value: unknown): number | undefined {
346
+ return typeof value === "number" && Number.isFinite(value) && value >= 0
347
+ ? value
348
+ : undefined;
349
+ }
350
+
351
+ function projectTerminalUsage(message: AssistantMessage): AssistantMessage["usage"] | undefined {
352
+ const rawUsage = (message as unknown as { usage?: unknown }).usage;
353
+ if (typeof rawUsage !== "object" || rawUsage === null) return undefined;
354
+ const usage = rawUsage as Record<string, unknown>;
355
+ const input = finiteNonNegative(usage.input);
356
+ const output = finiteNonNegative(usage.output);
357
+ const cacheRead = finiteNonNegative(usage.cacheRead);
358
+ const cacheWrite = finiteNonNegative(usage.cacheWrite);
359
+ const totalTokens = finiteNonNegative(usage.totalTokens);
360
+ const rawCost = usage.cost;
361
+ if (
362
+ input === undefined ||
363
+ output === undefined ||
364
+ cacheRead === undefined ||
365
+ cacheWrite === undefined ||
366
+ totalTokens === undefined ||
367
+ typeof rawCost !== "object" ||
368
+ rawCost === null
369
+ ) {
370
+ return undefined;
371
+ }
372
+ const cost = rawCost as Record<string, unknown>;
373
+ const costInput = finiteNonNegative(cost.input);
374
+ const costOutput = finiteNonNegative(cost.output);
375
+ const costCacheRead = finiteNonNegative(cost.cacheRead);
376
+ const costCacheWrite = finiteNonNegative(cost.cacheWrite);
377
+ const costTotal = finiteNonNegative(cost.total);
378
+ if (
379
+ costInput === undefined ||
380
+ costOutput === undefined ||
381
+ costCacheRead === undefined ||
382
+ costCacheWrite === undefined ||
383
+ costTotal === undefined
384
+ ) {
385
+ return undefined;
386
+ }
387
+ const cacheWrite1h = finiteNonNegative(usage.cacheWrite1h);
388
+ if (usage.cacheWrite1h !== undefined && cacheWrite1h === undefined) return undefined;
389
+ const reasoning = finiteNonNegative(usage.reasoning);
390
+ if (usage.reasoning !== undefined && reasoning === undefined) return undefined;
391
+ return {
392
+ input,
393
+ output,
394
+ cacheRead,
395
+ cacheWrite,
396
+ ...(cacheWrite1h === undefined ? {} : { cacheWrite1h }),
397
+ ...(reasoning === undefined ? {} : { reasoning }),
398
+ totalTokens,
399
+ cost: {
400
+ input: costInput,
401
+ output: costOutput,
402
+ cacheRead: costCacheRead,
403
+ cacheWrite: costCacheWrite,
404
+ total: costTotal,
405
+ },
406
+ };
407
+ }
408
+
409
+ interface ExhaustionLengthMatch {
410
+ readonly usage: AssistantMessage["usage"];
411
+ readonly timestamp: number;
412
+ }
413
+
414
+ function exhaustionLengthMatch(input: {
415
+ readonly message: AssistantMessage;
416
+ readonly model: unknown;
417
+ readonly options: SimpleStreamOptions | undefined;
418
+ readonly account: LogicalPhysicalAccount;
419
+ readonly currentAccounts: () => LogicalPhysicalAccount[];
420
+ }): ExhaustionLengthMatch | undefined {
421
+ try {
422
+ if (input.message.stopReason !== "length") return undefined;
423
+ if (logicalProviderType(input.account) !== "subscription") return undefined;
424
+ if (typeof input.model !== "object" || input.model === null) return undefined;
425
+ const model = input.model as Record<string, unknown>;
426
+ const contextWindow = finiteNonNegative(model.contextWindow);
427
+ const modelMaxTokens = finiteNonNegative(model.maxTokens);
428
+ if (
429
+ contextWindow === undefined ||
430
+ contextWindow === 0 ||
431
+ modelMaxTokens === undefined
432
+ ) {
433
+ return undefined;
434
+ }
435
+ const optionMaxTokens = input.options?.maxTokens;
436
+ const allowance =
437
+ optionMaxTokens === undefined
438
+ ? modelMaxTokens
439
+ : finiteNonNegative(optionMaxTokens);
440
+ if (allowance === undefined) return undefined;
441
+
442
+ const usage = projectTerminalUsage(input.message);
443
+ const timestamp = finiteNonNegative(input.message.timestamp);
444
+ if (usage === undefined || timestamp === undefined) return undefined;
445
+ if (usage.output > EXHAUSTION_LENGTH_MAX_OUTPUT_TOKENS) return undefined;
446
+ if (allowance < EXHAUSTION_LENGTH_MIN_ALLOWANCE_TOKENS) return undefined;
447
+ if (
448
+ allowance <
449
+ EXHAUSTION_LENGTH_ALLOWANCE_MULTIPLIER * Math.max(usage.output, 1)
450
+ ) {
451
+ return undefined;
452
+ }
453
+ const inputFromTotal = usage.totalTokens - usage.output;
454
+ const inputFromBreakdown = usage.input + usage.cacheRead + usage.cacheWrite;
455
+ if (
456
+ inputFromTotal < 0 ||
457
+ !Number.isFinite(inputFromBreakdown) ||
458
+ Math.max(inputFromTotal, inputFromBreakdown) >
459
+ contextWindow * EXHAUSTION_LENGTH_MAX_CONTEXT_FRACTION
460
+ ) {
461
+ return undefined;
462
+ }
463
+
464
+ const currentMatches = input.currentAccounts().filter(
465
+ (currentAccount) =>
466
+ currentAccount.providerId === input.account.providerId &&
467
+ currentAccount.family === input.account.family &&
468
+ logicalProviderType(currentAccount) === "subscription",
469
+ );
470
+ if (currentMatches.length !== 1) return undefined;
471
+ const currentAccount = currentMatches[0]!;
472
+ if (!(currentAccount.exhausted === true)) return undefined;
473
+ const requestFingerprint = input.account.accountFingerprint;
474
+ if (
475
+ typeof requestFingerprint === "string" &&
476
+ requestFingerprint.length > 0 &&
477
+ currentAccount.accountFingerprint !== requestFingerprint
478
+ ) {
479
+ return undefined;
480
+ }
481
+ return { usage, timestamp };
482
+ } catch {
483
+ return undefined;
484
+ }
485
+ }
486
+
487
+ /**
488
+ * Project a caller's thrown error into the narrow failure signal routing reads.
489
+ *
490
+ * Named fields only, each validated against the shipped vocabulary. A spread or
491
+ * a cast here would carry whatever else the caller hung on its error — message
492
+ * text, headers, a whole response body — into a structure that is classified,
493
+ * retained, and reported through a diagnostic sink. The model id comes from the
494
+ * selection this provider just made, never from the error.
495
+ */
496
+ function projectFailureSignal(
497
+ error: unknown,
498
+ modelId: string,
499
+ ): ProviderFailureSignal {
500
+ const source =
501
+ typeof error === "object" && error !== null
502
+ ? (error as Record<string, unknown>)
503
+ : {};
504
+ const httpStatus =
505
+ finiteNumber(source.httpStatus) ?? finiteNumber(source.status);
506
+ const rawCode = source.code;
507
+ const parsedCode = providerErrorCodeFromMessage(
508
+ typeof source.errorMessage === "string"
509
+ ? source.errorMessage
510
+ : typeof source.message === "string"
511
+ ? source.message
512
+ : undefined,
513
+ );
514
+ const code =
515
+ typeof rawCode === "string" && PROVIDER_ERROR_CODE_SET.has(rawCode)
516
+ ? (rawCode as ProviderErrorCode)
517
+ : parsedCode;
518
+ const transportKind = source.transportKind;
519
+ const retryAfterSeconds = finiteNumber(source.retryAfterSeconds);
520
+ const resetAtMs = finiteNumber(source.resetAtMs);
521
+ return {
522
+ modelId,
523
+ ...(httpStatus === undefined ? {} : { httpStatus }),
524
+ ...(code === undefined ? {} : { code }),
525
+ ...(typeof transportKind === "string" &&
526
+ TRANSPORT_FAILURE_KIND_SET.has(transportKind)
527
+ ? { transportKind: transportKind as TransportFailureKind }
528
+ : {}),
529
+ ...(retryAfterSeconds === undefined ? {} : { retryAfterSeconds }),
530
+ ...(resetAtMs === undefined ? {} : { resetAtMs }),
531
+ };
532
+ }
533
+
534
+ function safeProjectFailureSignal(
535
+ error: unknown,
536
+ modelId: string,
537
+ ): ProviderFailureSignal {
538
+ try {
539
+ return projectFailureSignal(error, modelId);
540
+ } catch {
541
+ return { modelId };
542
+ }
543
+ }
544
+
545
+ /**
546
+ * Project a logical account into the account shape routing understands.
547
+ *
548
+ * Provider-reported exhaustion becomes the usage shape routing actually reads.
549
+ * Without that translation an exhausted account looks healthy on the retry
550
+ * path while the first route excludes it, which is two notions of exhaustion
551
+ * free to drift apart.
552
+ */
553
+ function projectManagedAccount(
554
+ account: LogicalPhysicalAccount,
555
+ ): ManagedAccount {
556
+ return {
557
+ providerId: account.providerId,
558
+ family: account.family,
559
+ credentialType:
560
+ logicalProviderType(account) === "owning-vendor-api" ? "api_key" : "oauth",
561
+ modelIds: [...account.modelIds],
562
+ ...(account.exhausted === true
563
+ ? { fleetUsage: { remainingRequests: 0 } }
564
+ : {}),
565
+ ...(account.accountFingerprint === undefined
566
+ ? {}
567
+ : { accountFingerprint: account.accountFingerprint }),
568
+ };
569
+ }
570
+
571
+ interface HostRetryCooldownReceipt {
572
+ readonly alreadyCooled: boolean;
573
+ readonly rollback: () => void;
574
+ }
575
+
576
+ /** Synchronous, cooldown-only bookkeeping shared with host retry selection. */
577
+ export interface HostRetryCoordinator {
578
+ readonly state: RuntimeState;
579
+ recordFailure(input: {
580
+ account: LogicalPhysicalAccount;
581
+ requestedModelId: string;
582
+ dispatchedModelId: string;
583
+ error: unknown;
584
+ }): HostRetryCooldownReceipt;
585
+ }
586
+
587
+ export function createHostRetryCoordinator(
588
+ deps: LogicalProviderDeps,
589
+ ): HostRetryCoordinator {
590
+ const state = deps.state ?? new RuntimeState();
591
+ const config: MultiAccountConfig = {
592
+ ...DEFAULT_CONFIG,
593
+ tierModelMap: deps.tierModelMap ?? DEFAULT_CONFIG.tierModelMap,
594
+ };
595
+
596
+ const reportSwallowed = (): void => {
597
+ try {
598
+ deps.onDiagnostic?.(
599
+ "logical host retry bookkeeping failed and was ignored.",
600
+ );
601
+ } catch {
602
+ // Deliberately empty: bookkeeping cannot replace a provider result.
603
+ }
604
+ };
605
+ const noCooldownReceipt = (): HostRetryCooldownReceipt => ({
606
+ alreadyCooled: false,
607
+ rollback: () => {},
608
+ });
609
+
610
+ return {
611
+ state,
612
+ recordFailure(input) {
613
+ try {
614
+ const failure = projectFailureSignal(input.error, input.dispatchedModelId);
615
+ const managed = deps.accounts
616
+ .filter((account) => account.authenticated !== false)
617
+ .map(projectManagedAccount);
618
+ const failedAccount = managed.find(
619
+ (account) => account.providerId === input.account.providerId,
620
+ );
621
+ if (failedAccount === undefined) return noCooldownReceipt();
622
+ const failedFingerprint = failedAccount.accountFingerprint;
623
+ const affectedProviderIds = managed
624
+ .filter(
625
+ (account) =>
626
+ account.providerId === failedAccount.providerId ||
627
+ (failedFingerprint !== undefined &&
628
+ failedFingerprint.length > 0 &&
629
+ account.accountFingerprint === failedFingerprint),
630
+ )
631
+ .map((account) => account.providerId);
632
+ const transaction = state.runReversibleCooldownMutation(
633
+ affectedProviderIds,
634
+ () =>
635
+ recordFailureCooldown({
636
+ failedAccount,
637
+ accounts: managed,
638
+ failure,
639
+ state,
640
+ config,
641
+ nowMs: Date.now(),
642
+ }),
643
+ );
644
+ return {
645
+ alreadyCooled: transaction.value,
646
+ rollback: transaction.rollback,
647
+ };
648
+ } catch {
649
+ reportSwallowed();
650
+ return noCooldownReceipt();
651
+ }
652
+ },
653
+ };
654
+ }
655
+
656
+ /**
657
+ * Build the logical provider.
658
+ *
659
+ * Selection is exact and unforgiving by design. A model id no account serves is
660
+ * refused rather than mapped onto something close to it: substituting a
661
+ * catalog head, a same-family sibling or a vendor default would quietly run a
662
+ * different model than the operator chose, and the resulting answer would look
663
+ * entirely normal.
664
+ */
665
+ export function createLogicalProvider(
666
+ deps: LogicalProviderDeps,
667
+ ): LogicalProvider {
668
+ const coordinator = createHostRetryCoordinator(deps);
669
+
670
+ const diagnose = (message: string): void => {
671
+ deps.onDiagnostic?.(message);
672
+ };
673
+
674
+ const selectAccount = (
675
+ modelId: string,
676
+ consumeRoutePin: boolean,
677
+ ): LogicalServingAccount => {
678
+ const subscriptionVendors = new Set(
679
+ deps.accounts
680
+ .filter(
681
+ (account) =>
682
+ logicalProviderType(account) === "subscription" &&
683
+ account.modelIds.includes(modelId),
684
+ )
685
+ .map((account) => vendorForFamily(account.family)),
686
+ );
687
+ let modelVendor = deps.modelVendor?.(modelId);
688
+ if (modelVendor === undefined && deps.modelVendor === undefined) {
689
+ if (subscriptionVendors.size === 1) {
690
+ modelVendor = subscriptionVendors.values().next().value as Vendor;
691
+ } else if (subscriptionVendors.size > 1 && deps.originFamily !== undefined) {
692
+ const originVendor = vendorForFamily(deps.originFamily);
693
+ if (subscriptionVendors.has(originVendor)) modelVendor = originVendor;
694
+ }
695
+ }
696
+ if (modelVendor === undefined) {
697
+ if (subscriptionVendors.size > 1) {
698
+ throw new Error(
699
+ `the model id ${modelId} is served by more than one managed family, so no route can be chosen for it`,
700
+ );
701
+ }
702
+ throw new Error(
703
+ `no managed account serves the exact model id ${modelId}`,
704
+ );
705
+ }
706
+
707
+ const tierModelMap = deps.tierModelMap ?? DEFAULT_CONFIG.tierModelMap;
708
+ const serving = deps.accounts.flatMap((account) => {
709
+ if (vendorForFamily(account.family) !== modelVendor) return [];
710
+ const resolved = resolveLogicalServingAccount(account, modelId, tierModelMap);
711
+ return resolved === undefined ? [] : [resolved];
712
+ });
713
+ if (serving.length === 0) {
714
+ // Refusal, not substitution. This is also the load-bearing half of the
715
+ // exact-identity contract: a virtual id that is neither exact nor explicitly
716
+ // mapped to a catalog member must never reach a physical provider.
717
+ throw new Error(
718
+ `no managed account serves the exact model id ${modelId}`,
719
+ );
720
+ }
721
+
722
+ // Physical OpenAI subscription (`openai-codex`) and owning-vendor API
723
+ // (`openai`) accounts are one routing family: vendor `openai`. Ownership is
724
+ // fixed from the declared subscription catalog before API exact/map matches are
725
+ // admitted, so another vendor's API catalog cannot make a row ambiguous.
726
+ const nowMs = Date.now();
727
+ const eligible = serving.filter(({ account }) =>
728
+ logicalAccountEligible(
729
+ {
730
+ providerId: account.providerId,
731
+ exhausted: account.exhausted,
732
+ authenticated: account.authenticated,
733
+ },
734
+ // The coordinator's carrier, not `deps.state` directly. A failure
735
+ // this turn already recorded lives there, and reading anywhere else
736
+ // would re-select the account that just failed.
737
+ coordinator.state,
738
+ nowMs,
739
+ ),
740
+ );
741
+ if (consumeRoutePin) {
742
+ const pin = deps.routePin?.get();
743
+ if (pin !== undefined) {
744
+ const pinned = eligible.find(
745
+ ({ account, resolvedModelId }) =>
746
+ account.providerId === pin.destinationProviderId &&
747
+ account.family === pin.destinationFamily &&
748
+ logicalProviderType(account) === "subscription" &&
749
+ resolvedModelId === modelId &&
750
+ pin.requestedModelId === modelId,
751
+ );
752
+ if (pinned === undefined) {
753
+ deps.routePin?.clear();
754
+ } else {
755
+ const consumed = deps.routePin?.consume(pin.generation, modelId);
756
+ if (
757
+ consumed !== undefined &&
758
+ consumed.destinationProviderId === pinned.account.providerId &&
759
+ consumed.destinationFamily === pinned.account.family
760
+ ) {
761
+ return pinned;
762
+ }
763
+ }
764
+ }
765
+ }
766
+ eligible.sort(
767
+ (left, right) =>
768
+ tierRank(logicalProviderType(left.account)) -
769
+ tierRank(logicalProviderType(right.account)),
770
+ );
771
+ const chosen = eligible[0];
772
+ if (chosen === undefined) {
773
+ // Every account across every tier that serves this model is currently
774
+ // ineligible. The turn is refused before any physical request is issued;
775
+ // it is not parked, and no account is mutated by having been asked.
776
+ throw new Error(
777
+ `every managed account serving ${modelId} is currently ineligible`,
778
+ );
779
+ }
780
+ return chosen;
781
+ };
782
+
783
+ /**
784
+ * Watch a stream that opened successfully.
785
+ *
786
+ * A provider can accept a request, return 200, and fail part-way through the
787
+ * stream — Anthropic emits `event: error` that way. That failure is exactly
788
+ * as real as a rejected dispatch, so it records exactly the same cooldown;
789
+ * without this, a mid-stream rate limit would leave the account looking
790
+ * healthy and the host would retry straight back onto it.
791
+ *
792
+ * It also guarantees the host sees a terminal event.
793
+ *
794
+ * The host settles a turn only on a `done` or `error` event: `EventStream`
795
+ * resolves its final result from `isComplete(event)` on push, and the
796
+ * `forwardStream` adapter that wraps every provider stream ends with
797
+ * `end(undefined)` when the source exposes no `result()`, which the
798
+ * `result !== undefined` guard makes a no-op. A dispatched stream that simply
799
+ * runs out therefore leaves the caller's `prompt()` pending forever, with no
800
+ * timeout anywhere to break it.
801
+ *
802
+ * `deps.dispatch` is caller-supplied, so that is a careless caller wedging the
803
+ * host permanently. Real Pi streams always terminate, so this is defense in
804
+ * depth at an injectable boundary rather than a repair of a reachable hang.
805
+ *
806
+ * The synthesized event is `error`, never `done`. A stream that stopped
807
+ * without terminating did not answer, and reporting `done` would invent an
808
+ * answer that never arrived.
809
+ *
810
+ * It records no cooldown. Not terminating is a protocol fault in the dispatch
811
+ * implementation, not evidence the account is rate limited, and
812
+ * `classifyFailure` would read an unrecognizable failure as
813
+ * `cooldown("unknown", …)` — cooling a healthy account for someone else's bug.
814
+ * A genuinely failing account still throws, and the catch below cools it.
815
+ *
816
+ * `result()` is deliberately not forwarded. Exposing it would make
817
+ * `forwardStream` take its `await source.result()` branch, which for exactly
818
+ * this non-terminating source never settles.
819
+ */
820
+ const terminalAttribution = (
821
+ event: unknown,
822
+ ):
823
+ | { readonly outcome: "finish" | "fail" | "abort"; readonly message: AssistantMessage }
824
+ | undefined => {
825
+ if (typeof event !== "object" || event === null) return undefined;
826
+ const candidate = event as {
827
+ type?: unknown;
828
+ reason?: unknown;
829
+ message?: unknown;
830
+ error?: unknown;
831
+ };
832
+ const terminal = candidate.type === "done" ? candidate.message : candidate.error;
833
+ if (
834
+ (candidate.type !== "done" && candidate.type !== "error") ||
835
+ typeof terminal !== "object" ||
836
+ terminal === null ||
837
+ (terminal as { role?: unknown }).role !== "assistant"
838
+ ) {
839
+ return undefined;
840
+ }
841
+ return {
842
+ outcome:
843
+ candidate.type === "done"
844
+ ? "finish"
845
+ : candidate.reason === "aborted"
846
+ ? "abort"
847
+ : "fail",
848
+ message: terminal as AssistantMessage,
849
+ };
850
+ };
851
+
852
+ const syntheticErrorMessage = (
853
+ modelId: string,
854
+ errorMessage: string,
855
+ usage: AssistantMessage["usage"] = {
856
+ input: 0,
857
+ output: 0,
858
+ cacheRead: 0,
859
+ cacheWrite: 0,
860
+ totalTokens: 0,
861
+ cost: {
862
+ input: 0,
863
+ output: 0,
864
+ cacheRead: 0,
865
+ cacheWrite: 0,
866
+ total: 0,
867
+ },
868
+ },
869
+ timestamp = Date.now(),
870
+ ): AssistantMessage => ({
871
+ role: "assistant",
872
+ content: [],
873
+ api: LOGICAL_PROVIDER_ID,
874
+ provider: LOGICAL_PROVIDER_ID,
875
+ model: modelId,
876
+ usage,
877
+ stopReason: "error",
878
+ errorMessage,
879
+ timestamp,
880
+ });
881
+
882
+ const exhaustionLengthError = (
883
+ modelId: string,
884
+ match: ExhaustionLengthMatch,
885
+ ): AssistantMessage & { readonly code: "quota_exhausted" } => ({
886
+ ...syntheticErrorMessage(
887
+ modelId,
888
+ EXHAUSTION_LENGTH_ERROR_MESSAGE,
889
+ match.usage,
890
+ match.timestamp,
891
+ ),
892
+ code: "quota_exhausted",
893
+ });
894
+
895
+ const hostWouldRetry = (error: unknown): boolean => {
896
+ try {
897
+ const errorMessage =
898
+ error instanceof Error
899
+ ? error.message
900
+ : typeof error === "string"
901
+ ? error
902
+ : String(error);
903
+ return isRetryableAssistantError({
904
+ stopReason: "error",
905
+ errorMessage,
906
+ } as AssistantMessage);
907
+ } catch {
908
+ return false;
909
+ }
910
+ };
911
+
912
+ const classifiedErrorMessage = (
913
+ failure: ProviderFailureSignal,
914
+ hostRetryable: boolean,
915
+ ): string => {
916
+ const label = (() => {
917
+ switch (classifyFailure(failure).category) {
918
+ case "quota-rate-limit":
919
+ return "usage-limit";
920
+ case "terminal-auth":
921
+ case "transient-auth":
922
+ return "authentication";
923
+ case "permission":
924
+ return "permission";
925
+ case "config":
926
+ return "configuration";
927
+ case "transport":
928
+ return "network";
929
+ case "unknown":
930
+ return "unknown";
931
+ }
932
+ })();
933
+ const prefix = hostRetryable ? "provider returned error" : "provider_error";
934
+ return `${prefix} (${label})`;
935
+ };
936
+
937
+ const projectMessage = (message: AssistantMessage, modelId: string): AssistantMessage => ({
938
+ ...message, api: LOGICAL_PROVIDER_ID, provider: LOGICAL_PROVIDER_ID, model: modelId,
939
+ });
940
+
941
+ const projectEvent = (event: unknown, modelId: string): unknown => {
942
+ if (typeof event !== "object" || event === null) return event;
943
+ const candidate = event as Record<string, unknown>;
944
+ const key = candidate.type === "done" ? "message" : candidate.type === "error" ? "error" :
945
+ ["start", "text_start", "text_delta", "text_end", "thinking_start",
946
+ "thinking_delta", "thinking_end", "toolcall_start", "toolcall_delta",
947
+ "toolcall_end"].includes(String(candidate.type)) ? "partial" : undefined;
948
+ if (key === undefined) return event;
949
+ const value = candidate[key];
950
+ if (typeof value !== "object" || value === null) return event;
951
+ return { ...candidate, [key]: projectMessage(value as AssistantMessage, modelId) };
952
+ };
953
+
954
+ const watchStream = (
955
+ stream: AsyncIterable<unknown>,
956
+ model: unknown,
957
+ options: SimpleStreamOptions | undefined,
958
+ account: LogicalPhysicalAccount,
959
+ requestedModelId: string,
960
+ dispatchedModelId: string,
961
+ attempt: LogicalAttributionAttempt,
962
+ ): AsyncIterable<unknown> => ({
963
+ async *[Symbol.asyncIterator]() {
964
+ let sawTerminal = false;
965
+ let failureReceipt: HostRetryCooldownReceipt | undefined;
966
+ const recordFailureOnce = (error: unknown): HostRetryCooldownReceipt => {
967
+ failureReceipt ??= coordinator.recordFailure({
968
+ account,
969
+ requestedModelId,
970
+ dispatchedModelId,
971
+ error,
972
+ });
973
+ return failureReceipt;
974
+ };
975
+ const attributeFailure = (
976
+ message: AssistantMessage,
977
+ failure: ProviderFailureSignal,
978
+ receipt: HostRetryCooldownReceipt,
979
+ ): void => {
980
+ // Host retry needs the cooldown before the terminal reaches Pi, while
981
+ // exact identity cannot be accepted until association/message_end. The
982
+ // receipt makes the early write provisional and reverses it on uncertainty.
983
+ try {
984
+ const accepted = attempt.fail(message, {
985
+ alreadyCooled: receipt.alreadyCooled,
986
+ dispatchedModelId,
987
+ failure,
988
+ ...(receipt.alreadyCooled
989
+ ? { rollbackCooldown: receipt.rollback }
990
+ : {}),
991
+ });
992
+ if (accepted === false && deps.attribution !== undefined) receipt.rollback();
993
+ } catch {
994
+ if (deps.attribution !== undefined) receipt.rollback();
995
+ }
996
+ };
997
+ try {
998
+ // Keep physical events intact. A corroborated subscription-exhaustion
999
+ // length terminal still becomes a bounded retryable public error.
1000
+ for await (const event of stream) {
1001
+ const eventType =
1002
+ typeof event === "object" && event !== null
1003
+ ? (event as { type?: unknown }).type
1004
+ : undefined;
1005
+ if (eventType === "done" || eventType === "error") sawTerminal = true;
1006
+ const terminal = terminalAttribution(event);
1007
+ if (terminal !== undefined) {
1008
+ const { message, outcome } = terminal;
1009
+ if (outcome === "finish") {
1010
+ const match = exhaustionLengthMatch({
1011
+ message,
1012
+ model,
1013
+ options,
1014
+ account,
1015
+ currentAccounts: () => deps.accounts,
1016
+ });
1017
+ if (match !== undefined) {
1018
+ const replacement = exhaustionLengthError(requestedModelId, match);
1019
+ const failure = safeProjectFailureSignal(
1020
+ replacement,
1021
+ dispatchedModelId,
1022
+ );
1023
+ attributeFailure(
1024
+ replacement,
1025
+ failure,
1026
+ recordFailureOnce(replacement),
1027
+ );
1028
+ await attempt.waitForTerminal();
1029
+ yield { type: "error", reason: "error", error: replacement };
1030
+ continue;
1031
+ }
1032
+ safeAttributionCall(() => attempt.finish(message));
1033
+ } else if (outcome === "abort") {
1034
+ safeAttributionCall(() => attempt.abort(message));
1035
+ } else {
1036
+ const failure = safeProjectFailureSignal(message, dispatchedModelId);
1037
+ attributeFailure(message, failure, recordFailureOnce(message));
1038
+ }
1039
+ await attempt.waitForTerminal();
1040
+ }
1041
+ const publicEvent = projectEvent(event, requestedModelId);
1042
+ if (terminal !== undefined && publicEvent !== event) {
1043
+ const publicTerminal = terminalAttribution(publicEvent);
1044
+ if (publicTerminal !== undefined) {
1045
+ safeAttributionCall(() => attempt.bindPublicTerminal?.(terminal.message, publicTerminal.message));
1046
+ safeAttributionCall(() => deps.onPublicTerminal?.(terminal.message, publicTerminal.message));
1047
+ }
1048
+ }
1049
+ yield publicEvent;
1050
+ }
1051
+ } catch (error) {
1052
+ recordFailureOnce(error);
1053
+ throw error;
1054
+ }
1055
+ if (sawTerminal) return;
1056
+ deps.onDiagnostic?.(
1057
+ `logical dispatch stream for ${account.providerId} ended with no terminal event; ` +
1058
+ "reporting a synthetic failure so the turn cannot hang",
1059
+ );
1060
+ const syntheticMessage = syntheticErrorMessage(
1061
+ requestedModelId,
1062
+ "the dispatched stream ended without a terminal event",
1063
+ );
1064
+ safeAttributionCall(() => attempt.fail(syntheticMessage));
1065
+ await attempt.waitForTerminal();
1066
+ yield { type: "error", reason: "error", error: syntheticMessage };
1067
+ },
1068
+ });
1069
+
1070
+ return {
1071
+ api: LOGICAL_PROVIDER_ID,
1072
+
1073
+ // Deliberately `async`: a refusal must arrive as a rejected promise, not
1074
+ // as a synchronous throw. Callers invoke this and then await the result,
1075
+ // so a synchronous throw would escape past their error handling entirely.
1076
+ async streamSimple(model, context, rawOptions) {
1077
+ const options =
1078
+ typeof rawOptions === "object" && rawOptions !== null
1079
+ ? (rawOptions as SimpleStreamOptions)
1080
+ : undefined;
1081
+ const modelId = requestedModelId(model);
1082
+ if (modelId === undefined) {
1083
+ throw new Error(
1084
+ "the logical provider was asked for a request with no model id",
1085
+ );
1086
+ }
1087
+ const { account, resolvedModelId } = selectAccount(modelId, true);
1088
+ const providerType = logicalProviderType(account);
1089
+ deps.onObservation?.({
1090
+ providerId: account.providerId,
1091
+ modelId: resolvedModelId,
1092
+ family: account.family,
1093
+ providerType,
1094
+ });
1095
+ const route: LogicalRouteFact = Object.freeze({
1096
+ providerId: account.providerId,
1097
+ family: account.family,
1098
+ providerType,
1099
+ accountFingerprint: account.providerId,
1100
+ });
1101
+ const attempt = safeBeginAttributionAttempt(deps.attribution, route);
1102
+ if (account.authenticated !== undefined) {
1103
+ safeAttributionCall(() => attempt.onAuthentication(account.authenticated!));
1104
+ }
1105
+ if (account.modelSupported !== undefined) {
1106
+ safeAttributionCall(() => attempt.onModelSupport(account.modelSupported!));
1107
+ }
1108
+ if (account.health !== undefined) {
1109
+ safeAttributionCall(() => attempt.onHealth(account.health));
1110
+ }
1111
+ const originalOnPayload = options?.onPayload;
1112
+ const originalOnResponse = options?.onResponse;
1113
+ const wrappedOnPayload: NonNullable<SimpleStreamOptions["onPayload"]> = async (
1114
+ payload,
1115
+ payloadModel,
1116
+ ) => {
1117
+ safeAttributionCall(() => attempt.onPayload(payload));
1118
+ if (originalOnPayload === undefined) return undefined;
1119
+ return await originalOnPayload(payload, payloadModel);
1120
+ };
1121
+ const wrappedOnResponse: NonNullable<SimpleStreamOptions["onResponse"]> = async (
1122
+ response,
1123
+ responseModel,
1124
+ ) => {
1125
+ safeAttributionCall(() => attempt.onResponse(response));
1126
+ if (originalOnResponse !== undefined) {
1127
+ await originalOnResponse(response, responseModel);
1128
+ }
1129
+ };
1130
+ const attributedOptions: SimpleStreamOptions = {
1131
+ ...options,
1132
+ onPayload: wrappedOnPayload,
1133
+ onResponse: wrappedOnResponse,
1134
+ };
1135
+ // Dispatch only the exact requested id or the catalog-checked destination id
1136
+ // from the operator-authored tier map. A catalog head is never a fallback.
1137
+ let stream: AsyncIterable<unknown>;
1138
+ try {
1139
+ stream = await deps.dispatch({
1140
+ providerId: account.providerId,
1141
+ modelId: resolvedModelId,
1142
+ context,
1143
+ options: attributedOptions,
1144
+ });
1145
+ } catch (error) {
1146
+ // Cool synchronously so host retry cannot reselect this account, then
1147
+ // surface the rejection as a self-owned terminal that can be correlated
1148
+ // by exact object identity at message_end.
1149
+ const failure = safeProjectFailureSignal(error, resolvedModelId);
1150
+ const receipt = coordinator.recordFailure({
1151
+ account,
1152
+ requestedModelId: modelId,
1153
+ dispatchedModelId: resolvedModelId,
1154
+ error,
1155
+ });
1156
+ const syntheticMessage = syntheticErrorMessage(
1157
+ modelId,
1158
+ classifiedErrorMessage(failure, hostWouldRetry(error)),
1159
+ );
1160
+ try {
1161
+ const accepted = attempt.fail(syntheticMessage, {
1162
+ alreadyCooled: receipt.alreadyCooled,
1163
+ dispatchedModelId: resolvedModelId,
1164
+ failure,
1165
+ ...(receipt.alreadyCooled
1166
+ ? { rollbackCooldown: receipt.rollback }
1167
+ : {}),
1168
+ });
1169
+ if (accepted === false && deps.attribution !== undefined) receipt.rollback();
1170
+ } catch {
1171
+ if (deps.attribution !== undefined) receipt.rollback();
1172
+ }
1173
+ return (async function* () {
1174
+ await attempt.waitForTerminal();
1175
+ yield { type: "error", reason: "error", error: syntheticMessage };
1176
+ })();
1177
+ }
1178
+ return watchStream(
1179
+ stream,
1180
+ model,
1181
+ options,
1182
+ account,
1183
+ modelId,
1184
+ resolvedModelId,
1185
+ attempt,
1186
+ );
1187
+ },
1188
+
1189
+ preflight(model) {
1190
+ const modelId = requestedModelId(model);
1191
+ if (modelId === undefined) return undefined;
1192
+ try {
1193
+ const { account, resolvedModelId } = selectAccount(modelId, false);
1194
+ const providerType = logicalProviderType(account);
1195
+ // Everything below is read off the account that was actually
1196
+ // chosen. The logical provider has no health, no credential and no
1197
+ // expiry of its own, so reporting anything not observed here would
1198
+ // be reporting an invention as a measurement.
1199
+ const route: LogicalRouteFact = {
1200
+ providerId: account.providerId,
1201
+ family: account.family,
1202
+ providerType,
1203
+ accountFingerprint: account.providerId,
1204
+ };
1205
+ deps.onObservation?.({
1206
+ providerId: account.providerId,
1207
+ modelId: resolvedModelId,
1208
+ family: account.family,
1209
+ providerType,
1210
+ kind: "preflight",
1211
+ route,
1212
+ });
1213
+ return {
1214
+ route,
1215
+ ...(account.health === undefined ? {} : { health: account.health }),
1216
+ ...(account.authenticated === undefined
1217
+ ? {}
1218
+ : { authenticated: account.authenticated }),
1219
+ ...(account.modelSupported === undefined
1220
+ ? {}
1221
+ : { modelSupported: account.modelSupported }),
1222
+ };
1223
+ } catch (error) {
1224
+ diagnose(
1225
+ error instanceof Error
1226
+ ? `logical preflight refused ${modelId}: ${error.message}`
1227
+ : `logical preflight refused ${modelId}`,
1228
+ );
1229
+ return undefined;
1230
+ }
1231
+ },
1232
+
1233
+ shutdown() {
1234
+ safeAttributionCall(() => deps.attribution?.shutdown());
1235
+ },
1236
+ };
1237
+ }