@centerforagenticai/pi-multi-account 0.1.4 → 0.1.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/packages/pi-anthropic-oauth/src/stream.ts +8 -0
- package/src/codex-adapter.ts +81 -7
- package/src/config.ts +151 -0
- package/src/host-final-stop-message.ts +16 -4
- package/src/index.ts +17 -0
- package/src/logical-provider.ts +127 -1
- package/src/model-fallback-policy.ts +384 -0
- package/src/recovery-engine.ts +354 -68
- package/src/recovery-plan.ts +7 -1
- package/src/recovery-send-evidence.ts +29 -0
- package/src/refusal-advice.ts +139 -0
- package/src/shared-usage.ts +21 -4
- package/src/usage-fetch.ts +42 -59
package/package.json
CHANGED
|
@@ -221,10 +221,18 @@ export function streamAnthropicOAuth(
|
|
|
221
221
|
// which throws under fine-grained-tool-streaming (input may be invalid
|
|
222
222
|
// mid-flight) and aborts the turn. The raw stream yields the same
|
|
223
223
|
// RawMessageStreamEvents; tool args are already parsed leniently below.
|
|
224
|
+
// A finite non-negative integer request-local maxRetries bounds the SDK's
|
|
225
|
+
// own retries (0 sends exactly once); any other value keeps the SDK default.
|
|
226
|
+
const maxRetries = options?.maxRetries;
|
|
224
227
|
const { data: anthropicStream, response: httpResponse } =
|
|
225
228
|
await client.messages
|
|
226
229
|
.create(params, {
|
|
227
230
|
signal: options?.signal,
|
|
231
|
+
...(Number.isFinite(maxRetries) &&
|
|
232
|
+
Number.isInteger(maxRetries) &&
|
|
233
|
+
(maxRetries ?? -1) >= 0
|
|
234
|
+
? { maxRetries }
|
|
235
|
+
: {}),
|
|
228
236
|
})
|
|
229
237
|
.withResponse();
|
|
230
238
|
|
package/src/codex-adapter.ts
CHANGED
|
@@ -193,14 +193,81 @@ function withAliasEvent(
|
|
|
193
193
|
}
|
|
194
194
|
}
|
|
195
195
|
|
|
196
|
+
/**
|
|
197
|
+
* One attributed, sanitized error terminal for an upstream that failed before
|
|
198
|
+
* producing a stream (a synchronous throw, a rejected setup, or a throwing
|
|
199
|
+
* iterator). The shape matches the host's own setup-error terminal: no content,
|
|
200
|
+
* zero usage, no diagnostics. Only the bounded, redacted error text survives.
|
|
201
|
+
* A failure after the caller's signal fired is a cancellation: it ends as
|
|
202
|
+
* `aborted`, matching the maintained stream, so it never cools the account.
|
|
203
|
+
*/
|
|
204
|
+
function aliasSetupErrorMessage(
|
|
205
|
+
error: unknown,
|
|
206
|
+
aliasModel: Model<Api>,
|
|
207
|
+
aborted: boolean,
|
|
208
|
+
): AssistantMessage & { stopReason: "error" | "aborted" } {
|
|
209
|
+
let detail: string;
|
|
210
|
+
try {
|
|
211
|
+
detail = error instanceof Error ? error.message : String(error);
|
|
212
|
+
} catch {
|
|
213
|
+
detail = "Codex alias stream setup failed";
|
|
214
|
+
}
|
|
215
|
+
return {
|
|
216
|
+
role: "assistant",
|
|
217
|
+
content: [],
|
|
218
|
+
api: aliasModel.api,
|
|
219
|
+
provider: aliasModel.provider,
|
|
220
|
+
model: aliasModel.id,
|
|
221
|
+
usage: {
|
|
222
|
+
input: 0,
|
|
223
|
+
output: 0,
|
|
224
|
+
cacheRead: 0,
|
|
225
|
+
cacheWrite: 0,
|
|
226
|
+
totalTokens: 0,
|
|
227
|
+
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
|
|
228
|
+
},
|
|
229
|
+
stopReason: aborted ? "aborted" : "error",
|
|
230
|
+
errorMessage: sanitizeDiagnosticText(detail),
|
|
231
|
+
timestamp: Date.now(),
|
|
232
|
+
};
|
|
233
|
+
}
|
|
234
|
+
|
|
235
|
+
function isTerminalEvent(event: AssistantMessageEvent): boolean {
|
|
236
|
+
return event.type === "done" || event.type === "error";
|
|
237
|
+
}
|
|
238
|
+
|
|
239
|
+
/**
|
|
240
|
+
* Forwards upstream events with alias attribution. Every failure mode ends in
|
|
241
|
+
* exactly one terminal: a rejected setup or a throwing iterator before any
|
|
242
|
+
* terminal becomes one attributed error event, and the attributed stream always
|
|
243
|
+
* ends, so no failure escapes as an unhandled rejection or a hang.
|
|
244
|
+
*/
|
|
196
245
|
function reattributeStream(
|
|
197
|
-
upstream: AssistantMessageEventStream
|
|
246
|
+
upstream: AssistantMessageEventStream | PromiseLike<AssistantMessageEventStream>,
|
|
198
247
|
aliasModel: Model<Api>,
|
|
248
|
+
signal: AbortSignal | undefined,
|
|
199
249
|
): AssistantMessageEventStream {
|
|
200
250
|
const attributed = createAssistantMessageEventStream();
|
|
201
251
|
void (async () => {
|
|
202
|
-
|
|
203
|
-
|
|
252
|
+
let sawTerminal = false;
|
|
253
|
+
try {
|
|
254
|
+
for await (const event of await upstream) {
|
|
255
|
+
if (sawTerminal) continue;
|
|
256
|
+
if (isTerminalEvent(event)) sawTerminal = true;
|
|
257
|
+
attributed.push(withAliasEvent(event, aliasModel));
|
|
258
|
+
}
|
|
259
|
+
} catch (error) {
|
|
260
|
+
if (!sawTerminal) {
|
|
261
|
+
sawTerminal = true;
|
|
262
|
+
const message = aliasSetupErrorMessage(
|
|
263
|
+
error,
|
|
264
|
+
aliasModel,
|
|
265
|
+
signal?.aborted === true,
|
|
266
|
+
);
|
|
267
|
+
attributed.push({ type: "error", reason: message.stopReason, error: message });
|
|
268
|
+
}
|
|
269
|
+
} finally {
|
|
270
|
+
attributed.end();
|
|
204
271
|
}
|
|
205
272
|
})();
|
|
206
273
|
return attributed;
|
|
@@ -296,10 +363,17 @@ export function createCodexAliasStream(
|
|
|
296
363
|
};
|
|
297
364
|
}
|
|
298
365
|
|
|
299
|
-
|
|
300
|
-
|
|
301
|
-
|
|
302
|
-
)
|
|
366
|
+
// Defensive only under the live host: its `lazyApi` stream catches a setup
|
|
367
|
+
// throw first and returns its own setup-error terminal. A non-lazy upstream
|
|
368
|
+
// (plain Node) can still throw synchronously; convert that into the same
|
|
369
|
+
// single attributed, sanitized terminal as any other failure (UPSTREAM.md).
|
|
370
|
+
let upstreamStream: ReturnType<CodexUpstreamStream>;
|
|
371
|
+
try {
|
|
372
|
+
upstreamStream = upstream(upstreamModel, upstreamContext, upstreamOptions);
|
|
373
|
+
} catch (error) {
|
|
374
|
+
upstreamStream = Promise.reject(error) as unknown as ReturnType<CodexUpstreamStream>;
|
|
375
|
+
}
|
|
376
|
+
return reattributeStream(upstreamStream, aliasModel, options?.signal);
|
|
303
377
|
};
|
|
304
378
|
return aliasStream as unknown as NonNullable<ProviderConfig["streamSimple"]>;
|
|
305
379
|
}
|
package/src/config.ts
CHANGED
|
@@ -29,6 +29,7 @@ import {
|
|
|
29
29
|
} from "node:fs";
|
|
30
30
|
import { randomUUID } from "node:crypto";
|
|
31
31
|
import { basename, dirname, isAbsolute, join } from "node:path";
|
|
32
|
+
import { sanitizeDiagnosticText } from "./diagnostics.js";
|
|
32
33
|
import { PROJECT_KEY_PATTERN } from "./project-identity.js";
|
|
33
34
|
import {
|
|
34
35
|
AccountRateHistoryError,
|
|
@@ -167,6 +168,15 @@ export interface CrossFamilyChain {
|
|
|
167
168
|
readonly to: AllowedFamily;
|
|
168
169
|
}
|
|
169
170
|
|
|
171
|
+
/** One exact, directional model-substitution egress authorization. */
|
|
172
|
+
export interface ModelFallbackEgressAuthorization {
|
|
173
|
+
readonly sourceModelId: string;
|
|
174
|
+
readonly destinationModelId: string;
|
|
175
|
+
}
|
|
176
|
+
|
|
177
|
+
/** Exact source unified model id -> ordered exact fallback model ids. */
|
|
178
|
+
export type ModelFallbackMap = Readonly<Record<string, readonly string[]>>;
|
|
179
|
+
|
|
170
180
|
export interface MultiAccountConfig {
|
|
171
181
|
readonly accountLimit: number;
|
|
172
182
|
readonly sameFamilyFailover: boolean;
|
|
@@ -238,6 +248,10 @@ export interface MultiAccountConfig {
|
|
|
238
248
|
*/
|
|
239
249
|
readonly preferredModels: Readonly<Record<string, readonly string[]>>;
|
|
240
250
|
readonly tierModelMap: TierModelMap;
|
|
251
|
+
/** Explicit ordered fallback policy. Absent or empty disables model substitution. */
|
|
252
|
+
readonly modelFallbacks?: ModelFallbackMap;
|
|
253
|
+
/** Directional authorization required in addition to policy for cross-vendor edges. */
|
|
254
|
+
readonly modelFallbackEgress?: readonly ModelFallbackEgressAuthorization[];
|
|
241
255
|
/**
|
|
242
256
|
* How close to expiry a credential may get before routing prefers a fresher
|
|
243
257
|
* same-family account, in milliseconds. Pre-emption avoids spending a turn to
|
|
@@ -284,6 +298,8 @@ export const DEFAULT_CONFIG: MultiAccountConfig = {
|
|
|
284
298
|
accountRateHistory: {},
|
|
285
299
|
preferredModels: {},
|
|
286
300
|
tierModelMap: Object.freeze({}),
|
|
301
|
+
modelFallbacks: Object.freeze({}),
|
|
302
|
+
modelFallbackEgress: Object.freeze([]),
|
|
287
303
|
// Comfortably longer than a turn, short enough that accounts are not retired
|
|
288
304
|
// while they still have useful life.
|
|
289
305
|
preemptiveExpiryWindowMs: 120_000,
|
|
@@ -309,6 +325,8 @@ const CONFIG_KEYS = new Set<keyof MultiAccountConfig>([
|
|
|
309
325
|
"accountRateHistory",
|
|
310
326
|
"preferredModels",
|
|
311
327
|
"tierModelMap",
|
|
328
|
+
"modelFallbacks",
|
|
329
|
+
"modelFallbackEgress",
|
|
312
330
|
"preemptiveExpiryWindowMs",
|
|
313
331
|
"usageFetchEnabled",
|
|
314
332
|
]);
|
|
@@ -689,6 +707,135 @@ function isValidTierModelId(value: unknown): value is string {
|
|
|
689
707
|
);
|
|
690
708
|
}
|
|
691
709
|
|
|
710
|
+
export const MAX_MODEL_FALLBACK_SOURCES = 128;
|
|
711
|
+
export const MAX_MODEL_FALLBACK_DESTINATIONS = 16;
|
|
712
|
+
export const MAX_MODEL_FALLBACK_EDGES = 256;
|
|
713
|
+
/**
|
|
714
|
+
* Exact catalog ids: managed wire ids, OpenRouter `vendor/model[:variant]`
|
|
715
|
+
* slugs, and the `~vendor/...` and `@cf/...` forms in the pinned catalog.
|
|
716
|
+
*/
|
|
717
|
+
const MODEL_FALLBACK_ID_PATTERN = /^[A-Za-z0-9~@][A-Za-z0-9._:/@~-]{0,255}$/;
|
|
718
|
+
|
|
719
|
+
/**
|
|
720
|
+
* A fallback id may later appear in routing diagnostics, so it is accepted only
|
|
721
|
+
* when the shared diagnostic sanitizer would leave it byte-for-byte unchanged.
|
|
722
|
+
* Anything the sanitizer would redact anywhere in the string (credential
|
|
723
|
+
* prefixes, JWTs, token/canary shapes, long opaque runs) is rejected here, and
|
|
724
|
+
* the rejection message never echoes the value.
|
|
725
|
+
*/
|
|
726
|
+
function isValidModelFallbackId(value: unknown): value is string {
|
|
727
|
+
return (
|
|
728
|
+
typeof value === "string" &&
|
|
729
|
+
MODEL_FALLBACK_ID_PATTERN.test(value) &&
|
|
730
|
+
sanitizeDiagnosticText(value) === value
|
|
731
|
+
);
|
|
732
|
+
}
|
|
733
|
+
|
|
734
|
+
function invalidModelFallbackId(field: string): ConfigValidationError {
|
|
735
|
+
return new ConfigValidationError(
|
|
736
|
+
`${field} must use an exact, non-credential model id of at most 256 characters.`,
|
|
737
|
+
);
|
|
738
|
+
}
|
|
739
|
+
|
|
740
|
+
function parseModelFallbacks(value: unknown): ModelFallbackMap {
|
|
741
|
+
if (value === undefined) return DEFAULT_CONFIG.modelFallbacks ?? Object.freeze({});
|
|
742
|
+
if (!isPlainObject(value)) {
|
|
743
|
+
throw new ConfigValidationError("modelFallbacks must be a plain JSON object.");
|
|
744
|
+
}
|
|
745
|
+
const entries = Object.entries(value);
|
|
746
|
+
if (entries.length > MAX_MODEL_FALLBACK_SOURCES) {
|
|
747
|
+
throw new ConfigValidationError(
|
|
748
|
+
`modelFallbacks must contain at most ${MAX_MODEL_FALLBACK_SOURCES} sources.`,
|
|
749
|
+
);
|
|
750
|
+
}
|
|
751
|
+
const parsed = Object.create(null) as Record<string, readonly string[]>;
|
|
752
|
+
for (const [sourceModelId, rawDestinations] of entries) {
|
|
753
|
+
if (!isValidModelFallbackId(sourceModelId)) {
|
|
754
|
+
throw invalidModelFallbackId("modelFallbacks source ids");
|
|
755
|
+
}
|
|
756
|
+
if (!Array.isArray(rawDestinations) || rawDestinations.length === 0) {
|
|
757
|
+
throw new ConfigValidationError(
|
|
758
|
+
"modelFallbacks values must be non-empty arrays of model ids.",
|
|
759
|
+
);
|
|
760
|
+
}
|
|
761
|
+
if (rawDestinations.length > MAX_MODEL_FALLBACK_DESTINATIONS) {
|
|
762
|
+
throw new ConfigValidationError(
|
|
763
|
+
`modelFallbacks may contain at most ${MAX_MODEL_FALLBACK_DESTINATIONS} destinations per source.`,
|
|
764
|
+
);
|
|
765
|
+
}
|
|
766
|
+
const destinations: string[] = [];
|
|
767
|
+
const seen = new Set<string>();
|
|
768
|
+
for (const destinationModelId of rawDestinations) {
|
|
769
|
+
if (!isValidModelFallbackId(destinationModelId)) {
|
|
770
|
+
throw invalidModelFallbackId("modelFallbacks destination ids");
|
|
771
|
+
}
|
|
772
|
+
if (destinationModelId === sourceModelId) {
|
|
773
|
+
throw new ConfigValidationError(
|
|
774
|
+
"modelFallbacks cannot list the source model as its own destination.",
|
|
775
|
+
);
|
|
776
|
+
}
|
|
777
|
+
if (seen.has(destinationModelId)) {
|
|
778
|
+
throw new ConfigValidationError(
|
|
779
|
+
"modelFallbacks cannot contain duplicate destinations for one source.",
|
|
780
|
+
);
|
|
781
|
+
}
|
|
782
|
+
seen.add(destinationModelId);
|
|
783
|
+
destinations.push(destinationModelId);
|
|
784
|
+
}
|
|
785
|
+
parsed[sourceModelId] = Object.freeze(destinations);
|
|
786
|
+
}
|
|
787
|
+
return Object.freeze(parsed);
|
|
788
|
+
}
|
|
789
|
+
|
|
790
|
+
function parseModelFallbackEgress(
|
|
791
|
+
value: unknown,
|
|
792
|
+
): readonly ModelFallbackEgressAuthorization[] {
|
|
793
|
+
if (value === undefined) return DEFAULT_CONFIG.modelFallbackEgress ?? Object.freeze([]);
|
|
794
|
+
if (!Array.isArray(value)) {
|
|
795
|
+
throw new ConfigValidationError("modelFallbackEgress must be an array.");
|
|
796
|
+
}
|
|
797
|
+
if (value.length > MAX_MODEL_FALLBACK_EDGES) {
|
|
798
|
+
throw new ConfigValidationError(
|
|
799
|
+
`modelFallbackEgress must contain at most ${MAX_MODEL_FALLBACK_EDGES} edges.`,
|
|
800
|
+
);
|
|
801
|
+
}
|
|
802
|
+
const parsed: ModelFallbackEgressAuthorization[] = [];
|
|
803
|
+
const seen = new Set<string>();
|
|
804
|
+
for (const candidate of value) {
|
|
805
|
+
if (!isPlainObject(candidate)) {
|
|
806
|
+
throw new ConfigValidationError("modelFallbackEgress entries must be plain objects.");
|
|
807
|
+
}
|
|
808
|
+
const unknownKeys = Object.keys(candidate).filter(
|
|
809
|
+
(key) => key !== "sourceModelId" && key !== "destinationModelId",
|
|
810
|
+
);
|
|
811
|
+
if (unknownKeys.length > 0) {
|
|
812
|
+
throw new ConfigValidationError(
|
|
813
|
+
"modelFallbackEgress entries contain an unsupported field.",
|
|
814
|
+
);
|
|
815
|
+
}
|
|
816
|
+
const sourceModelId = candidate["sourceModelId"];
|
|
817
|
+
const destinationModelId = candidate["destinationModelId"];
|
|
818
|
+
if (!isValidModelFallbackId(sourceModelId)) {
|
|
819
|
+
throw invalidModelFallbackId("modelFallbackEgress sourceModelId");
|
|
820
|
+
}
|
|
821
|
+
if (!isValidModelFallbackId(destinationModelId)) {
|
|
822
|
+
throw invalidModelFallbackId("modelFallbackEgress destinationModelId");
|
|
823
|
+
}
|
|
824
|
+
if (sourceModelId === destinationModelId) {
|
|
825
|
+
throw new ConfigValidationError(
|
|
826
|
+
"modelFallbackEgress cannot authorize a model to itself.",
|
|
827
|
+
);
|
|
828
|
+
}
|
|
829
|
+
const key = `${sourceModelId}\u0000${destinationModelId}`;
|
|
830
|
+
if (seen.has(key)) {
|
|
831
|
+
throw new ConfigValidationError("modelFallbackEgress cannot contain duplicate edges.");
|
|
832
|
+
}
|
|
833
|
+
seen.add(key);
|
|
834
|
+
parsed.push(Object.freeze({ sourceModelId, destinationModelId }));
|
|
835
|
+
}
|
|
836
|
+
return Object.freeze(parsed);
|
|
837
|
+
}
|
|
838
|
+
|
|
692
839
|
export function parseTierModelMap(value: unknown): TierModelMap {
|
|
693
840
|
if (value === undefined) return DEFAULT_CONFIG.tierModelMap;
|
|
694
841
|
if (!isRecord(value)) {
|
|
@@ -1119,6 +1266,8 @@ export function parseConfig(value: unknown): MultiAccountConfig {
|
|
|
1119
1266
|
const accountRateHistory = parseAccountRateHistory(value["accountRateHistory"]);
|
|
1120
1267
|
const preferredModels = parsePreferredModels(value["preferredModels"]);
|
|
1121
1268
|
const tierModelMap = parseTierModelMap(value["tierModelMap"]);
|
|
1269
|
+
const modelFallbacks = parseModelFallbacks(value["modelFallbacks"]);
|
|
1270
|
+
const modelFallbackEgress = parseModelFallbackEgress(value["modelFallbackEgress"]);
|
|
1122
1271
|
const preemptiveExpiryWindowMs =
|
|
1123
1272
|
value["preemptiveExpiryWindowMs"] ??
|
|
1124
1273
|
DEFAULT_CONFIG.preemptiveExpiryWindowMs;
|
|
@@ -1206,6 +1355,8 @@ export function parseConfig(value: unknown): MultiAccountConfig {
|
|
|
1206
1355
|
accountRateHistory,
|
|
1207
1356
|
preferredModels,
|
|
1208
1357
|
tierModelMap,
|
|
1358
|
+
modelFallbacks,
|
|
1359
|
+
modelFallbackEgress,
|
|
1209
1360
|
preemptiveExpiryWindowMs,
|
|
1210
1361
|
usageFetchEnabled,
|
|
1211
1362
|
};
|
|
@@ -10,6 +10,13 @@ export const REFUSAL_FALLBACK_MESSAGE = "The model refused to complete the reque
|
|
|
10
10
|
export const UNKNOWN_STOP_FALLBACK_MESSAGE =
|
|
11
11
|
"Provider stopped with an unrecognized stop reason";
|
|
12
12
|
|
|
13
|
+
/**
|
|
14
|
+
* Appended when a reworded unknown stop reason omits words a host predicate acts
|
|
15
|
+
* on, so the published reason does not read as complete. Neither pinned
|
|
16
|
+
* predicate matches it, and the marked message is probed again before use.
|
|
17
|
+
*/
|
|
18
|
+
export const PARTLY_WITHHELD_MARKER = " (partly withheld)";
|
|
19
|
+
|
|
13
20
|
/** Upper bound on the normalized stop reason named in a reworded message. */
|
|
14
21
|
const MAX_NAMED_REASON_LENGTH = 64;
|
|
15
22
|
|
|
@@ -23,7 +30,7 @@ const MAX_NAMED_REASON_LENGTH = 64;
|
|
|
23
30
|
* An unreadable predicate result counts as a match, so the caller keeps looking
|
|
24
31
|
* for a safer form.
|
|
25
32
|
*/
|
|
26
|
-
function hostWouldRedispatch(message: AssistantMessage, text: string): boolean {
|
|
33
|
+
export function hostWouldRedispatch(message: AssistantMessage, text: string): boolean {
|
|
27
34
|
try {
|
|
28
35
|
const probe = { ...message, errorMessage: text };
|
|
29
36
|
return isRetryableAssistantError(probe) || isContextOverflow(probe, 0);
|
|
@@ -68,8 +75,9 @@ function normalizedReasonWords(rawStopReason: unknown): string[] {
|
|
|
68
75
|
* The first form keeps every normalized word. When that still matches a host
|
|
69
76
|
* predicate (a reason containing "overloaded" or "timeout", say), the second
|
|
70
77
|
* form admits words in order and drops each one whose addition would make the
|
|
71
|
-
* message match.
|
|
72
|
-
*
|
|
78
|
+
* message match. When any word was dropped the result ends with
|
|
79
|
+
* `PARTLY_WITHHELD_MARKER`; that marked text is probed too, and an unsafe or
|
|
80
|
+
* empty result yields `undefined`.
|
|
73
81
|
*/
|
|
74
82
|
function unknownStopNamingReason(
|
|
75
83
|
message: AssistantMessage,
|
|
@@ -84,7 +92,11 @@ function unknownStopNamingReason(
|
|
|
84
92
|
const tentative = namedReasonMessage([...kept, word].join(" "));
|
|
85
93
|
if (!hostWouldRedispatch(message, tentative)) kept.push(word);
|
|
86
94
|
}
|
|
87
|
-
|
|
95
|
+
if (kept.length === 0) return undefined;
|
|
96
|
+
const named = namedReasonMessage(kept.join(" "));
|
|
97
|
+
if (kept.length === words.length) return named;
|
|
98
|
+
const marked = `${named}${PARTLY_WITHHELD_MARKER}`;
|
|
99
|
+
return hostWouldRedispatch(message, marked) ? undefined : marked;
|
|
88
100
|
}
|
|
89
101
|
|
|
90
102
|
/**
|
package/src/index.ts
CHANGED
|
@@ -80,6 +80,7 @@ import {
|
|
|
80
80
|
import { DiagnosticStore } from "./diagnostic-store.js";
|
|
81
81
|
import { DeclarationNoticeMarker } from "./declaration-notice-marker.js";
|
|
82
82
|
import { DiagnosticLog } from "./diagnostics.js";
|
|
83
|
+
import { createRefusalAdvisor } from "./refusal-advice.js";
|
|
83
84
|
import {
|
|
84
85
|
classifyProviderId,
|
|
85
86
|
createPublicAuthStorageAdapter,
|
|
@@ -4159,6 +4160,9 @@ export const createMultiAccountExtension =
|
|
|
4159
4160
|
const lastFailure = new Map<string, ProviderFailureSignal>();
|
|
4160
4161
|
const handledFailures = new WeakSet<object>();
|
|
4161
4162
|
const observedMessages = new WeakSet<object>();
|
|
4163
|
+
// Operator advice for a structured refusal. A delegate-owned in-process
|
|
4164
|
+
// session shares the foreground UI, so only the foreground advises.
|
|
4165
|
+
const refusalAdvisor = createRefusalAdvisor({ foreground: !delegateOwnedSession });
|
|
4162
4166
|
const recordUsage = (
|
|
4163
4167
|
observation: () => UsageObservation | undefined,
|
|
4164
4168
|
): void => {
|
|
@@ -5153,6 +5157,17 @@ export const createMultiAccountExtension =
|
|
|
5153
5157
|
acceptedLogicalAssociation,
|
|
5154
5158
|
);
|
|
5155
5159
|
}
|
|
5160
|
+
// Advice only: names the physical route that refused, never routes.
|
|
5161
|
+
refusalAdvisor.advise(
|
|
5162
|
+
originalMessage,
|
|
5163
|
+
messageContext,
|
|
5164
|
+
acceptedLogicalAssociation && logicalAssociation !== undefined
|
|
5165
|
+
? {
|
|
5166
|
+
providerId: logicalAssociation.route.providerId,
|
|
5167
|
+
modelId: logicalAssociation.dispatchedModelId,
|
|
5168
|
+
}
|
|
5169
|
+
: {},
|
|
5170
|
+
);
|
|
5156
5171
|
|
|
5157
5172
|
try {
|
|
5158
5173
|
const validOriginalMessage = originalMessage as AssistantMessage;
|
|
@@ -5240,6 +5255,8 @@ export const createMultiAccountExtension =
|
|
|
5240
5255
|
) {
|
|
5241
5256
|
return;
|
|
5242
5257
|
}
|
|
5258
|
+
// Advice only: a managed account's structured refusal never routes.
|
|
5259
|
+
refusalAdvisor.advise(event.message, messageContext, identity);
|
|
5243
5260
|
const subscriptionFamily = isRoutingEligibleAccountFamily(messageSlot)
|
|
5244
5261
|
? messageSlot.family
|
|
5245
5262
|
: undefined;
|
package/src/logical-provider.ts
CHANGED
|
@@ -16,6 +16,7 @@
|
|
|
16
16
|
import { logicalAccountEligible, recordFailureCooldown } from "./routing.js";
|
|
17
17
|
import type { ManagedAccount } from "./routing.js";
|
|
18
18
|
import {
|
|
19
|
+
isContextOverflow,
|
|
19
20
|
isRetryableAssistantError,
|
|
20
21
|
type AssistantMessage,
|
|
21
22
|
type ProviderResponse,
|
|
@@ -353,6 +354,36 @@ const EXHAUSTION_LENGTH_ALLOWANCE_MULTIPLIER = 8;
|
|
|
353
354
|
const EXHAUSTION_LENGTH_MAX_CONTEXT_FRACTION = 0.8;
|
|
354
355
|
const EXHAUSTION_LENGTH_ERROR_MESSAGE = "provider returned error (usage-limit)";
|
|
355
356
|
|
|
357
|
+
/**
|
|
358
|
+
* Fixed public text for a setup-shaped context overflow. The pinned host's
|
|
359
|
+
* `isContextOverflow` matches it (so the host compacts and retries once) and
|
|
360
|
+
* `isRetryableAssistantError` does not (so the host does not fail over).
|
|
361
|
+
*/
|
|
362
|
+
export const SETUP_CONTEXT_OVERFLOW_MESSAGE = "context_length_exceeded (provider_error)";
|
|
363
|
+
|
|
364
|
+
type SetupFailureDisposition = "context-overflow" | "retryable" | "host-final";
|
|
365
|
+
|
|
366
|
+
/**
|
|
367
|
+
* How the pinned host treats the raw setup text. The host checks the two
|
|
368
|
+
* predicates separately: `_handlePostAgentRun` compacts and retries once on
|
|
369
|
+
* `isContextOverflow`, while `_isRetryableError` excludes overflow and fails
|
|
370
|
+
* over on `isRetryableAssistantError`. An unreadable predicate result counts
|
|
371
|
+
* as host-final.
|
|
372
|
+
*/
|
|
373
|
+
function setupFailureDisposition(
|
|
374
|
+
message: AssistantMessage,
|
|
375
|
+
raw: unknown,
|
|
376
|
+
): SetupFailureDisposition {
|
|
377
|
+
if (typeof raw !== "string" || raw.length === 0) return "host-final";
|
|
378
|
+
try {
|
|
379
|
+
const probe = { ...message, errorMessage: raw };
|
|
380
|
+
if (isContextOverflow(probe, 0)) return "context-overflow";
|
|
381
|
+
return isRetryableAssistantError(probe) ? "retryable" : "host-final";
|
|
382
|
+
} catch {
|
|
383
|
+
return "host-final";
|
|
384
|
+
}
|
|
385
|
+
}
|
|
386
|
+
|
|
356
387
|
function finiteNonNegative(value: unknown): number | undefined {
|
|
357
388
|
return typeof value === "number" && Number.isFinite(value) && value >= 0
|
|
358
389
|
? value
|
|
@@ -542,6 +573,54 @@ function projectFailureSignal(
|
|
|
542
573
|
};
|
|
543
574
|
}
|
|
544
575
|
|
|
576
|
+
/**
|
|
577
|
+
* Whether a physical terminal is the host's setup-error shape: the first event
|
|
578
|
+
* of the stream is an `error` with no content, all-zero usage, no diagnostics,
|
|
579
|
+
* no structured stop code, and no structured failure evidence.
|
|
580
|
+
*
|
|
581
|
+
* That is what pi-ai `lazyStream` (`createSetupErrorMessage`) publishes when a
|
|
582
|
+
* provider stream throws or rejects before it starts, so its `errorMessage` is
|
|
583
|
+
* raw exception text, not provider-authored failure prose. The production cause
|
|
584
|
+
* of the observed setup `TypeError` is not known (see UPSTREAM.md). The same
|
|
585
|
+
* shape also carries transient pre-start failures ("fetch failed", a 503 before
|
|
586
|
+
* `start`), so the caller decides retryability from the text, never publishes
|
|
587
|
+
* it. A real provider failure that carries a recognized code or status keeps
|
|
588
|
+
* its own text and routing.
|
|
589
|
+
*/
|
|
590
|
+
function isUnclassifiedSetupFailure(
|
|
591
|
+
message: AssistantMessage,
|
|
592
|
+
failure: ProviderFailureSignal,
|
|
593
|
+
): boolean {
|
|
594
|
+
try {
|
|
595
|
+
if (message.stopReason !== "error") return false;
|
|
596
|
+
if (!Array.isArray(message.content) || message.content.length !== 0) return false;
|
|
597
|
+
const diagnostics = (message as { diagnostics?: unknown }).diagnostics;
|
|
598
|
+
if (diagnostics !== undefined && !(Array.isArray(diagnostics) && diagnostics.length === 0)) {
|
|
599
|
+
return false;
|
|
600
|
+
}
|
|
601
|
+
if ((message as { code?: unknown }).code !== undefined) return false;
|
|
602
|
+
const usage = projectTerminalUsage(message);
|
|
603
|
+
if (
|
|
604
|
+
usage === undefined ||
|
|
605
|
+
usage.input !== 0 ||
|
|
606
|
+
usage.output !== 0 ||
|
|
607
|
+
usage.cacheRead !== 0 ||
|
|
608
|
+
usage.cacheWrite !== 0 ||
|
|
609
|
+
usage.totalTokens !== 0 ||
|
|
610
|
+
usage.cost.total !== 0
|
|
611
|
+
) {
|
|
612
|
+
return false;
|
|
613
|
+
}
|
|
614
|
+
return (
|
|
615
|
+
failure.code === undefined &&
|
|
616
|
+
failure.httpStatus === undefined &&
|
|
617
|
+
failure.transportKind === undefined
|
|
618
|
+
);
|
|
619
|
+
} catch {
|
|
620
|
+
return false;
|
|
621
|
+
}
|
|
622
|
+
}
|
|
623
|
+
|
|
545
624
|
function safeProjectFailureSignal(
|
|
546
625
|
error: unknown,
|
|
547
626
|
modelId: string,
|
|
@@ -964,6 +1043,18 @@ export function createLogicalProvider(
|
|
|
964
1043
|
return { ...candidate, [key]: projectMessage(value as AssistantMessage, modelId) };
|
|
965
1044
|
};
|
|
966
1045
|
|
|
1046
|
+
const withPublicErrorMessage = (event: unknown, errorMessage: string): unknown => {
|
|
1047
|
+
if (typeof event !== "object" || event === null) return event;
|
|
1048
|
+
const candidate = event as Record<string, unknown>;
|
|
1049
|
+
if (candidate.type !== "error" || typeof candidate.error !== "object" || candidate.error === null) {
|
|
1050
|
+
return event;
|
|
1051
|
+
}
|
|
1052
|
+
return {
|
|
1053
|
+
...candidate,
|
|
1054
|
+
error: { ...(candidate.error as AssistantMessage), errorMessage },
|
|
1055
|
+
};
|
|
1056
|
+
};
|
|
1057
|
+
|
|
967
1058
|
const watchStream = (
|
|
968
1059
|
stream: AsyncIterable<unknown>,
|
|
969
1060
|
model: unknown,
|
|
@@ -975,6 +1066,7 @@ export function createLogicalProvider(
|
|
|
975
1066
|
): AsyncIterable<unknown> => ({
|
|
976
1067
|
async *[Symbol.asyncIterator]() {
|
|
977
1068
|
let sawTerminal = false;
|
|
1069
|
+
let sawEvent = false;
|
|
978
1070
|
let failureReceipt: HostRetryCooldownReceipt | undefined;
|
|
979
1071
|
const recordFailureOnce = (error: unknown): HostRetryCooldownReceipt => {
|
|
980
1072
|
failureReceipt ??= coordinator.recordFailure({
|
|
@@ -1016,7 +1108,10 @@ export function createLogicalProvider(
|
|
|
1016
1108
|
? (event as { type?: unknown }).type
|
|
1017
1109
|
: undefined;
|
|
1018
1110
|
if (eventType === "done" || eventType === "error") sawTerminal = true;
|
|
1111
|
+
const firstEvent = !sawEvent;
|
|
1112
|
+
sawEvent = true;
|
|
1019
1113
|
const terminal = terminalAttribution(event);
|
|
1114
|
+
let setupFailureMessage: string | undefined;
|
|
1020
1115
|
if (terminal !== undefined) {
|
|
1021
1116
|
const { message, outcome } = terminal;
|
|
1022
1117
|
if (outcome === "finish") {
|
|
@@ -1048,10 +1143,41 @@ export function createLogicalProvider(
|
|
|
1048
1143
|
} else {
|
|
1049
1144
|
const failure = safeProjectFailureSignal(message, dispatchedModelId);
|
|
1050
1145
|
attributeFailure(message, failure, recordFailureOnce(message));
|
|
1146
|
+
if (firstEvent && isUnclassifiedSetupFailure(message, failure)) {
|
|
1147
|
+
// The physical account is cooled above exactly like any other
|
|
1148
|
+
// failure. Only the published text changes: the raw text is
|
|
1149
|
+
// replaced by the bounded classified message the
|
|
1150
|
+
// rejected-dispatch path uses, so it is never published. The
|
|
1151
|
+
// setup shape is shared by transient pre-start failures ("fetch
|
|
1152
|
+
// failed", a 503 before `start`), pre-start context overflows
|
|
1153
|
+
// (a Codex 400 or Anthropic 413), and deterministic setup throws.
|
|
1154
|
+
// The raw text decides the form: an overflow publishes a fixed
|
|
1155
|
+
// overflow message so the host compacts instead of failing
|
|
1156
|
+
// over; a host-retryable text keeps the retryable form and
|
|
1157
|
+
// fails over; anything else is host-final (`provider_error`),
|
|
1158
|
+
// so a deterministic fault is not repeated on the next account.
|
|
1159
|
+
const disposition = setupFailureDisposition(message, message.errorMessage);
|
|
1160
|
+
setupFailureMessage =
|
|
1161
|
+
disposition === "context-overflow"
|
|
1162
|
+
? SETUP_CONTEXT_OVERFLOW_MESSAGE
|
|
1163
|
+
: classifiedErrorMessage(failure, disposition === "retryable");
|
|
1164
|
+
try {
|
|
1165
|
+
deps.onDiagnostic?.(
|
|
1166
|
+
`logical dispatch for ${account.providerId} failed during stream setup; ` +
|
|
1167
|
+
"published a classified failure instead of the raw setup error",
|
|
1168
|
+
);
|
|
1169
|
+
} catch {
|
|
1170
|
+
// A diagnostic sink failure cannot replace a provider result.
|
|
1171
|
+
}
|
|
1172
|
+
}
|
|
1051
1173
|
}
|
|
1052
1174
|
await attempt.waitForTerminal();
|
|
1053
1175
|
}
|
|
1054
|
-
const
|
|
1176
|
+
const projectedEvent = projectEvent(event, requestedModelId);
|
|
1177
|
+
const publicEvent =
|
|
1178
|
+
setupFailureMessage === undefined
|
|
1179
|
+
? projectedEvent
|
|
1180
|
+
: withPublicErrorMessage(projectedEvent, setupFailureMessage);
|
|
1055
1181
|
if (terminal !== undefined && publicEvent !== event) {
|
|
1056
1182
|
const publicTerminal = terminalAttribution(publicEvent);
|
|
1057
1183
|
if (publicTerminal !== undefined) {
|