@centerforagenticai/pi-multi-account 0.1.2 → 0.1.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/src/anthropic-adaptive-stream.ts +81 -15
- package/src/anthropic-alias-stream.ts +9 -1
- package/src/codex-adapter.ts +28 -4
- package/src/error-classification.ts +13 -1
- package/src/host-final-stop-message.ts +124 -0
- package/src/index.ts +24 -4
- package/src/logical-provider.ts +30 -1
- package/src/routing.ts +17 -1
- package/src/shared-usage.ts +83 -176
- package/src/usage-fetch.ts +59 -22
- package/src/usage.ts +29 -26
package/package.json
CHANGED
|
@@ -24,8 +24,9 @@
|
|
|
24
24
|
* Provenance: pi-anthropic-oauth@0.2.4-intel.2, private pin
|
|
25
25
|
* 99ac00f290efbcb93e7f23e6b7482d8727367aa7, upstream baseline
|
|
26
26
|
* 53266ecb51b6d1890ef3f7251a64cb1d71d96099, source src/stream.ts.
|
|
27
|
-
* Local delta: adaptive thinking emits type=adaptive and mapped output effort
|
|
28
|
-
* without budget_tokens;
|
|
27
|
+
* Local delta: adaptive thinking emits type=adaptive and mapped output effort
|
|
28
|
+
* without budget_tokens; request-local retry/timeout options reach the SDK; and
|
|
29
|
+
* refusal or unknown stop reasons retain bounded, structured error details.
|
|
29
30
|
*/
|
|
30
31
|
|
|
31
32
|
import { Anthropic } from "@anthropic-ai/sdk";
|
|
@@ -41,6 +42,8 @@ import {
|
|
|
41
42
|
type SimpleStreamOptions,
|
|
42
43
|
type StopReason,
|
|
43
44
|
} from "@earendil-works/pi-ai";
|
|
45
|
+
import { sanitizeDiagnosticText } from "./diagnostics.js";
|
|
46
|
+
import type { ProviderErrorCode } from "./error-classification.js";
|
|
44
47
|
type IndexedBlock =
|
|
45
48
|
| ({ type: "text"; text: string } & { index: number })
|
|
46
49
|
| ({ type: "thinking"; thinking: string; thinkingSignature?: string } & {
|
|
@@ -95,18 +98,51 @@ const REQUIRED_BETAS = [
|
|
|
95
98
|
"interleaved-thinking-2025-05-14",
|
|
96
99
|
] as const;
|
|
97
100
|
|
|
98
|
-
|
|
101
|
+
type StopReasonResult = Readonly<{
|
|
102
|
+
stopReason: StopReason;
|
|
103
|
+
errorMessage?: string;
|
|
104
|
+
code?: Extract<ProviderErrorCode, "refusal" | "unknown_stop">;
|
|
105
|
+
}>;
|
|
106
|
+
|
|
107
|
+
function mapStopReason(
|
|
108
|
+
reason: string | null | undefined,
|
|
109
|
+
stopDetails?: unknown,
|
|
110
|
+
): StopReasonResult {
|
|
99
111
|
switch (reason) {
|
|
100
112
|
case "end_turn":
|
|
101
113
|
case "pause_turn":
|
|
102
114
|
case "stop_sequence":
|
|
103
|
-
return "stop";
|
|
115
|
+
return { stopReason: "stop" };
|
|
104
116
|
case "max_tokens":
|
|
105
|
-
return "length";
|
|
117
|
+
return { stopReason: "length" };
|
|
106
118
|
case "tool_use":
|
|
107
|
-
return "toolUse";
|
|
108
|
-
|
|
109
|
-
|
|
119
|
+
return { stopReason: "toolUse" };
|
|
120
|
+
case "refusal": {
|
|
121
|
+
const explanation =
|
|
122
|
+
typeof stopDetails === "object" &&
|
|
123
|
+
stopDetails !== null &&
|
|
124
|
+
typeof (stopDetails as { explanation?: unknown }).explanation === "string"
|
|
125
|
+
? sanitizeDiagnosticText(
|
|
126
|
+
(stopDetails as { explanation: string }).explanation,
|
|
127
|
+
)
|
|
128
|
+
: "";
|
|
129
|
+
return {
|
|
130
|
+
stopReason: "error",
|
|
131
|
+
errorMessage:
|
|
132
|
+
explanation || "The model refused to complete the request",
|
|
133
|
+
code: "refusal",
|
|
134
|
+
};
|
|
135
|
+
}
|
|
136
|
+
default: {
|
|
137
|
+
const boundedReason = sanitizeDiagnosticText(reason ?? "unknown");
|
|
138
|
+
return {
|
|
139
|
+
stopReason: "error",
|
|
140
|
+
errorMessage: sanitizeDiagnosticText(
|
|
141
|
+
`Provider stopped with: ${boundedReason}`,
|
|
142
|
+
),
|
|
143
|
+
code: "unknown_stop",
|
|
144
|
+
};
|
|
145
|
+
}
|
|
110
146
|
}
|
|
111
147
|
}
|
|
112
148
|
|
|
@@ -163,7 +199,7 @@ export function streamAnthropicAdaptive(
|
|
|
163
199
|
const stream = createAssistantMessageEventStream();
|
|
164
200
|
|
|
165
201
|
void (async () => {
|
|
166
|
-
const output: AssistantMessage = {
|
|
202
|
+
const output: AssistantMessage & { code?: ProviderErrorCode } = {
|
|
167
203
|
role: "assistant",
|
|
168
204
|
content: [],
|
|
169
205
|
api: model.api,
|
|
@@ -276,10 +312,24 @@ export function streamAnthropicAdaptive(
|
|
|
276
312
|
// which throws under fine-grained-tool-streaming (input may be invalid
|
|
277
313
|
// mid-flight) and aborts the turn. The raw stream yields the same
|
|
278
314
|
// RawMessageStreamEvents; tool args are already parsed leniently below.
|
|
315
|
+
const maxRetries = options?.maxRetries;
|
|
316
|
+
const timeoutMs = options?.timeoutMs;
|
|
279
317
|
const { data: anthropicStream, response: httpResponse } =
|
|
280
318
|
await client.messages
|
|
281
319
|
.create(params, {
|
|
282
320
|
signal: options?.signal,
|
|
321
|
+
...(Number.isFinite(maxRetries) &&
|
|
322
|
+
Number.isInteger(maxRetries) &&
|
|
323
|
+
(maxRetries ?? -1) >= 0
|
|
324
|
+
? { maxRetries }
|
|
325
|
+
: {}),
|
|
326
|
+
// Any finite positive timeout is honored; the SDK timer takes whole
|
|
327
|
+
// milliseconds, so a fractional value is floored to at least 1.
|
|
328
|
+
...(typeof timeoutMs === "number" &&
|
|
329
|
+
Number.isFinite(timeoutMs) &&
|
|
330
|
+
timeoutMs > 0
|
|
331
|
+
? { timeout: Math.max(1, Math.floor(timeoutMs)) }
|
|
332
|
+
: {}),
|
|
283
333
|
})
|
|
284
334
|
.withResponse();
|
|
285
335
|
|
|
@@ -474,7 +524,19 @@ export function streamAnthropicAdaptive(
|
|
|
474
524
|
}
|
|
475
525
|
|
|
476
526
|
if (event.type === "message_delta") {
|
|
477
|
-
|
|
527
|
+
const rawStopReason = event.delta.stop_reason;
|
|
528
|
+
const mapped = mapStopReason(
|
|
529
|
+
rawStopReason,
|
|
530
|
+
(event.delta as { stop_details?: unknown }).stop_details,
|
|
531
|
+
);
|
|
532
|
+
output.stopReason = mapped.stopReason;
|
|
533
|
+
if (typeof rawStopReason === "string") {
|
|
534
|
+
output.rawStopReason = sanitizeDiagnosticText(rawStopReason);
|
|
535
|
+
}
|
|
536
|
+
if (mapped.errorMessage !== undefined) {
|
|
537
|
+
output.errorMessage = mapped.errorMessage;
|
|
538
|
+
}
|
|
539
|
+
if (mapped.code !== undefined) output.code = mapped.code;
|
|
478
540
|
output.usage.input =
|
|
479
541
|
(event.usage as { input_tokens?: number }).input_tokens ||
|
|
480
542
|
output.usage.input;
|
|
@@ -505,11 +567,15 @@ export function streamAnthropicAdaptive(
|
|
|
505
567
|
}
|
|
506
568
|
|
|
507
569
|
if (options?.signal?.aborted) throw new Error("Request aborted");
|
|
508
|
-
|
|
509
|
-
type: "
|
|
510
|
-
|
|
511
|
-
|
|
512
|
-
|
|
570
|
+
if (output.stopReason === "error") {
|
|
571
|
+
stream.push({ type: "error", reason: "error", error: output });
|
|
572
|
+
} else {
|
|
573
|
+
stream.push({
|
|
574
|
+
type: "done",
|
|
575
|
+
reason: output.stopReason as "stop" | "length" | "toolUse",
|
|
576
|
+
message: output,
|
|
577
|
+
});
|
|
578
|
+
}
|
|
513
579
|
stream.end();
|
|
514
580
|
} catch (error) {
|
|
515
581
|
for (const block of output.content as Array<{
|
|
@@ -13,6 +13,7 @@ import {
|
|
|
13
13
|
sanitizeDiagnosticText,
|
|
14
14
|
sanitizeHeaderValue,
|
|
15
15
|
} from "./diagnostics.js";
|
|
16
|
+
import { hostFinalStopMessage } from "./host-final-stop-message.js";
|
|
16
17
|
|
|
17
18
|
export const ANTHROPIC_ALIAS_API = "hypha-anthropic-oauth" as const;
|
|
18
19
|
|
|
@@ -53,12 +54,19 @@ function withAliasAttribution(
|
|
|
53
54
|
};
|
|
54
55
|
}
|
|
55
56
|
|
|
57
|
+
/**
|
|
58
|
+
* Bounds an upstream error and, for a structured refusal or unknown stop,
|
|
59
|
+
* publishes the shared host-final message. A direct alias turn reaches the
|
|
60
|
+
* host's retry and compaction predicates without the unified provider, so
|
|
61
|
+
* provider-authored stop wording must not make the host resend the request.
|
|
62
|
+
*/
|
|
56
63
|
function sanitizeUpstreamError(message: AssistantMessage): AssistantMessage {
|
|
57
64
|
if (message.errorMessage === undefined) return message;
|
|
58
|
-
|
|
65
|
+
const sanitized: AssistantMessage = {
|
|
59
66
|
...message,
|
|
60
67
|
errorMessage: sanitizeDiagnosticText(message.errorMessage),
|
|
61
68
|
};
|
|
69
|
+
return { ...sanitized, ...hostFinalStopMessage(sanitized) };
|
|
62
70
|
}
|
|
63
71
|
|
|
64
72
|
function withAliasEvent(
|
package/src/codex-adapter.ts
CHANGED
|
@@ -235,9 +235,33 @@ function normalizeAliasContext(
|
|
|
235
235
|
}
|
|
236
236
|
|
|
237
237
|
/**
|
|
238
|
-
*
|
|
238
|
+
* Temporary containment for the Pi 0.99 Codex WebSocket failure: a WebSocket
|
|
239
|
+
* error leaves the session in a state where the next Codex turn crashes with
|
|
240
|
+
* "Cannot read properties of undefined (reading 'length')". Routes this
|
|
241
|
+
* extension owns always request SSE. The base `openai-codex` provider is never
|
|
242
|
+
* touched; only options on calls we already route are adjusted. Remove once the
|
|
243
|
+
* WebSocket path is fixed upstream.
|
|
244
|
+
*/
|
|
245
|
+
export const CODEX_FORCED_TRANSPORT = "sse" as const;
|
|
246
|
+
|
|
247
|
+
/** Returns options with Codex transport pinned to SSE, and whether a value was overridden. */
|
|
248
|
+
export function forceCodexSseOptions<T extends SimpleStreamOptions>(
|
|
249
|
+
options: T | undefined,
|
|
250
|
+
): { options: T; overridden: boolean } {
|
|
251
|
+
if (options?.transport === CODEX_FORCED_TRANSPORT) {
|
|
252
|
+
return { options, overridden: false };
|
|
253
|
+
}
|
|
254
|
+
return {
|
|
255
|
+
options: { ...(options ?? {}), transport: CODEX_FORCED_TRANSPORT } as T,
|
|
256
|
+
overridden: true,
|
|
257
|
+
};
|
|
258
|
+
}
|
|
259
|
+
|
|
260
|
+
/**
|
|
261
|
+
* Wraps the exact maintained stream without replacing OAuth. Pi's
|
|
239
262
|
* alias-resolved apiKey and every other option field are forwarded unchanged;
|
|
240
|
-
*
|
|
263
|
+
* model/context identity and callback attribution are adapted, and transport
|
|
264
|
+
* is pinned to SSE (see {@link CODEX_FORCED_TRANSPORT}).
|
|
241
265
|
*/
|
|
242
266
|
export function createCodexAliasStream(
|
|
243
267
|
maintainedStream: NonNullable<ProviderConfig["streamSimple"]>,
|
|
@@ -251,12 +275,12 @@ export function createCodexAliasStream(
|
|
|
251
275
|
};
|
|
252
276
|
const upstreamContext = normalizeAliasContext(context, aliasModel);
|
|
253
277
|
|
|
254
|
-
let upstreamOptions = options;
|
|
278
|
+
let upstreamOptions = forceCodexSseOptions(options).options;
|
|
255
279
|
if (options?.onPayload || options?.onResponse) {
|
|
256
280
|
const aliasOnPayload = options.onPayload;
|
|
257
281
|
const aliasOnResponse = options.onResponse;
|
|
258
282
|
upstreamOptions = {
|
|
259
|
-
...
|
|
283
|
+
...upstreamOptions,
|
|
260
284
|
...(aliasOnPayload
|
|
261
285
|
? {
|
|
262
286
|
onPayload: (payload: unknown) =>
|
|
@@ -13,6 +13,8 @@ export const PROVIDER_ERROR_CODES = [
|
|
|
13
13
|
"model_not_found",
|
|
14
14
|
"unsupported_api_version",
|
|
15
15
|
"invalid_request",
|
|
16
|
+
"refusal",
|
|
17
|
+
"unknown_stop",
|
|
16
18
|
] as const;
|
|
17
19
|
export type ProviderErrorCode = (typeof PROVIDER_ERROR_CODES)[number];
|
|
18
20
|
|
|
@@ -50,10 +52,12 @@ export type FailureCategory =
|
|
|
50
52
|
|
|
51
53
|
export interface FailureClassification {
|
|
52
54
|
readonly category: FailureCategory;
|
|
55
|
+
readonly kind?: "refusal" | "unknown-stop";
|
|
53
56
|
readonly accountAction:
|
|
54
57
|
| "cooldown-and-route"
|
|
55
58
|
| "invalidate-and-route"
|
|
56
|
-
| "route-without-retry"
|
|
59
|
+
| "route-without-retry"
|
|
60
|
+
| "retain-account";
|
|
57
61
|
readonly cooldownReason?: CooldownReason;
|
|
58
62
|
readonly serverHint?: NumericServerHint;
|
|
59
63
|
}
|
|
@@ -248,6 +252,14 @@ export function classifyFailure(
|
|
|
248
252
|
return { category: "config", accountAction: "route-without-retry" };
|
|
249
253
|
}
|
|
250
254
|
|
|
255
|
+
if (signal.code === "refusal" || signal.code === "unknown_stop") {
|
|
256
|
+
return {
|
|
257
|
+
category: "unknown",
|
|
258
|
+
kind: signal.code === "refusal" ? "refusal" : "unknown-stop",
|
|
259
|
+
accountAction: "retain-account",
|
|
260
|
+
};
|
|
261
|
+
}
|
|
262
|
+
|
|
251
263
|
if (isTransportFailure(signal)) {
|
|
252
264
|
return cooldown("transport", "transport", signal);
|
|
253
265
|
}
|
|
@@ -0,0 +1,124 @@
|
|
|
1
|
+
import {
|
|
2
|
+
isContextOverflow,
|
|
3
|
+
isRetryableAssistantError,
|
|
4
|
+
type AssistantMessage,
|
|
5
|
+
} from "@earendil-works/pi-ai";
|
|
6
|
+
import { sanitizeDiagnosticText } from "./diagnostics.js";
|
|
7
|
+
|
|
8
|
+
/** Fixed fallbacks for a structured provider stop; none matches a host re-dispatch pattern. */
|
|
9
|
+
export const REFUSAL_FALLBACK_MESSAGE = "The model refused to complete the request";
|
|
10
|
+
export const UNKNOWN_STOP_FALLBACK_MESSAGE =
|
|
11
|
+
"Provider stopped with an unrecognized stop reason";
|
|
12
|
+
|
|
13
|
+
/** Upper bound on the normalized stop reason named in a reworded message. */
|
|
14
|
+
const MAX_NAMED_REASON_LENGTH = 64;
|
|
15
|
+
|
|
16
|
+
/**
|
|
17
|
+
* Whether the pinned host would re-dispatch an error terminal carrying `text`.
|
|
18
|
+
*
|
|
19
|
+
* The host re-dispatches when either retry predicate matches the prose:
|
|
20
|
+
* `AgentSession._isRetryableError` runs pi-ai `isRetryableAssistantError`, and
|
|
21
|
+
* `_checkCompaction` runs `isContextOverflow` before compacting and retrying once
|
|
22
|
+
* (pi-coding-agent dist/core/agent-session.js, both from `@earendil-works/pi-ai`).
|
|
23
|
+
* An unreadable predicate result counts as a match, so the caller keeps looking
|
|
24
|
+
* for a safer form.
|
|
25
|
+
*/
|
|
26
|
+
function hostWouldRedispatch(message: AssistantMessage, text: string): boolean {
|
|
27
|
+
try {
|
|
28
|
+
const probe = { ...message, errorMessage: text };
|
|
29
|
+
return isRetryableAssistantError(probe) || isContextOverflow(probe, 0);
|
|
30
|
+
} catch {
|
|
31
|
+
return true;
|
|
32
|
+
}
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
function namedReasonMessage(reason: string): string {
|
|
36
|
+
return `Provider stopped (reason: ${reason})`;
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
/**
|
|
40
|
+
* Normalizes a bounded, sanitized raw stop reason into plain words: every run of
|
|
41
|
+
* non-alphanumeric characters (underscores included) becomes one space, and the
|
|
42
|
+
* result is capped at a word boundary where possible.
|
|
43
|
+
*/
|
|
44
|
+
function normalizedReasonWords(rawStopReason: unknown): string[] {
|
|
45
|
+
if (typeof rawStopReason !== "string") return [];
|
|
46
|
+
const words = sanitizeDiagnosticText(rawStopReason)
|
|
47
|
+
.replace(/[^A-Za-z0-9]+/g, " ")
|
|
48
|
+
.trim()
|
|
49
|
+
.split(" ")
|
|
50
|
+
.filter((word) => word.length > 0);
|
|
51
|
+
const kept: string[] = [];
|
|
52
|
+
let length = 0;
|
|
53
|
+
for (const word of words) {
|
|
54
|
+
const next = length === 0 ? word.length : length + 1 + word.length;
|
|
55
|
+
if (next > MAX_NAMED_REASON_LENGTH) {
|
|
56
|
+
if (kept.length === 0) kept.push(word.slice(0, MAX_NAMED_REASON_LENGTH));
|
|
57
|
+
break;
|
|
58
|
+
}
|
|
59
|
+
kept.push(word);
|
|
60
|
+
length = next;
|
|
61
|
+
}
|
|
62
|
+
return kept;
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
/**
|
|
66
|
+
* Names an unknown stop reason in a message neither host predicate acts on.
|
|
67
|
+
*
|
|
68
|
+
* The first form keeps every normalized word. When that still matches a host
|
|
69
|
+
* predicate (a reason containing "overloaded" or "timeout", say), the second
|
|
70
|
+
* form admits words in order and drops each one whose addition would make the
|
|
71
|
+
* message match. Every admitted prefix was probed, so the result is host-final
|
|
72
|
+
* by construction. An empty result yields `undefined`.
|
|
73
|
+
*/
|
|
74
|
+
function unknownStopNamingReason(
|
|
75
|
+
message: AssistantMessage,
|
|
76
|
+
rawStopReason: unknown,
|
|
77
|
+
): string | undefined {
|
|
78
|
+
const words = normalizedReasonWords(rawStopReason);
|
|
79
|
+
if (words.length === 0) return undefined;
|
|
80
|
+
const reworded = namedReasonMessage(words.join(" "));
|
|
81
|
+
if (!hostWouldRedispatch(message, reworded)) return reworded;
|
|
82
|
+
const kept: string[] = [];
|
|
83
|
+
for (const word of words) {
|
|
84
|
+
const tentative = namedReasonMessage([...kept, word].join(" "));
|
|
85
|
+
if (!hostWouldRedispatch(message, tentative)) kept.push(word);
|
|
86
|
+
}
|
|
87
|
+
return kept.length === 0 ? undefined : namedReasonMessage(kept.join(" "));
|
|
88
|
+
}
|
|
89
|
+
|
|
90
|
+
/**
|
|
91
|
+
* The public error text for a structured refusal or unknown provider stop.
|
|
92
|
+
*
|
|
93
|
+
* The pinned host re-dispatches an error terminal from its prose alone (see
|
|
94
|
+
* `hostWouldRedispatch`). A refusal explanation or stop reason is
|
|
95
|
+
* provider-authored, so text such as "overloaded", "prompt is too long", or
|
|
96
|
+
* "model_context_window_exceeded" would make the host send the stopped request
|
|
97
|
+
* again. The structured code alone selects this path; the provider text is kept
|
|
98
|
+
* only when neither host predicate would act on it. An unknown stop is then
|
|
99
|
+
* reworded so it still names its reason, and only when no reworded form is safe
|
|
100
|
+
* is a fixed fallback published. A refusal whose explanation is unsafe publishes
|
|
101
|
+
* its fixed fallback directly. A message without the structured code, or that is
|
|
102
|
+
* not an error terminal, yields `undefined` and must be published unchanged.
|
|
103
|
+
*/
|
|
104
|
+
export function hostFinalStopMessage(
|
|
105
|
+
message: AssistantMessage,
|
|
106
|
+
): { readonly errorMessage: string } | undefined {
|
|
107
|
+
const code = (message as { code?: unknown }).code;
|
|
108
|
+
if (message.stopReason !== "error") return undefined;
|
|
109
|
+
if (code !== "refusal" && code !== "unknown_stop") return undefined;
|
|
110
|
+
const candidate = message.errorMessage;
|
|
111
|
+
if (
|
|
112
|
+
typeof candidate === "string" &&
|
|
113
|
+
candidate.length > 0 &&
|
|
114
|
+
!hostWouldRedispatch(message, candidate)
|
|
115
|
+
) {
|
|
116
|
+
return { errorMessage: candidate };
|
|
117
|
+
}
|
|
118
|
+
if (code === "refusal") return { errorMessage: REFUSAL_FALLBACK_MESSAGE };
|
|
119
|
+
return {
|
|
120
|
+
errorMessage:
|
|
121
|
+
unknownStopNamingReason(message, message.rawStopReason) ??
|
|
122
|
+
UNKNOWN_STOP_FALLBACK_MESSAGE,
|
|
123
|
+
};
|
|
124
|
+
}
|
package/src/index.ts
CHANGED
|
@@ -3890,6 +3890,18 @@ export const createMultiAccountExtension =
|
|
|
3890
3890
|
modelSupport,
|
|
3891
3891
|
alreadyCooled: classified.alreadyCooled === true,
|
|
3892
3892
|
});
|
|
3893
|
+
if (decision.status === "retained") {
|
|
3894
|
+
// A structured refusal or unknown stop keeps the account: no switch,
|
|
3895
|
+
// no park, no OpenRouter, and no continuation. The same context would
|
|
3896
|
+
// stop the same way on any account.
|
|
3897
|
+
diagnostics.record(
|
|
3898
|
+
"info",
|
|
3899
|
+
"routing.retained",
|
|
3900
|
+
"The provider stopped the turn with a refusal or unknown stop reason; the account was kept and no follow-up was scheduled.",
|
|
3901
|
+
{ providerId, kind: decision.classification.kind ?? "unknown" },
|
|
3902
|
+
);
|
|
3903
|
+
return;
|
|
3904
|
+
}
|
|
3893
3905
|
if (decision.status === "paused") {
|
|
3894
3906
|
// OpenRouter is an explicitly enabled, metered FINAL rung. The helper
|
|
3895
3907
|
// re-checks every managed family so family-chain policy cannot bypass
|
|
@@ -4873,7 +4885,7 @@ export const createMultiAccountExtension =
|
|
|
4873
4885
|
config,
|
|
4874
4886
|
nowMs,
|
|
4875
4887
|
});
|
|
4876
|
-
if (reactive.status
|
|
4888
|
+
if (reactive.status !== "selected") return;
|
|
4877
4889
|
reactiveCandidate = reactive;
|
|
4878
4890
|
destinationProviderId = reactive.destination.providerId;
|
|
4879
4891
|
}
|
|
@@ -5299,9 +5311,17 @@ export const createMultiAccountExtension =
|
|
|
5299
5311
|
turnRouteOrigin = origin;
|
|
5300
5312
|
handledFailures.add(event.message);
|
|
5301
5313
|
const responseFailure = lastFailure.get(providerId) ?? {};
|
|
5302
|
-
|
|
5303
|
-
|
|
5304
|
-
|
|
5314
|
+
// A structured provider stop is copied only from this two-value allowlist,
|
|
5315
|
+
// never spread, and it wins over any code parsed from provider-authored
|
|
5316
|
+
// errorMessage prose: refusal text must not read as a rate limit.
|
|
5317
|
+
const rawStopCode = (event.message as { code?: unknown }).code;
|
|
5318
|
+
const stopCode =
|
|
5319
|
+
rawStopCode === "refusal" || rawStopCode === "unknown_stop"
|
|
5320
|
+
? rawStopCode
|
|
5321
|
+
: undefined;
|
|
5322
|
+
const messageCode =
|
|
5323
|
+
stopCode ??
|
|
5324
|
+
providerErrorCodeFromMessage(event.message.errorMessage);
|
|
5305
5325
|
const failure: ProviderFailureSignal = {
|
|
5306
5326
|
...responseFailure,
|
|
5307
5327
|
...(messageCode === undefined ? {} : { code: messageCode }),
|
package/src/logical-provider.ts
CHANGED
|
@@ -22,6 +22,7 @@ import {
|
|
|
22
22
|
type SimpleStreamOptions,
|
|
23
23
|
} from "@earendil-works/pi-ai";
|
|
24
24
|
import { RuntimeState, type LogicalRoutePin } from "./runtime-state.js";
|
|
25
|
+
import { hostFinalStopMessage } from "./host-final-stop-message.js";
|
|
25
26
|
import { DEFAULT_CONFIG } from "./config.js";
|
|
26
27
|
import type {
|
|
27
28
|
AllowedFamily,
|
|
@@ -47,6 +48,16 @@ import type { ProviderType, Vendor } from "./vendor.js";
|
|
|
47
48
|
|
|
48
49
|
export { LOGICAL_PROVIDER_ID } from "./models-declaration.js";
|
|
49
50
|
import { LOGICAL_PROVIDER_ID } from "./models-declaration.js";
|
|
51
|
+
import { forceCodexSseOptions } from "./codex-adapter.js";
|
|
52
|
+
|
|
53
|
+
/** Fixed diagnostic label for a caller-supplied transport; never echoes the raw value. */
|
|
54
|
+
function codexTransportLabel(
|
|
55
|
+
transport: unknown,
|
|
56
|
+
): "websocket" | "websocket-cached" | "auto" | "other" {
|
|
57
|
+
return transport === "websocket" || transport === "websocket-cached" || transport === "auto"
|
|
58
|
+
? transport
|
|
59
|
+
: "other";
|
|
60
|
+
}
|
|
50
61
|
|
|
51
62
|
/** One physical account the logical provider may dispatch to. */
|
|
52
63
|
export interface LogicalPhysicalAccount {
|
|
@@ -666,6 +677,7 @@ export function createLogicalProvider(
|
|
|
666
677
|
deps: LogicalProviderDeps,
|
|
667
678
|
): LogicalProvider {
|
|
668
679
|
const coordinator = createHostRetryCoordinator(deps);
|
|
680
|
+
let codexTransportNoticeSent = false;
|
|
669
681
|
|
|
670
682
|
const diagnose = (message: string): void => {
|
|
671
683
|
deps.onDiagnostic?.(message);
|
|
@@ -936,6 +948,7 @@ export function createLogicalProvider(
|
|
|
936
948
|
|
|
937
949
|
const projectMessage = (message: AssistantMessage, modelId: string): AssistantMessage => ({
|
|
938
950
|
...message, api: LOGICAL_PROVIDER_ID, provider: LOGICAL_PROVIDER_ID, model: modelId,
|
|
951
|
+
...hostFinalStopMessage(message),
|
|
939
952
|
});
|
|
940
953
|
|
|
941
954
|
const projectEvent = (event: unknown, modelId: string): unknown => {
|
|
@@ -1127,8 +1140,24 @@ export function createLogicalProvider(
|
|
|
1127
1140
|
await originalOnResponse(response, responseModel);
|
|
1128
1141
|
}
|
|
1129
1142
|
};
|
|
1143
|
+
let routedOptions = options;
|
|
1144
|
+
if (account.family === "openai-codex") {
|
|
1145
|
+
// Temporary Pi 0.99 WebSocket containment; see CODEX_FORCED_TRANSPORT.
|
|
1146
|
+
const forced = forceCodexSseOptions(options);
|
|
1147
|
+
routedOptions = forced.options;
|
|
1148
|
+
if (forced.overridden && options?.transport !== undefined && !codexTransportNoticeSent) {
|
|
1149
|
+
codexTransportNoticeSent = true;
|
|
1150
|
+
try {
|
|
1151
|
+
deps.onDiagnostic?.(
|
|
1152
|
+
`Codex transport "${codexTransportLabel(options.transport)}" overridden to "sse" on routed calls (Pi 0.99 WebSocket containment).`,
|
|
1153
|
+
);
|
|
1154
|
+
} catch {
|
|
1155
|
+
// A diagnostic sink failure cannot replace a provider result.
|
|
1156
|
+
}
|
|
1157
|
+
}
|
|
1158
|
+
}
|
|
1130
1159
|
const attributedOptions: SimpleStreamOptions = {
|
|
1131
|
-
...
|
|
1160
|
+
...routedOptions,
|
|
1132
1161
|
onPayload: wrappedOnPayload,
|
|
1133
1162
|
onResponse: wrappedOnResponse,
|
|
1134
1163
|
};
|
package/src/routing.ts
CHANGED
|
@@ -98,7 +98,18 @@ export interface PausedRoute {
|
|
|
98
98
|
readonly retryAfterMs: number | null;
|
|
99
99
|
}
|
|
100
100
|
|
|
101
|
-
|
|
101
|
+
/**
|
|
102
|
+
* A structured refusal or unknown provider stop. The failed account is kept
|
|
103
|
+
* as-is: no cooldown, no invalidation, no alternative, and no continuation.
|
|
104
|
+
* The stop is deterministic for the request context, so another account would
|
|
105
|
+
* only spend another request on the same answer.
|
|
106
|
+
*/
|
|
107
|
+
export interface RetainedRoute {
|
|
108
|
+
readonly status: "retained";
|
|
109
|
+
readonly classification: FailureClassification;
|
|
110
|
+
}
|
|
111
|
+
|
|
112
|
+
export type ReactiveRouteDecision = SelectedRoute | PausedRoute | RetainedRoute;
|
|
102
113
|
|
|
103
114
|
export interface HealthSelectionDecision {
|
|
104
115
|
readonly providerId: string;
|
|
@@ -1092,6 +1103,11 @@ export function routeAfterFailure(options: {
|
|
|
1092
1103
|
throw new TypeError("failedAccount must be a canonical managed provider.");
|
|
1093
1104
|
}
|
|
1094
1105
|
const classification = classifyFailure(failure);
|
|
1106
|
+
// Before any state observation: a retained stop writes nothing and routes
|
|
1107
|
+
// nowhere, so settlement cannot turn it into a switch or a continuation.
|
|
1108
|
+
if (classification.accountAction === "retain-account") {
|
|
1109
|
+
return { status: "retained", classification };
|
|
1110
|
+
}
|
|
1095
1111
|
if (
|
|
1096
1112
|
classification.category === "config" &&
|
|
1097
1113
|
failure.code === "model_not_found" &&
|
package/src/shared-usage.ts
CHANGED
|
@@ -65,25 +65,22 @@ export const EXHAUSTION_HOLD_MS = 60 * 60_000;
|
|
|
65
65
|
* policy.
|
|
66
66
|
*/
|
|
67
67
|
const MAX_REFRESH_DEBOUNCE_MS = 60 * 60_000;
|
|
68
|
+
/** Longest persisted usage-attempt delay produced by the capped retry ladder. */
|
|
69
|
+
export const MAX_USAGE_ATTEMPT_DELAY_MS = 15 * 60_000;
|
|
68
70
|
|
|
69
71
|
export const SHARED_USAGE_MAX_BYTES = 512 * 1024;
|
|
70
72
|
const MAX_RECORD_BYTES = 4_096;
|
|
71
73
|
const MAX_OBSERVER_ID_LENGTH = 256;
|
|
72
74
|
/**
|
|
73
|
-
*
|
|
74
|
-
*
|
|
75
|
-
*
|
|
76
|
-
*
|
|
77
|
-
*
|
|
78
|
-
*
|
|
79
|
-
*
|
|
80
|
-
* identifiers.
|
|
81
|
-
*
|
|
82
|
-
* There is no companion value-length bound: unknown string VALUES are refused
|
|
83
|
-
* outright rather than length-capped, because a bounded short string is
|
|
84
|
-
* exactly the shape of a leaked bearer token.
|
|
75
|
+
* Exactly the grammar `defaultObserverId()` can emit: a hostname mapped into
|
|
76
|
+
* `[A-Za-z0-9._-]`, then at most one `:` followed only by decimal pid digits,
|
|
77
|
+
* sliced to MAX_OBSERVER_ID_LENGTH. The rule is derived from the producer
|
|
78
|
+
* rather than from a guess about what a credential looks like
|
|
79
|
+
* (internal issue #108), and it deliberately does not require the
|
|
80
|
+
* colon or the pid: on a host whose name fills the slice, the delimiter and pid
|
|
81
|
+
* are cut off, and the store must not refuse or delete those records.
|
|
85
82
|
*/
|
|
86
|
-
const
|
|
83
|
+
const OBSERVER_ID_PATTERN = /^(?=.{1,256}$)[A-Za-z0-9._-]*(?::[0-9]*)?$/;
|
|
87
84
|
|
|
88
85
|
export type UsageObservationSource = "rate-limit-header" | "usage-endpoint";
|
|
89
86
|
|
|
@@ -123,6 +120,25 @@ const USAGE_FAILURE_DETAILS = new Set<UsageFailureDetail>([
|
|
|
123
120
|
"quota-summary-error",
|
|
124
121
|
]);
|
|
125
122
|
|
|
123
|
+
/**
|
|
124
|
+
* The closed set `failureReason()` in usage-fetch.ts can produce. A persisted
|
|
125
|
+
* attempt may carry only one of these, so the field cannot hold caller text.
|
|
126
|
+
*/
|
|
127
|
+
export type SharedUsageFailureReason =
|
|
128
|
+
| "rate-limit"
|
|
129
|
+
| "server-error"
|
|
130
|
+
| "credential-unavailable"
|
|
131
|
+
| "malformed-response"
|
|
132
|
+
| "network-error";
|
|
133
|
+
|
|
134
|
+
const USAGE_FAILURE_REASONS = new Set<SharedUsageFailureReason>([
|
|
135
|
+
"rate-limit",
|
|
136
|
+
"server-error",
|
|
137
|
+
"credential-unavailable",
|
|
138
|
+
"malformed-response",
|
|
139
|
+
"network-error",
|
|
140
|
+
]);
|
|
141
|
+
|
|
126
142
|
/** Machine-global fetch-attempt state; deliberately ignored by usage aggregation. */
|
|
127
143
|
export interface SharedUsageAttemptRecord {
|
|
128
144
|
readonly recordType: "usage-attempt";
|
|
@@ -133,9 +149,10 @@ export interface SharedUsageAttemptRecord {
|
|
|
133
149
|
readonly observedAtMs: number;
|
|
134
150
|
readonly observerId: string;
|
|
135
151
|
readonly failureCount: number;
|
|
152
|
+
/** Bounded relative to `observedAtMs` by the capped retry ladder. */
|
|
136
153
|
readonly nextAttemptAtMs: number;
|
|
137
154
|
readonly disabled: boolean;
|
|
138
|
-
readonly failureReason?:
|
|
155
|
+
readonly failureReason?: SharedUsageFailureReason;
|
|
139
156
|
/** Fixed, sanitized classification detail; never an upstream error body. */
|
|
140
157
|
readonly failureDetail?: UsageFailureDetail;
|
|
141
158
|
/**
|
|
@@ -238,141 +255,9 @@ function validTimestamp(value: unknown): value is number {
|
|
|
238
255
|
return typeof value === "number" && Number.isFinite(value) && value >= 0;
|
|
239
256
|
}
|
|
240
257
|
|
|
241
|
-
/**
|
|
242
|
-
* Whether a record this build cannot interpret may survive compaction.
|
|
243
|
-
*
|
|
244
|
-
* Compaction rebuilds the file from recognised records, so anything not carried
|
|
245
|
-
* here is deleted. That is how an older build destroys a record type added
|
|
246
|
-
* after it. Carrying unknown records forward keeps a newer build's state alive
|
|
247
|
-
* across a mixed-version fleet.
|
|
248
|
-
*
|
|
249
|
-
* The filter exists because "unrecognised" also covers records this build
|
|
250
|
-
* rejected as malformed, including the credential-bearing ones that
|
|
251
|
-
* `append` refuses and compaction currently scrubs. Carrying those would turn a
|
|
252
|
-
* durability fix into a privacy regression, so a carried record must still look
|
|
253
|
-
* like a usage record for a known account: a `recordType` string this build
|
|
254
|
-
* does not know, the same account identity fields every record carries, and no
|
|
255
|
-
* field outside that shape. A future record type satisfies this; a leaked
|
|
256
|
-
* authorization header does not.
|
|
257
|
-
*/
|
|
258
|
-
function carryableUnknownRecord(value: unknown): boolean {
|
|
259
|
-
if (typeof value !== "object" || value === null || Array.isArray(value))
|
|
260
|
-
return false;
|
|
261
|
-
const record = value as Record<string, unknown>;
|
|
262
|
-
// `recordType` must look like a record type, not merely be non-empty.
|
|
263
|
-
//
|
|
264
|
-
// Round 2 proved "non-empty string" is not a constraint: `Bearer <token>`
|
|
265
|
-
// is a non-empty string, so a credential placed here survived compaction.
|
|
266
|
-
// A real record type is a lower-kebab identifier, and nothing that fails
|
|
267
|
-
// this shape is a record type this build should carry blind.
|
|
268
|
-
if (
|
|
269
|
-
typeof record.recordType !== "string" ||
|
|
270
|
-
!CARRYABLE_RECORD_TYPE.test(record.recordType)
|
|
271
|
-
)
|
|
272
|
-
return false;
|
|
273
|
-
if (
|
|
274
|
-
typeof record.providerId !== "string" ||
|
|
275
|
-
typeof record.family !== "string" ||
|
|
276
|
-
!isAllowedFamily(record.family) ||
|
|
277
|
-
!isCanonicalManagedProviderId(record.providerId, record.family) ||
|
|
278
|
-
!validTimestamp(record.observedAtMs) ||
|
|
279
|
-
!validObserverId(record.observerId) ||
|
|
280
|
-
// Stricter than `validObserverId` on purpose: that only bounds length,
|
|
281
|
-
// and a bearer token is a bounded string. See CARRYABLE_OBSERVER_ID.
|
|
282
|
-
!CARRYABLE_OBSERVER_ID.test(record.observerId)
|
|
283
|
-
)
|
|
284
|
-
return false;
|
|
285
|
-
// Beyond the identity fields above, a carried record may hold only
|
|
286
|
-
// NON-STRING values.
|
|
287
|
-
//
|
|
288
|
-
// The first version of this filter bounded value shape -- primitives only,
|
|
289
|
-
// length-capped -- and review proved it unsound: `{ authorization: "Bearer
|
|
290
|
-
// SHORT" }` is a bounded primitive and survived compaction, which is exactly
|
|
291
|
-
// the privacy regression the carry-through was not allowed to create.
|
|
292
|
-
//
|
|
293
|
-
// Refusing unknown strings outright is the only defensible rule here. A
|
|
294
|
-
// denylist of sensitive-looking field names would be a guess about what a
|
|
295
|
-
// future record type calls its fields, and every credential this project
|
|
296
|
-
// handles is a string. Timestamps, counts, fractions and flags -- what a
|
|
297
|
-
// forward-compatible usage record actually needs -- are unaffected. A future
|
|
298
|
-
// type that genuinely needs a string field must teach this build about
|
|
299
|
-
// itself rather than rely on being carried blind.
|
|
300
|
-
for (const [key, entry] of Object.entries(record)) {
|
|
301
|
-
// Field NAMES are constrained by shape, not only length. Round 3 found
|
|
302
|
-
// a length bound alone lets a key called `Bearer SECRET` through: the
|
|
303
|
-
// credential rides in the key rather than the value. A field name in a
|
|
304
|
-
// JSON record written by this project is a lower-camel identifier.
|
|
305
|
-
if (!CARRYABLE_FIELD_NAME.test(key)) return false;
|
|
306
|
-
if (CARRYABLE_IDENTITY_FIELDS.has(key)) continue;
|
|
307
|
-
if (!carryableUnknownValue(entry)) return false;
|
|
308
|
-
}
|
|
309
|
-
return true;
|
|
310
|
-
}
|
|
311
|
-
|
|
312
|
-
/**
|
|
313
|
-
* Record types compaction may carry forward: the project's own namespace.
|
|
314
|
-
*
|
|
315
|
-
* Two weaker rules were tried and both refuted. "Non-empty" fell to
|
|
316
|
-
* `Bearer <token>` in round 2. A lower-kebab shape fell in round 3 to
|
|
317
|
-
* `sk-ant-api03-deadbeef`, which IS lower-kebab -- an API key and a record
|
|
318
|
-
* type are not distinguishable by shape, so no amount of character-class
|
|
319
|
-
* tightening can separate them.
|
|
320
|
-
*
|
|
321
|
-
* A namespace can. Every record type this file writes is `usage-`-prefixed
|
|
322
|
-
* (`usage-attempt`, `usage-exhaustion-hold`), so a future type from a newer
|
|
323
|
-
* build will be too. That is a property of the writer rather than a guess
|
|
324
|
-
* about what a credential looks like, which is why it holds where the shape
|
|
325
|
-
* checks did not.
|
|
326
|
-
*/
|
|
327
|
-
const CARRYABLE_RECORD_TYPE = /^usage-[a-z][a-z0-9-]{0,56}$/;
|
|
328
|
-
|
|
329
|
-
/**
|
|
330
|
-
* Shape a carried record's `observerId` must have.
|
|
331
|
-
*
|
|
332
|
-
* `validObserverId` only bounds length, and round 2 proved that insufficient:
|
|
333
|
-
* a bearer token is a bounded string. This restricts the CHARACTER SET instead,
|
|
334
|
-
* which is what makes the field unusable for smuggling while still accepting
|
|
335
|
-
* everything the producer can emit.
|
|
336
|
-
*
|
|
337
|
-
* The colon is deliberately NOT required. `defaultObserverId` builds
|
|
338
|
-
* `${hostname()}:${process.pid}` and then truncates to MAX_OBSERVER_ID_LENGTH,
|
|
339
|
-
* so on a host with a very long name the pid -- and the colon with it -- is cut
|
|
340
|
-
* off entirely. Round 3 found an earlier version of this expression required
|
|
341
|
-
* the colon, which would have made compaction DELETE legitimate records on such
|
|
342
|
-
* a machine: the precise data loss this carry-through exists to prevent, caused
|
|
343
|
-
* by the fix for it. A verified probe produced a 256-character id with no colon
|
|
344
|
-
* at all.
|
|
345
|
-
*
|
|
346
|
-
* The length bound matches MAX_OBSERVER_ID_LENGTH rather than guessing a
|
|
347
|
-
* narrower one, so the accepted domain covers every value the producer can
|
|
348
|
-
* actually return.
|
|
349
|
-
*/
|
|
350
|
-
const CARRYABLE_OBSERVER_ID = /^[A-Za-z0-9._:-]{1,256}$/;
|
|
351
|
-
|
|
352
|
-
/**
|
|
353
|
-
* Identity fields every record carries. They are the only strings a carried
|
|
354
|
-
* unknown record may contain, and each is format-checked above rather than
|
|
355
|
-
* merely bounded.
|
|
356
|
-
*/
|
|
357
|
-
const CARRYABLE_IDENTITY_FIELDS = new Set([
|
|
358
|
-
"recordType",
|
|
359
|
-
"providerId",
|
|
360
|
-
"family",
|
|
361
|
-
"observerId",
|
|
362
|
-
]);
|
|
363
|
-
|
|
364
|
-
function carryableUnknownValue(value: unknown): boolean {
|
|
365
|
-
return (
|
|
366
|
-
value === null || typeof value === "boolean" || typeof value === "number"
|
|
367
|
-
);
|
|
368
|
-
}
|
|
369
258
|
|
|
370
259
|
function validObserverId(value: unknown): value is string {
|
|
371
|
-
return (
|
|
372
|
-
typeof value === "string" &&
|
|
373
|
-
value.length > 0 &&
|
|
374
|
-
value.length <= MAX_OBSERVER_ID_LENGTH
|
|
375
|
-
);
|
|
260
|
+
return typeof value === "string" && OBSERVER_ID_PATTERN.test(value);
|
|
376
261
|
}
|
|
377
262
|
|
|
378
263
|
const TOKEN_FIELDS = [
|
|
@@ -474,10 +359,15 @@ function validAttemptRecord(value: unknown): value is SharedUsageAttemptRecord {
|
|
|
474
359
|
validObserverId(record.observerId) &&
|
|
475
360
|
nonNegativeInteger(record.failureCount) &&
|
|
476
361
|
validTimestamp(record.nextAttemptAtMs) &&
|
|
362
|
+
record.nextAttemptAtMs >= record.observedAtMs &&
|
|
363
|
+
record.nextAttemptAtMs - record.observedAtMs <=
|
|
364
|
+
MAX_USAGE_ATTEMPT_DELAY_MS &&
|
|
477
365
|
typeof record.disabled === "boolean" &&
|
|
478
366
|
(record.failureReason === undefined ||
|
|
479
367
|
(typeof record.failureReason === "string" &&
|
|
480
|
-
|
|
368
|
+
USAGE_FAILURE_REASONS.has(
|
|
369
|
+
record.failureReason as SharedUsageFailureReason,
|
|
370
|
+
))) &&
|
|
481
371
|
(record.failureDetail === undefined ||
|
|
482
372
|
(typeof record.failureDetail === "string" &&
|
|
483
373
|
USAGE_FAILURE_DETAILS.has(record.failureDetail as UsageFailureDetail))) &&
|
|
@@ -551,9 +441,19 @@ function defaultStorePath(): string {
|
|
|
551
441
|
return join(agentDir, "pi-multi-account", "usage.ndjson");
|
|
552
442
|
}
|
|
553
443
|
|
|
554
|
-
/**
|
|
555
|
-
|
|
556
|
-
|
|
444
|
+
/**
|
|
445
|
+
* Identity is bounded metadata only; it contains no credential-derived value.
|
|
446
|
+
*
|
|
447
|
+
* Hostname characters outside the observer-id alphabet are mapped to `-`, so
|
|
448
|
+
* every value this producer returns is one `validObserverId` accepts. Without
|
|
449
|
+
* that, an unusual hostname would make the default store constructor throw.
|
|
450
|
+
*/
|
|
451
|
+
export function defaultObserverId(
|
|
452
|
+
host: string = hostname(),
|
|
453
|
+
pid: number = process.pid,
|
|
454
|
+
): string {
|
|
455
|
+
const safeHost = host.replace(/[^A-Za-z0-9._-]/g, "-");
|
|
456
|
+
return `${safeHost}:${pid}`.slice(0, MAX_OBSERVER_ID_LENGTH);
|
|
557
457
|
}
|
|
558
458
|
|
|
559
459
|
function projectRecord(record: SharedUsageRecord): SharedUsageRecord {
|
|
@@ -829,24 +729,26 @@ function compactUsageFile(options: {
|
|
|
829
729
|
const latestRateLimits = new Map<string, SharedUsageRecord>();
|
|
830
730
|
const latestAttempts = new Map<string, SharedUsageAttemptRecord>();
|
|
831
731
|
const latestHolds = new Map<string, SharedUsageExhaustionHoldRecord>();
|
|
832
|
-
//
|
|
833
|
-
//
|
|
834
|
-
//
|
|
835
|
-
// process running an older build silently deletes every record type added
|
|
836
|
-
// after it -- including the exhaustion holds that keep a spent account out
|
|
837
|
-
// of rotation. The account then looks healthy
|
|
838
|
-
// to the next process and gets routed to again.
|
|
732
|
+
// Compaction is a CLOSED REGISTRY: only the three record types this
|
|
733
|
+
// build validates survive, and each survives as the JSON of its own
|
|
734
|
+
// validated projection, never as the source bytes.
|
|
839
735
|
//
|
|
840
|
-
//
|
|
841
|
-
//
|
|
842
|
-
//
|
|
736
|
+
// internal MR !83 carried unrecognised records verbatim so an
|
|
737
|
+
// older build would not delete a newer build's record types. Four rounds
|
|
738
|
+
// of shape rules (non-empty, lower-kebab, `usage-` namespace, lower-camel
|
|
739
|
+
// field names) tried to tell a future record from credential text and
|
|
740
|
+
// all were refuted (internal issue #110): `sk-ant-api03-deadbeef`
|
|
741
|
+
// is indistinguishable by shape from an identifier, a `usage-` prefix is
|
|
742
|
+
// free to any writer, and a retained source line keeps every field name
|
|
743
|
+
// and raw numeric text too. A record this build cannot interpret cannot
|
|
744
|
+
// be proved credential-free, so it is dropped.
|
|
843
745
|
//
|
|
844
|
-
//
|
|
845
|
-
//
|
|
846
|
-
//
|
|
847
|
-
//
|
|
848
|
-
//
|
|
849
|
-
|
|
746
|
+
// Mixed-version safety is kept for every type this build knows, the
|
|
747
|
+
// exhaustion hold included: each is re-emitted from its validated fields,
|
|
748
|
+
// so a peer on this build never deletes a live hold. The accepted cost
|
|
749
|
+
// is that a record type added after this build is dropped when this
|
|
750
|
+
// build compacts; a new type must stay advisory until every build in the
|
|
751
|
+
// fleet recognises it.
|
|
850
752
|
|
|
851
753
|
for (const line of completeUsageLines(options.path, completePrefixBytes)) {
|
|
852
754
|
let parsed: unknown;
|
|
@@ -893,10 +795,7 @@ function compactUsageFile(options: {
|
|
|
893
795
|
if (!previous || hold.observedAtMs >= previous.observedAtMs) {
|
|
894
796
|
latestHolds.set(hold.providerId, hold);
|
|
895
797
|
}
|
|
896
|
-
} else if (carryableUnknownRecord(parsed)) {
|
|
897
|
-
unrecognisedLines.push(line);
|
|
898
798
|
}
|
|
899
|
-
|
|
900
799
|
}
|
|
901
800
|
const compactedRecords = new Set([
|
|
902
801
|
...latest.values(),
|
|
@@ -905,12 +804,9 @@ function compactUsageFile(options: {
|
|
|
905
804
|
...latestAttempts.values(),
|
|
906
805
|
...latestHolds.values(),
|
|
907
806
|
]);
|
|
908
|
-
|
|
909
|
-
|
|
910
|
-
|
|
911
|
-
...unrecognisedLines.map((line) => `${line}\n`),
|
|
912
|
-
...[...compactedRecords].map((record) => `${JSON.stringify(record)}\n`),
|
|
913
|
-
].join("");
|
|
807
|
+
const compacted = [...compactedRecords]
|
|
808
|
+
.map((record) => `${JSON.stringify(record)}\n`)
|
|
809
|
+
.join("");
|
|
914
810
|
const compactedBytes = Buffer.byteLength(compacted, "utf8");
|
|
915
811
|
if (compactedBytes > options.maxBytes) return false;
|
|
916
812
|
|
|
@@ -1009,6 +905,11 @@ export class SharedUsageStore {
|
|
|
1009
905
|
|
|
1010
906
|
append(record: SharedUsageLogRecord): boolean {
|
|
1011
907
|
try {
|
|
908
|
+
// Judge the caller's own value, not the projection: projecting slices
|
|
909
|
+
// the observer id, and a value this store would have to rewrite is
|
|
910
|
+
// not one the producer emitted.
|
|
911
|
+
if (!validObserverId((record as { observerId?: unknown }).observerId))
|
|
912
|
+
return false;
|
|
1012
913
|
const projected = validExhaustionHoldRecord(record)
|
|
1013
914
|
? projectExhaustionHold(record)
|
|
1014
915
|
: validAttemptRecord(record)
|
|
@@ -1134,16 +1035,22 @@ export class SharedUsageStore {
|
|
|
1134
1035
|
* records, so a hold that has lapsed simply stops being reported. Callers get
|
|
1135
1036
|
* a time to compare, not a boolean, because routing has to combine it with an
|
|
1136
1037
|
* authoritative recovery time and take whichever is later.
|
|
1038
|
+
*
|
|
1039
|
+
* `clearedAtMs` excludes holds whose own validated `failedAtMs` is at or
|
|
1040
|
+
* before it, so a caller's operator clear is judged against the failure
|
|
1041
|
+
* the record states rather than one reconstructed from its deadline.
|
|
1137
1042
|
*/
|
|
1138
1043
|
activeExhaustionHoldUntilMs(
|
|
1139
1044
|
providerId: string,
|
|
1140
1045
|
family: AllowedFamily,
|
|
1141
1046
|
nowMs: number,
|
|
1047
|
+
clearedAtMs?: number,
|
|
1142
1048
|
): number | undefined {
|
|
1143
1049
|
let latest: number | undefined;
|
|
1144
1050
|
for (const hold of this.readExhaustionHolds()) {
|
|
1145
1051
|
if (hold.providerId !== providerId || hold.family !== family) continue;
|
|
1146
1052
|
if (hold.holdUntilMs <= nowMs) continue;
|
|
1053
|
+
if (clearedAtMs !== undefined && hold.failedAtMs <= clearedAtMs) continue;
|
|
1147
1054
|
// A relative bound alone is not enough. `holdUntilMs - failedAtMs` can
|
|
1148
1055
|
// be a legitimate 60 minutes while `failedAtMs` itself sits in the year
|
|
1149
1056
|
// 3138, which would exclude the account for centuries. No honest hold
|
package/src/usage-fetch.ts
CHANGED
|
@@ -6,8 +6,10 @@ import {
|
|
|
6
6
|
type MachineLeaseHandle,
|
|
7
7
|
} from "./machine-lease.js";
|
|
8
8
|
import {
|
|
9
|
+
MAX_USAGE_ATTEMPT_DELAY_MS,
|
|
9
10
|
normalizeUsageEndpointPercent,
|
|
10
11
|
type SharedUsageAttemptRecord,
|
|
12
|
+
type SharedUsageFailureReason,
|
|
11
13
|
type SharedUsageStore,
|
|
12
14
|
type UsageFailureDetail,
|
|
13
15
|
} from "./shared-usage.js";
|
|
@@ -39,9 +41,8 @@ export const USAGE_FETCH_TIMEOUT_MS = 10_000;
|
|
|
39
41
|
const MAX_RESPONSE_BYTES = 128 * 1024;
|
|
40
42
|
const MAX_ERROR_CHARS = 512;
|
|
41
43
|
const BASE_BACKOFF_MS = 30_000;
|
|
42
|
-
const MAX_BACKOFF_MS = 15 * 60_000;
|
|
43
44
|
const BACKOFF_LADDER_STEPS =
|
|
44
|
-
Math.ceil(Math.log2(
|
|
45
|
+
Math.ceil(Math.log2(MAX_USAGE_ATTEMPT_DELAY_MS / BASE_BACKOFF_MS)) + 1;
|
|
45
46
|
/** Disable only after the capped rung has failed once more. */
|
|
46
47
|
export const USAGE_FETCH_DISABLE_AFTER_FAILURES = BACKOFF_LADDER_STEPS + 1;
|
|
47
48
|
|
|
@@ -104,12 +105,7 @@ export interface UsageFetchStatus {
|
|
|
104
105
|
readonly disabled: boolean;
|
|
105
106
|
readonly failureCount: number;
|
|
106
107
|
readonly nextAttemptAtMs?: number;
|
|
107
|
-
readonly disabledReason?:
|
|
108
|
-
| "rate-limit"
|
|
109
|
-
| "server-error"
|
|
110
|
-
| "credential-unavailable"
|
|
111
|
-
| "malformed-response"
|
|
112
|
-
| "network-error";
|
|
108
|
+
readonly disabledReason?: SharedUsageFailureReason;
|
|
113
109
|
}
|
|
114
110
|
|
|
115
111
|
export type UsageFetchResultStatus =
|
|
@@ -217,7 +213,10 @@ function retryAfterMs(response: UsageFetchResponse): number | undefined {
|
|
|
217
213
|
}
|
|
218
214
|
const seconds = Number(value);
|
|
219
215
|
if (!Number.isFinite(seconds) || seconds < 0) return undefined;
|
|
220
|
-
return Math.min(
|
|
216
|
+
return Math.min(
|
|
217
|
+
MAX_USAGE_ATTEMPT_DELAY_MS,
|
|
218
|
+
Math.max(0, Math.ceil(seconds * 1_000)),
|
|
219
|
+
);
|
|
221
220
|
}
|
|
222
221
|
|
|
223
222
|
/** Redacts bearer and access-token values before an error can escape this module. */
|
|
@@ -642,14 +641,37 @@ async function queryEndpoint(
|
|
|
642
641
|
: normalizeAnthropicUsagePayload(payload);
|
|
643
642
|
}
|
|
644
643
|
|
|
644
|
+
/**
|
|
645
|
+
* The attempt's deadline, or undefined when no honest ladder could have set it.
|
|
646
|
+
*
|
|
647
|
+
* The store already bounds `nextAttemptAtMs` against the record's own
|
|
648
|
+
* `observedAtMs` on append and read (internal issue #108). That span
|
|
649
|
+
* check alone is not enough for a reader: a record whose ORIGIN sits centuries
|
|
650
|
+
* ahead carries a legitimate-looking span and would still suppress polling for
|
|
651
|
+
* that account forever. No honest deadline lies further than one capped rung
|
|
652
|
+
* from now, so anything beyond that is ignored rather than trusted.
|
|
653
|
+
*/
|
|
654
|
+
function plausibleAttemptDeadline(
|
|
655
|
+
attempt: SharedUsageAttemptRecord | undefined,
|
|
656
|
+
nowMs: number,
|
|
657
|
+
): number | undefined {
|
|
658
|
+
if (attempt === undefined) return undefined;
|
|
659
|
+
if (attempt.nextAttemptAtMs > nowMs + MAX_USAGE_ATTEMPT_DELAY_MS) {
|
|
660
|
+
return undefined;
|
|
661
|
+
}
|
|
662
|
+
return attempt.nextAttemptAtMs;
|
|
663
|
+
}
|
|
664
|
+
|
|
645
665
|
function statusFromAttempt(
|
|
646
666
|
attempt: SharedUsageAttemptRecord | undefined,
|
|
647
667
|
enabled: boolean,
|
|
668
|
+
nowMs: number,
|
|
648
669
|
): UsageFetchStatus {
|
|
649
670
|
const nextAttemptAtMs =
|
|
650
|
-
attempt?.failureCount === 0
|
|
651
|
-
|
|
652
|
-
|
|
671
|
+
attempt?.failureCount === 0
|
|
672
|
+
? undefined
|
|
673
|
+
: plausibleAttemptDeadline(attempt, nowMs);
|
|
674
|
+
const disabledReason = attempt?.failureReason;
|
|
653
675
|
return {
|
|
654
676
|
enabled,
|
|
655
677
|
disabled: attempt?.disabled ?? false,
|
|
@@ -833,6 +855,7 @@ export class UsageFetcher {
|
|
|
833
855
|
return statusFromAttempt(
|
|
834
856
|
this.#sharedStore.latestAttempt(providerId, family),
|
|
835
857
|
enabled,
|
|
858
|
+
this.#now(),
|
|
836
859
|
);
|
|
837
860
|
}
|
|
838
861
|
|
|
@@ -997,11 +1020,9 @@ export class UsageFetcher {
|
|
|
997
1020
|
// forever. The bound belongs here, on the record, before any clause
|
|
998
1021
|
// reads it.
|
|
999
1022
|
//
|
|
1000
|
-
// `nextAttemptAtMs` is
|
|
1001
|
-
//
|
|
1002
|
-
//
|
|
1003
|
-
// MAX_BACKOFF_MS, and bounding that would break genuine backoff
|
|
1004
|
-
// suppression. An honest record cannot have been OBSERVED in the future.
|
|
1023
|
+
// `nextAttemptAtMs` is bounded relative to `observedAtMs` by the store and
|
|
1024
|
+
// checked again here before it may suppress a send. An honest record cannot
|
|
1025
|
+
// have been OBSERVED in the future either.
|
|
1005
1026
|
if (
|
|
1006
1027
|
attempt.observedAtMs > nowMs ||
|
|
1007
1028
|
(attempt.failureTriggeredAtMs !== undefined &&
|
|
@@ -1009,7 +1030,12 @@ export class UsageFetcher {
|
|
|
1009
1030
|
) {
|
|
1010
1031
|
return undefined;
|
|
1011
1032
|
}
|
|
1012
|
-
|
|
1033
|
+
const nextAttemptAtMs = plausibleAttemptDeadline(attempt, nowMs);
|
|
1034
|
+
if (
|
|
1035
|
+
attempt.failureCount > 0 &&
|
|
1036
|
+
nextAttemptAtMs !== undefined &&
|
|
1037
|
+
nextAttemptAtMs > nowMs
|
|
1038
|
+
) {
|
|
1013
1039
|
return "backoff";
|
|
1014
1040
|
}
|
|
1015
1041
|
if (attempt.observedAtMs >= failedAtMs) return "already-answered";
|
|
@@ -1190,8 +1216,13 @@ export class UsageFetcher {
|
|
|
1190
1216
|
// Once the recorded backoff has elapsed, the account is retried whether or
|
|
1191
1217
|
// not it was disabled. A retry that fails again simply re-arms the ladder at
|
|
1192
1218
|
// its capped rung, so a genuinely dead endpoint is polled at most once per
|
|
1193
|
-
//
|
|
1194
|
-
|
|
1219
|
+
// MAX_USAGE_ATTEMPT_DELAY_MS rather than hammered.
|
|
1220
|
+
const priorDeadline = plausibleAttemptDeadline(prior, nowMs);
|
|
1221
|
+
if (
|
|
1222
|
+
prior !== undefined &&
|
|
1223
|
+
priorDeadline !== undefined &&
|
|
1224
|
+
priorDeadline > nowMs
|
|
1225
|
+
) {
|
|
1195
1226
|
await this.#persistWindowGap(
|
|
1196
1227
|
account,
|
|
1197
1228
|
nowMs,
|
|
@@ -1223,7 +1254,13 @@ export class UsageFetcher {
|
|
|
1223
1254
|
// over a stale `disabled` flag, so a recovered account can be observed
|
|
1224
1255
|
// again. Re-read under the lease because a peer may have attempted in
|
|
1225
1256
|
// between.
|
|
1226
|
-
|
|
1257
|
+
const currentNowMs = this.#now();
|
|
1258
|
+
const currentDeadline = plausibleAttemptDeadline(current, currentNowMs);
|
|
1259
|
+
if (
|
|
1260
|
+
current !== undefined &&
|
|
1261
|
+
currentDeadline !== undefined &&
|
|
1262
|
+
currentDeadline > currentNowMs
|
|
1263
|
+
) {
|
|
1227
1264
|
await this.#persistWindowGap(
|
|
1228
1265
|
account,
|
|
1229
1266
|
this.#now(),
|
|
@@ -1555,7 +1592,7 @@ export class UsageFetcher {
|
|
|
1555
1592
|
: new UsageEndpointError("usage endpoint failed", "network-error");
|
|
1556
1593
|
const failureCount = (prior?.failureCount ?? 0) + 1;
|
|
1557
1594
|
const backoff = Math.min(
|
|
1558
|
-
|
|
1595
|
+
MAX_USAGE_ATTEMPT_DELAY_MS,
|
|
1559
1596
|
BASE_BACKOFF_MS * 2 ** Math.max(0, failureCount - 1),
|
|
1560
1597
|
);
|
|
1561
1598
|
const disabled = failureCount >= USAGE_FETCH_DISABLE_AFTER_FAILURES;
|
package/src/usage.ts
CHANGED
|
@@ -337,8 +337,8 @@ export class UsageLedger {
|
|
|
337
337
|
* exhaustion hold is one. But a hold is machine-global: deleting the record
|
|
338
338
|
* would revoke it for every other session on the machine, and those peers
|
|
339
339
|
* may still be refusing work on that account. So the override is
|
|
340
|
-
* PROCESS-LOCAL -- this session stops honouring holds
|
|
341
|
-
*
|
|
340
|
+
* PROCESS-LOCAL -- this session stops honouring holds whose failure happened
|
|
341
|
+
* at or before the clear, and peers keep theirs.
|
|
342
342
|
*
|
|
343
343
|
* Stored as the clear time rather than a boolean so a LATER failure still
|
|
344
344
|
* installs a hold this session honours. A boolean would make one `clear`
|
|
@@ -346,6 +346,13 @@ export class UsageLedger {
|
|
|
346
346
|
* which is a permanent effect from a transient instruction.
|
|
347
347
|
*/
|
|
348
348
|
readonly #holdOverrides = new Map<string, number>();
|
|
349
|
+
/**
|
|
350
|
+
* When `clear()` with no account last ran. Clear-all is one point in time
|
|
351
|
+
* for EVERY account, including ones with no hold visible at that instant:
|
|
352
|
+
* enumerating the holds present then let a peer publish a pre-clear hold a
|
|
353
|
+
* moment later and still have it honoured (internal issue #110).
|
|
354
|
+
*/
|
|
355
|
+
#clearAllAtMs: number | undefined;
|
|
349
356
|
readonly #sharedStore: SharedUsageStore | undefined;
|
|
350
357
|
readonly #now: () => number;
|
|
351
358
|
|
|
@@ -714,29 +721,29 @@ export class UsageLedger {
|
|
|
714
721
|
family: AllowedFamily,
|
|
715
722
|
nowMs = this.#now(),
|
|
716
723
|
): number | undefined {
|
|
717
|
-
|
|
724
|
+
// An operator `clear` in THIS process stops us honouring holds whose
|
|
725
|
+
// failure happened at or before it. A hold from a LATER failure is
|
|
726
|
+
// honoured again: the override is a point in time, not a permanent
|
|
727
|
+
// exemption. Peers keep honouring the record either way, because it is
|
|
728
|
+
// not deleted.
|
|
729
|
+
//
|
|
730
|
+
// The store compares each hold's own validated `failedAtMs`. Deriving
|
|
731
|
+
// the failure time from the deadline (`holdUntilMs - EXHAUSTION_HOLD_MS`)
|
|
732
|
+
// is only right for an exact one-hour hold, and the store accepts every
|
|
733
|
+
// shorter span (internal issue #110).
|
|
734
|
+
const accountClearedAtMs = this.#holdOverrides.get(providerId);
|
|
735
|
+
const clearedAtMs =
|
|
736
|
+
accountClearedAtMs === undefined
|
|
737
|
+
? this.#clearAllAtMs
|
|
738
|
+
: this.#clearAllAtMs === undefined
|
|
739
|
+
? accountClearedAtMs
|
|
740
|
+
: Math.max(accountClearedAtMs, this.#clearAllAtMs);
|
|
741
|
+
return this.#sharedStore?.activeExhaustionHoldUntilMs(
|
|
718
742
|
providerId,
|
|
719
743
|
family,
|
|
720
744
|
nowMs,
|
|
745
|
+
clearedAtMs,
|
|
721
746
|
);
|
|
722
|
-
if (holdUntilMs === undefined) return undefined;
|
|
723
|
-
// An operator `clear` in THIS process stops us honouring holds that were
|
|
724
|
-
// already installed when it ran. A hold from a LATER failure is honoured
|
|
725
|
-
// again: the override is a point in time, not a permanent exemption.
|
|
726
|
-
// Peers keep honouring the record either way, because it is not deleted.
|
|
727
|
-
//
|
|
728
|
-
// `holdUntilMs - EXHAUSTION_HOLD_MS` recovers the failure time from the
|
|
729
|
-
// reported deadline without a second store read. The writer always sets
|
|
730
|
-
// exactly that span (asserted by the hold-duration test), and the
|
|
731
|
-
// store's own read bound refuses any record claiming more.
|
|
732
|
-
const clearedAtMs = this.#holdOverrides.get(providerId);
|
|
733
|
-
if (
|
|
734
|
-
clearedAtMs !== undefined &&
|
|
735
|
-
holdUntilMs - EXHAUSTION_HOLD_MS <= clearedAtMs
|
|
736
|
-
) {
|
|
737
|
-
return undefined;
|
|
738
|
-
}
|
|
739
|
-
return holdUntilMs;
|
|
740
747
|
}
|
|
741
748
|
|
|
742
749
|
/**
|
|
@@ -811,11 +818,7 @@ export class UsageLedger {
|
|
|
811
818
|
if (providerId === undefined) {
|
|
812
819
|
this.#snapshots.clear();
|
|
813
820
|
this.#quotaObservations.clear();
|
|
814
|
-
|
|
815
|
-
?.readExhaustionHolds()
|
|
816
|
-
.map((hold) => hold.providerId) ?? []) {
|
|
817
|
-
this.#holdOverrides.set(id, nowMs);
|
|
818
|
-
}
|
|
821
|
+
this.#clearAllAtMs = Math.max(this.#clearAllAtMs ?? nowMs, nowMs);
|
|
819
822
|
} else {
|
|
820
823
|
this.#snapshots.delete(providerId);
|
|
821
824
|
this.#quotaObservations.delete(providerId);
|