@oh-my-pi/pi-ai 18.3.0 → 18.3.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +20 -0
- package/THIRD-PARTY-NOTICES.txt +2 -2
- package/dist/types/auth/policy.d.ts +8 -1
- package/dist/types/auth/pool.d.ts +8 -0
- package/dist/types/auth/types.d.ts +17 -7
- package/dist/types/auth/usage.d.ts +2 -0
- package/dist/types/auth-broker/discover.d.ts +22 -1
- package/dist/types/auth-storage.d.ts +35 -11
- package/dist/types/error/rate-limit.d.ts +3 -2
- package/dist/types/index.d.ts +1 -0
- package/dist/types/providers/anthropic-slow-mode.d.ts +109 -0
- package/dist/types/providers/anthropic-wire.d.ts +4 -0
- package/dist/types/providers/mock.d.ts +3 -1
- package/dist/types/providers/openai-codex/live-steering.d.ts +77 -0
- package/dist/types/providers/transform-messages.d.ts +9 -1
- package/dist/types/types.d.ts +56 -0
- package/dist/types/usage/xai-oauth.d.ts +6 -1
- package/dist/types/utils/http-inspector.d.ts +6 -0
- package/package.json +6 -6
- package/src/auth/policy.ts +27 -6
- package/src/auth/pool.ts +35 -2
- package/src/auth/refresh.ts +2 -2
- package/src/auth/types.ts +17 -7
- package/src/auth/usage.ts +5 -0
- package/src/auth-broker/discover.ts +57 -27
- package/src/auth-storage.ts +121 -35
- package/src/error/flags.ts +2 -1
- package/src/error/rate-limit.ts +8 -4
- package/src/index.ts +1 -0
- package/src/providers/anthropic-slow-mode.ts +232 -0
- package/src/providers/anthropic-wire.ts +4 -0
- package/src/providers/anthropic.ts +322 -22
- package/src/providers/cowork-fetch.ts +11 -4
- package/src/providers/google-shared.ts +30 -7
- package/src/providers/inference-headers.ts +7 -1
- package/src/providers/mock.ts +4 -0
- package/src/providers/openai-codex/live-steering.ts +237 -0
- package/src/providers/openai-codex-responses.ts +372 -73
- package/src/providers/transform-messages.ts +27 -7
- package/src/stream.ts +32 -2
- package/src/types.ts +59 -0
- package/src/usage/registry.ts +2 -1
- package/src/usage/xai-oauth.ts +31 -1
- package/src/utils/http-inspector.ts +21 -2
- package/src/utils/openrouter-headers.ts +3 -3
|
@@ -77,6 +77,11 @@ import { AssistantMessageEventStream } from "../utils/event-stream";
|
|
|
77
77
|
import { isFoundryEnabled } from "../utils/foundry";
|
|
78
78
|
import { finalizeErrorMessage, type RawHttpRequestDump } from "../utils/http-inspector";
|
|
79
79
|
import { getStreamFirstEventTimeoutMs, getStreamIdleTimeoutMs, iterateWithIdleTimeout } from "../utils/idle-iterator";
|
|
80
|
+
import {
|
|
81
|
+
ANTHROPIC_SLOW_USAGE_LIMIT,
|
|
82
|
+
ANTHROPIC_USAGE_LIMIT_HEADER,
|
|
83
|
+
parseAnthropicSlowModeHeaders,
|
|
84
|
+
} from "./anthropic-slow-mode";
|
|
80
85
|
import { notifyProviderResponse } from "../utils/provider-response";
|
|
81
86
|
import { getHeadersFromError, getRetryAfterMsFromHeaders } from "../utils/retry-after";
|
|
82
87
|
import { COMBINATOR_KEYS, NO_STRICT, toolWireSchema } from "../utils/schema";
|
|
@@ -1487,6 +1492,54 @@ function unwrapAnthropicThinkingEnvelope(text: string): string | undefined {
|
|
|
1487
1492
|
return stripped ? current : undefined;
|
|
1488
1493
|
}
|
|
1489
1494
|
|
|
1495
|
+
/**
|
|
1496
|
+
* The refused response's content as the continuation prefix: client tool
|
|
1497
|
+
* calls (no matching tool_result) are omitted and a trailing text block is
|
|
1498
|
+
* right-trimmed, per the fallback-credit continuation contract.
|
|
1499
|
+
*/
|
|
1500
|
+
function refusalContinuationPrefix(content: AssistantMessage["content"]): AssistantMessage["content"] {
|
|
1501
|
+
const prefix = structuredClone(content.filter(block => block.type !== "toolCall"));
|
|
1502
|
+
const last = prefix.at(-1);
|
|
1503
|
+
if (last?.type === "text") last.text = last.text.trimEnd();
|
|
1504
|
+
return prefix;
|
|
1505
|
+
}
|
|
1506
|
+
|
|
1507
|
+
/** Wire form of {@link refusalContinuationPrefix} for the appended assistant message. */
|
|
1508
|
+
function formatEchoedRefusalContent(content: AssistantMessage["content"]): unknown[] {
|
|
1509
|
+
return refusalContinuationPrefix(content).map(block => {
|
|
1510
|
+
switch (block.type) {
|
|
1511
|
+
case "text":
|
|
1512
|
+
return { type: "text", text: block.text };
|
|
1513
|
+
case "thinking":
|
|
1514
|
+
return {
|
|
1515
|
+
type: "thinking",
|
|
1516
|
+
thinking: block.thinking,
|
|
1517
|
+
...(block.thinkingSignature ? { signature: block.thinkingSignature } : {}),
|
|
1518
|
+
};
|
|
1519
|
+
case "redactedThinking":
|
|
1520
|
+
return { type: "redacted_thinking", data: block.data };
|
|
1521
|
+
case "fallback":
|
|
1522
|
+
return { type: "fallback", from: block.from, to: block.to };
|
|
1523
|
+
case "anthropicServerTool":
|
|
1524
|
+
return block.block;
|
|
1525
|
+
default:
|
|
1526
|
+
return block;
|
|
1527
|
+
}
|
|
1528
|
+
});
|
|
1529
|
+
}
|
|
1530
|
+
|
|
1531
|
+
function isAnthropicBadRequest(error: unknown): boolean {
|
|
1532
|
+
if (!error) return false;
|
|
1533
|
+
if (typeof error === "object") {
|
|
1534
|
+
const rec = error as Record<string, unknown>;
|
|
1535
|
+
if (rec.status === 400 || rec.statusCode === 400) return true;
|
|
1536
|
+
if (typeof rec.message === "string" && (/\b400\b/.test(rec.message) || rec.message.includes("BadRequestError"))) {
|
|
1537
|
+
return true;
|
|
1538
|
+
}
|
|
1539
|
+
}
|
|
1540
|
+
return false;
|
|
1541
|
+
}
|
|
1542
|
+
|
|
1490
1543
|
function createEmptyUsage(premiumRequests?: number): Usage {
|
|
1491
1544
|
return {
|
|
1492
1545
|
input: 0,
|
|
@@ -2024,6 +2077,8 @@ const streamAnthropicOnce = (
|
|
|
2024
2077
|
let isOAuthToken: boolean;
|
|
2025
2078
|
// Retained so a Claude Code version bump can rebuild the client's fingerprint headers.
|
|
2026
2079
|
let clientArgs: AnthropicClientOptionsArgs | undefined;
|
|
2080
|
+
let requestExtraBetas: readonly string[] = [];
|
|
2081
|
+
let clientDefaultHeaders: Record<string, string> | undefined;
|
|
2027
2082
|
|
|
2028
2083
|
if (options?.client) {
|
|
2029
2084
|
client = options.client;
|
|
@@ -2112,6 +2167,21 @@ const streamAnthropicOnce = (
|
|
|
2112
2167
|
) {
|
|
2113
2168
|
extraBetas.push(extendedCacheTtlBeta);
|
|
2114
2169
|
}
|
|
2170
|
+
if (!isOAuth && isOfficialAnthropicApiUrl(baseUrl) && !extraBetas.includes(fallbackCreditBeta)) {
|
|
2171
|
+
extraBetas.push(fallbackCreditBeta);
|
|
2172
|
+
}
|
|
2173
|
+
if (options?.fallbackCreditRedemption) {
|
|
2174
|
+
const frozenBetas = (options.fallbackCreditRedemption.betas ?? []).filter(
|
|
2175
|
+
b => !b.startsWith("server-side-fallback-"),
|
|
2176
|
+
);
|
|
2177
|
+
extraBetas.length = 0;
|
|
2178
|
+
for (const beta of frozenBetas) {
|
|
2179
|
+
extraBetas.push(beta);
|
|
2180
|
+
}
|
|
2181
|
+
if (!extraBetas.includes(fallbackCreditBeta) && !extraBetas.includes("fallback-credit-2026-06-01")) {
|
|
2182
|
+
extraBetas.push(fallbackCreditBeta);
|
|
2183
|
+
}
|
|
2184
|
+
}
|
|
2115
2185
|
// Server-side fallback beta chain: opt-in via `options.fallbacks`.
|
|
2116
2186
|
// Nested overrides (`speed`, `output_config.effort`,
|
|
2117
2187
|
// `output_config.task_budget`) reuse the same top-level betas
|
|
@@ -2159,6 +2229,8 @@ const streamAnthropicOnce = (
|
|
|
2159
2229
|
const created = createClient(model, clientArgs);
|
|
2160
2230
|
client = created.client;
|
|
2161
2231
|
isOAuthToken = created.isOAuthToken;
|
|
2232
|
+
clientDefaultHeaders = created.defaultHeaders;
|
|
2233
|
+
requestExtraBetas = extraBetas;
|
|
2162
2234
|
}
|
|
2163
2235
|
const preparedContext = await prepareAnthropicManyImageContext(context, model.input.includes("image"));
|
|
2164
2236
|
const prepareParams = async (): Promise<MessageCreateParamsStreaming> => {
|
|
@@ -2200,7 +2272,50 @@ const streamAnthropicOnce = (
|
|
|
2200
2272
|
};
|
|
2201
2273
|
return nextParams;
|
|
2202
2274
|
};
|
|
2203
|
-
let
|
|
2275
|
+
let usingFallbackCredit = false;
|
|
2276
|
+
let fallbackCreditShape: "continuation" | "unchanged" | undefined = undefined;
|
|
2277
|
+
let fallbackCreditTransientRetries = 0;
|
|
2278
|
+
let params: MessageCreateParamsStreaming;
|
|
2279
|
+
if (
|
|
2280
|
+
options?.fallbackCreditRedemption &&
|
|
2281
|
+
Date.now() <= options.fallbackCreditRedemption.expiresAt &&
|
|
2282
|
+
options.fallbackCreditRedemption.params
|
|
2283
|
+
) {
|
|
2284
|
+
const redemption = options.fallbackCreditRedemption;
|
|
2285
|
+
usingFallbackCredit = true;
|
|
2286
|
+
const frozenParams = structuredClone(redemption.params as MessageCreateParamsStreaming);
|
|
2287
|
+
const targetModelId = options?.requestModelId ?? model.requestModelId ?? model.id;
|
|
2288
|
+
frozenParams.model = targetModelId;
|
|
2289
|
+
frozenParams.fallback_credit_token = redemption.token;
|
|
2290
|
+
delete frozenParams.fallbacks;
|
|
2291
|
+
if (
|
|
2292
|
+
redemption.prefillClaim !== false &&
|
|
2293
|
+
redemption.refusedContent &&
|
|
2294
|
+
redemption.refusedContent.length > 0
|
|
2295
|
+
) {
|
|
2296
|
+
fallbackCreditShape = "continuation";
|
|
2297
|
+
const echoed = formatEchoedRefusalContent(redemption.refusedContent);
|
|
2298
|
+
if (echoed.length > 0) {
|
|
2299
|
+
frozenParams.messages = [
|
|
2300
|
+
...frozenParams.messages,
|
|
2301
|
+
{ role: "assistant", content: echoed } as unknown as (typeof frozenParams.messages)[number],
|
|
2302
|
+
];
|
|
2303
|
+
}
|
|
2304
|
+
} else {
|
|
2305
|
+
fallbackCreditShape = "unchanged";
|
|
2306
|
+
}
|
|
2307
|
+
params = frozenParams;
|
|
2308
|
+
rawRequestDump = {
|
|
2309
|
+
provider: model.provider,
|
|
2310
|
+
api: output.api,
|
|
2311
|
+
model: model.id,
|
|
2312
|
+
method: "POST",
|
|
2313
|
+
url: `${baseUrl}/v1/messages${isOAuthToken ? "?beta=true" : ""}`,
|
|
2314
|
+
body: params,
|
|
2315
|
+
};
|
|
2316
|
+
} else {
|
|
2317
|
+
params = await prepareParams();
|
|
2318
|
+
}
|
|
2204
2319
|
const seenInputTransformations = new Set<string>();
|
|
2205
2320
|
const idleTimeoutMs = options?.streamIdleTimeoutMs ?? getStreamIdleTimeoutMs(model.compat.streamIdleTimeoutMs);
|
|
2206
2321
|
const firstEventTimeoutMs = options?.streamFirstEventTimeoutMs ?? getStreamFirstEventTimeoutMs(idleTimeoutMs);
|
|
@@ -2377,6 +2492,59 @@ const streamAnthropicOnce = (
|
|
|
2377
2492
|
const idleTimeoutAbortError = new AIError.StreamTimeoutError(
|
|
2378
2493
|
"Anthropic stream stalled while waiting for the next event",
|
|
2379
2494
|
);
|
|
2495
|
+
const resetStreamOutputState = (): void => {
|
|
2496
|
+
providerRetryAttempt = 0;
|
|
2497
|
+
output.content.length = 0;
|
|
2498
|
+
output.model = model.id;
|
|
2499
|
+
output.responseId = undefined;
|
|
2500
|
+
output.upstreamModel = undefined;
|
|
2501
|
+
output.errorMessage = undefined;
|
|
2502
|
+
output.stopDetails = undefined;
|
|
2503
|
+
output.inputTransformations = undefined;
|
|
2504
|
+
output.providerPayload = undefined;
|
|
2505
|
+
output.usage = createEmptyUsage(copilotDynamicHeaders?.premiumRequests);
|
|
2506
|
+
output.stopReason = "stop";
|
|
2507
|
+
firstTokenTime = undefined;
|
|
2508
|
+
};
|
|
2509
|
+
// A rebuilt body no longer matches the refused request, so it cannot carry
|
|
2510
|
+
// the credit token. When the refused turn already ran server tools, a
|
|
2511
|
+
// tokenless retry would re-run and re-bill them: surface the failure instead.
|
|
2512
|
+
const forfeitFallbackCredit = (streamFailure: unknown): void => {
|
|
2513
|
+
if (
|
|
2514
|
+
options?.fallbackCreditRedemption?.refusedContent?.some(block => block.type === "anthropicServerTool")
|
|
2515
|
+
) {
|
|
2516
|
+
logger.warn("anthropic: fallback credit cannot be forfeited after server tools ran; surfacing error", {
|
|
2517
|
+
model: model.id,
|
|
2518
|
+
});
|
|
2519
|
+
throw streamFailure;
|
|
2520
|
+
}
|
|
2521
|
+
usingFallbackCredit = false;
|
|
2522
|
+
fallbackCreditShape = undefined;
|
|
2523
|
+
};
|
|
2524
|
+
const rebuildParams = async (streamFailure: unknown): Promise<MessageCreateParamsStreaming> => {
|
|
2525
|
+
if (usingFallbackCredit) forfeitFallbackCredit(streamFailure);
|
|
2526
|
+
return prepareParams();
|
|
2527
|
+
};
|
|
2528
|
+
// Subscription slow mode (Claude Code `/low-priority`): first-party OAuth
|
|
2529
|
+
// requests only. Capacity waits are tracked per request so the caller's
|
|
2530
|
+
// max-wait budget covers the whole wait, not one attempt.
|
|
2531
|
+
const slowMode =
|
|
2532
|
+
options?.anthropicSlowMode !== undefined &&
|
|
2533
|
+
model.provider === "anthropic" &&
|
|
2534
|
+
isOAuthToken &&
|
|
2535
|
+
!options.client &&
|
|
2536
|
+
isOfficialAnthropicApiUrl(baseUrl)
|
|
2537
|
+
? options.anthropicSlowMode
|
|
2538
|
+
: undefined;
|
|
2539
|
+
// Slow-lane state is per Claude account: key it by the stored credential
|
|
2540
|
+
// that served this request, else by a digest of the bearer.
|
|
2541
|
+
const slowLane =
|
|
2542
|
+
options?.credentialId !== undefined
|
|
2543
|
+
? `cred:${options.credentialId}`
|
|
2544
|
+
: `key:${Bun.hash(apiKey).toString(36)}`;
|
|
2545
|
+
let slowWaitSinceMs: number | undefined;
|
|
2546
|
+
let slowWaitAttempts = 0;
|
|
2547
|
+
let sentSlow = false;
|
|
2380
2548
|
while (true) {
|
|
2381
2549
|
activeAbortTracker = createAbortSourceTracker(options?.signal);
|
|
2382
2550
|
const { requestSignal } = activeAbortTracker;
|
|
@@ -2418,14 +2586,29 @@ const streamAnthropicOnce = (
|
|
|
2418
2586
|
);
|
|
2419
2587
|
}
|
|
2420
2588
|
}
|
|
2421
|
-
|
|
2422
|
-
|
|
2589
|
+
sentSlow = slowMode?.isActive(slowLane) === true;
|
|
2590
|
+
let perRequestHeaders: Record<string, string> | undefined =
|
|
2591
|
+
umansGatewayWebSearchHeader || injectedClientBetaHeaders || options?.userProfileId || sentSlow
|
|
2423
2592
|
? {
|
|
2424
2593
|
...umansGatewayWebSearchHeader,
|
|
2425
2594
|
...injectedClientBetaHeaders,
|
|
2426
2595
|
...(options?.userProfileId ? { "anthropic-user-profile-id": options.userProfileId } : {}),
|
|
2596
|
+
...(sentSlow ? { [ANTHROPIC_USAGE_LIMIT_HEADER]: ANTHROPIC_SLOW_USAGE_LIMIT } : {}),
|
|
2427
2597
|
}
|
|
2428
2598
|
: undefined;
|
|
2599
|
+
if (usingFallbackCredit && options?.fallbackCreditRedemption?.betaHeader) {
|
|
2600
|
+
const rawBetas = options.fallbackCreditRedemption.betaHeader
|
|
2601
|
+
.split(",")
|
|
2602
|
+
.map(b => b.trim())
|
|
2603
|
+
.filter(b => b && !b.startsWith("server-side-fallback-"));
|
|
2604
|
+
if (!rawBetas.includes(fallbackCreditBeta) && !rawBetas.includes("fallback-credit-2026-06-01")) {
|
|
2605
|
+
rawBetas.push(fallbackCreditBeta);
|
|
2606
|
+
}
|
|
2607
|
+
perRequestHeaders = {
|
|
2608
|
+
...perRequestHeaders,
|
|
2609
|
+
"anthropic-beta": rawBetas.join(","),
|
|
2610
|
+
};
|
|
2611
|
+
}
|
|
2429
2612
|
const requestOptions = {
|
|
2430
2613
|
...createSdkStreamRequestOptions(requestSignal, requestTimeoutMs),
|
|
2431
2614
|
maxRetries: 0,
|
|
@@ -2465,6 +2648,10 @@ const streamAnthropicOnce = (
|
|
|
2465
2648
|
if (requestTimeout !== undefined) clearTimeout(requestTimeout);
|
|
2466
2649
|
}
|
|
2467
2650
|
await notifyProviderResponse(options, response, model, requestId);
|
|
2651
|
+
if (slowMode) {
|
|
2652
|
+
const slowSignal = parseAnthropicSlowModeHeaders(response.headers);
|
|
2653
|
+
if (slowSignal) slowMode.observe(slowSignal, slowLane);
|
|
2654
|
+
}
|
|
2468
2655
|
let sawEvent = false;
|
|
2469
2656
|
let sawMessageStart = false;
|
|
2470
2657
|
let sawTerminalEnvelope = false;
|
|
@@ -2890,7 +3077,23 @@ const streamAnthropicOnce = (
|
|
|
2890
3077
|
const category = stopDetails.category;
|
|
2891
3078
|
const label = category ? `Refusal (${category})` : "Refusal";
|
|
2892
3079
|
output.errorMessage = explanation ? `${label}: ${explanation}` : label;
|
|
2893
|
-
}
|
|
3080
|
+
}
|
|
3081
|
+
if (stopDetails?.fallback_credit_token) {
|
|
3082
|
+
const sentBetaHeader =
|
|
3083
|
+
getHeaderCaseInsensitive(perRequestHeaders ?? {}, "anthropic-beta") ??
|
|
3084
|
+
getHeaderCaseInsensitive(clientDefaultHeaders ?? {}, "anthropic-beta") ??
|
|
3085
|
+
requestExtraBetas.join(",");
|
|
3086
|
+
output.fallbackCreditHandle = {
|
|
3087
|
+
token: stopDetails.fallback_credit_token,
|
|
3088
|
+
prefillClaim: stopDetails.fallback_has_prefill_claim,
|
|
3089
|
+
params: structuredClone(params),
|
|
3090
|
+
betas: Array.from(requestExtraBetas),
|
|
3091
|
+
betaHeader: sentBetaHeader,
|
|
3092
|
+
expiresAt: Date.now() + 5 * 60 * 1000,
|
|
3093
|
+
refusedContent: output.content ? structuredClone(output.content) : undefined,
|
|
3094
|
+
};
|
|
3095
|
+
}
|
|
3096
|
+
if (!output.errorMessage) {
|
|
2894
3097
|
// Anthropic flagged an error-class stop (refusal / sensitive) without
|
|
2895
3098
|
// populating stop_details. Surface the raw reason instead of falling
|
|
2896
3099
|
// through to the generic "unknown error" string when we throw below.
|
|
@@ -3004,19 +3207,66 @@ const streamAnthropicOnce = (
|
|
|
3004
3207
|
providerSessionState.strictToolsDisabled = true;
|
|
3005
3208
|
}
|
|
3006
3209
|
disableStrictTools = true;
|
|
3007
|
-
params = await
|
|
3008
|
-
|
|
3009
|
-
output.content.length = 0;
|
|
3010
|
-
output.model = model.id;
|
|
3011
|
-
output.responseId = undefined;
|
|
3012
|
-
output.upstreamModel = undefined;
|
|
3013
|
-
output.errorMessage = undefined;
|
|
3014
|
-
output.providerPayload = undefined;
|
|
3015
|
-
output.usage = createEmptyUsage(copilotDynamicHeaders?.premiumRequests);
|
|
3016
|
-
output.stopReason = "stop";
|
|
3017
|
-
firstTokenTime = undefined;
|
|
3210
|
+
params = await rebuildParams(streamFailure);
|
|
3211
|
+
resetStreamOutputState();
|
|
3018
3212
|
continue;
|
|
3019
3213
|
}
|
|
3214
|
+
if (usingFallbackCredit && firstTokenTime === undefined && isAnthropicBadRequest(streamFailure)) {
|
|
3215
|
+
const errMessage = streamFailure instanceof Error ? streamFailure.message : String(streamFailure);
|
|
3216
|
+
const redemption = options!.fallbackCreditRedemption!;
|
|
3217
|
+
if (errMessage.includes("redemption temporarily unavailable")) {
|
|
3218
|
+
if (Date.now() < redemption.expiresAt && fallbackCreditTransientRetries < 2) {
|
|
3219
|
+
fallbackCreditTransientRetries++;
|
|
3220
|
+
logger.warn(
|
|
3221
|
+
"anthropic: fallback credit redemption temporarily unavailable, retrying same shape",
|
|
3222
|
+
{
|
|
3223
|
+
model: model.id,
|
|
3224
|
+
attempt: fallbackCreditTransientRetries,
|
|
3225
|
+
},
|
|
3226
|
+
);
|
|
3227
|
+
if (options?.providerRetryWait) {
|
|
3228
|
+
await options.providerRetryWait(500, options.signal);
|
|
3229
|
+
} else {
|
|
3230
|
+
await scheduler.wait(500, { signal: options?.signal });
|
|
3231
|
+
}
|
|
3232
|
+
resetStreamOutputState();
|
|
3233
|
+
continue;
|
|
3234
|
+
}
|
|
3235
|
+
throw streamFailure;
|
|
3236
|
+
}
|
|
3237
|
+
if (fallbackCreditShape === "continuation") {
|
|
3238
|
+
logger.warn(
|
|
3239
|
+
"anthropic: fallback credit continuation shape rejected, retrying with unchanged body",
|
|
3240
|
+
{
|
|
3241
|
+
model: model.id,
|
|
3242
|
+
error: errMessage,
|
|
3243
|
+
},
|
|
3244
|
+
);
|
|
3245
|
+
fallbackCreditShape = "unchanged";
|
|
3246
|
+
const frozenParams = structuredClone(redemption.params as MessageCreateParamsStreaming);
|
|
3247
|
+
const targetModelId = options?.requestModelId ?? model.requestModelId ?? model.id;
|
|
3248
|
+
frozenParams.model = targetModelId;
|
|
3249
|
+
frozenParams.fallback_credit_token = redemption.token;
|
|
3250
|
+
delete frozenParams.fallbacks;
|
|
3251
|
+
params = frozenParams;
|
|
3252
|
+
resetStreamOutputState();
|
|
3253
|
+
continue;
|
|
3254
|
+
}
|
|
3255
|
+
if (errMessage.includes("fallback_credit_token")) {
|
|
3256
|
+
forfeitFallbackCredit(streamFailure);
|
|
3257
|
+
logger.warn(
|
|
3258
|
+
"anthropic: fallback credit token rejected, falling back to standard request without token",
|
|
3259
|
+
{
|
|
3260
|
+
model: model.id,
|
|
3261
|
+
error: errMessage,
|
|
3262
|
+
},
|
|
3263
|
+
);
|
|
3264
|
+
dropAllThinking = true;
|
|
3265
|
+
params = await prepareParams();
|
|
3266
|
+
resetStreamOutputState();
|
|
3267
|
+
continue;
|
|
3268
|
+
}
|
|
3269
|
+
}
|
|
3020
3270
|
const streamFailureMessage =
|
|
3021
3271
|
streamFailure instanceof Error ? streamFailure.message : String(streamFailure);
|
|
3022
3272
|
if (
|
|
@@ -3030,7 +3280,8 @@ const streamAnthropicOnce = (
|
|
|
3030
3280
|
version: getClaudeCodeVersion(),
|
|
3031
3281
|
});
|
|
3032
3282
|
client = createClient(model, { ...clientArgs, disableStrictTools }).client;
|
|
3033
|
-
|
|
3283
|
+
// The version only changes client headers; a redemption keeps its frozen body.
|
|
3284
|
+
if (!usingFallbackCredit) params = await prepareParams();
|
|
3034
3285
|
providerRetryAttempt = 0;
|
|
3035
3286
|
output.content.length = 0;
|
|
3036
3287
|
output.model = model.id;
|
|
@@ -3058,7 +3309,7 @@ const streamAnthropicOnce = (
|
|
|
3058
3309
|
prefixBindingRetryAttempted = true;
|
|
3059
3310
|
prefixMismatchBehavior = undefined;
|
|
3060
3311
|
dropAllThinking = !rememberPrefixBindingFailure(params, streamFailureMessage, providerSessionState);
|
|
3061
|
-
params = await
|
|
3312
|
+
params = await rebuildParams(streamFailure);
|
|
3062
3313
|
providerRetryAttempt = 0;
|
|
3063
3314
|
output.content.length = 0;
|
|
3064
3315
|
output.model = model.id;
|
|
@@ -3092,7 +3343,7 @@ const streamAnthropicOnce = (
|
|
|
3092
3343
|
providerSessionState.replayUnsignedThinkingDisabled = true;
|
|
3093
3344
|
}
|
|
3094
3345
|
forceDemoteUnsignedThinking = true;
|
|
3095
|
-
params = await
|
|
3346
|
+
params = await rebuildParams(streamFailure);
|
|
3096
3347
|
providerRetryAttempt = 0;
|
|
3097
3348
|
output.content.length = 0;
|
|
3098
3349
|
output.model = model.id;
|
|
@@ -3134,7 +3385,7 @@ const streamAnthropicOnce = (
|
|
|
3134
3385
|
}
|
|
3135
3386
|
droppedAllThinkingForSignature = true;
|
|
3136
3387
|
dropAllThinking = true;
|
|
3137
|
-
params = await
|
|
3388
|
+
params = await rebuildParams(streamFailure);
|
|
3138
3389
|
providerRetryAttempt = 0;
|
|
3139
3390
|
output.content.length = 0;
|
|
3140
3391
|
output.model = model.id;
|
|
@@ -3162,7 +3413,7 @@ const streamAnthropicOnce = (
|
|
|
3162
3413
|
providerSessionState.fastModeDisabled = true;
|
|
3163
3414
|
}
|
|
3164
3415
|
dropFastMode = true;
|
|
3165
|
-
params = await
|
|
3416
|
+
params = await rebuildParams(streamFailure);
|
|
3166
3417
|
providerRetryAttempt = 0;
|
|
3167
3418
|
output.content.length = 0;
|
|
3168
3419
|
output.model = model.id;
|
|
@@ -3175,6 +3426,46 @@ const streamAnthropicOnce = (
|
|
|
3175
3426
|
firstTokenTime = undefined;
|
|
3176
3427
|
continue;
|
|
3177
3428
|
}
|
|
3429
|
+
if (
|
|
3430
|
+
slowMode &&
|
|
3431
|
+
firstTokenTime === undefined &&
|
|
3432
|
+
!streamedReplayUnsafeContent &&
|
|
3433
|
+
!activeAbortTracker.wasCallerAbort()
|
|
3434
|
+
) {
|
|
3435
|
+
const failureStatus = (streamFailure as { status?: unknown } | null)?.status;
|
|
3436
|
+
const httpStatus = typeof failureStatus === "number" ? failureStatus : undefined;
|
|
3437
|
+
const failureText = streamFailure instanceof Error ? streamFailure.message : String(streamFailure);
|
|
3438
|
+
const slowRetry = await slowMode.onFailure({
|
|
3439
|
+
lane: slowLane,
|
|
3440
|
+
httpStatus,
|
|
3441
|
+
overloaded: httpStatus === 529 || failureText.includes("overloaded_error"),
|
|
3442
|
+
signal: parseAnthropicSlowModeHeaders(getHeadersFromError(streamFailure)),
|
|
3443
|
+
sentSlow,
|
|
3444
|
+
waitedMs: slowWaitSinceMs === undefined ? 0 : Date.now() - slowWaitSinceMs,
|
|
3445
|
+
attempts: slowWaitAttempts,
|
|
3446
|
+
});
|
|
3447
|
+
if (slowRetry) {
|
|
3448
|
+
if (slowRetry.capacityWait) {
|
|
3449
|
+
slowWaitSinceMs ??= Date.now();
|
|
3450
|
+
slowWaitAttempts++;
|
|
3451
|
+
}
|
|
3452
|
+
logger.debug("anthropic: slow mode retry", {
|
|
3453
|
+
model: model.id,
|
|
3454
|
+
status: httpStatus,
|
|
3455
|
+
delayMs: slowRetry.delayMs,
|
|
3456
|
+
attempt: slowWaitAttempts,
|
|
3457
|
+
});
|
|
3458
|
+
if (slowRetry.delayMs > 0) {
|
|
3459
|
+
if (options?.providerRetryWait) {
|
|
3460
|
+
await options.providerRetryWait(slowRetry.delayMs, options.signal);
|
|
3461
|
+
} else {
|
|
3462
|
+
await scheduler.wait(slowRetry.delayMs, { signal: options?.signal });
|
|
3463
|
+
}
|
|
3464
|
+
}
|
|
3465
|
+
resetStreamOutputState();
|
|
3466
|
+
continue;
|
|
3467
|
+
}
|
|
3468
|
+
}
|
|
3178
3469
|
const isTransientEnvelopeFailure =
|
|
3179
3470
|
AIError.isTransientStreamParseError(streamFailure) || AIError.isStreamEnvelopeError(streamFailure);
|
|
3180
3471
|
const isLocalIdleTimeout =
|
|
@@ -3223,6 +3514,15 @@ const streamAnthropicOnce = (
|
|
|
3223
3514
|
firstTokenTime = undefined;
|
|
3224
3515
|
}
|
|
3225
3516
|
}
|
|
3517
|
+
if (
|
|
3518
|
+
usingFallbackCredit &&
|
|
3519
|
+
fallbackCreditShape === "continuation" &&
|
|
3520
|
+
options?.fallbackCreditRedemption?.refusedContent
|
|
3521
|
+
) {
|
|
3522
|
+
// The response continues the echoed prefix; keep it in the stored turn in
|
|
3523
|
+
// AssistantMessage form so later replays keep signatures and server tools.
|
|
3524
|
+
output.content.unshift(...refusalContinuationPrefix(options.fallbackCreditRedemption.refusedContent));
|
|
3525
|
+
}
|
|
3226
3526
|
output.duration = performance.now() - startTime;
|
|
3227
3527
|
if (firstTokenTime) output.ttft = firstTokenTime - startTime;
|
|
3228
3528
|
if (dropFastMode && model.provider === "anthropic" && options?.serviceTier === "priority") {
|
|
@@ -3530,10 +3830,10 @@ export function buildAnthropicClientOptions(args: AnthropicClientOptionsArgs): A
|
|
|
3530
3830
|
function createClient(
|
|
3531
3831
|
model: Model<"anthropic-messages">,
|
|
3532
3832
|
args: AnthropicClientOptionsArgs,
|
|
3533
|
-
): { client: AnthropicMessagesClient; isOAuthToken: boolean } {
|
|
3833
|
+
): { client: AnthropicMessagesClient; isOAuthToken: boolean; defaultHeaders?: Record<string, string> } {
|
|
3534
3834
|
const { isOAuthToken: oauthToken, ...clientOptions } = buildAnthropicClientOptions({ ...args, model });
|
|
3535
3835
|
const client = new AnthropicMessagesClient(clientOptions);
|
|
3536
|
-
return { client, isOAuthToken: oauthToken };
|
|
3836
|
+
return { client, isOAuthToken: oauthToken, defaultHeaders: clientOptions.defaultHeaders };
|
|
3537
3837
|
}
|
|
3538
3838
|
|
|
3539
3839
|
/** The compaction request is a standalone summary call, not a generation turn. */
|
|
@@ -104,18 +104,25 @@ function responseHeaders(message: IncomingMessage): Headers {
|
|
|
104
104
|
function decodedResponseStream(message: IncomingMessage): stream.Readable {
|
|
105
105
|
const rawEncoding = message.headers["content-encoding"];
|
|
106
106
|
const encoding = (Array.isArray(rawEncoding) ? rawEncoding[0] : rawEncoding)?.trim().toLowerCase();
|
|
107
|
+
let decoder: stream.Transform;
|
|
107
108
|
switch (encoding) {
|
|
108
109
|
case "gzip":
|
|
109
|
-
|
|
110
|
+
decoder = zlib.createGunzip();
|
|
111
|
+
break;
|
|
110
112
|
case "deflate":
|
|
111
|
-
|
|
113
|
+
decoder = zlib.createInflate();
|
|
114
|
+
break;
|
|
112
115
|
case "br":
|
|
113
|
-
|
|
116
|
+
decoder = zlib.createBrotliDecompress();
|
|
117
|
+
break;
|
|
114
118
|
case "zstd":
|
|
115
|
-
|
|
119
|
+
decoder = zlib.createZstdDecompress();
|
|
120
|
+
break;
|
|
116
121
|
default:
|
|
117
122
|
return message;
|
|
118
123
|
}
|
|
124
|
+
// Couple decoded-body cancellation to the source so its keep-alive socket cannot be stranded.
|
|
125
|
+
return stream.pipeline(message, decoder, () => {});
|
|
119
126
|
}
|
|
120
127
|
|
|
121
128
|
function createResponse(message: IncomingMessage, method: string): Response {
|
|
@@ -617,7 +617,13 @@ export async function consumeGoogleStream<T extends GoogleApiType>(args: {
|
|
|
617
617
|
|
|
618
618
|
for await (const chunk of googleStream) {
|
|
619
619
|
if (chunk.error) {
|
|
620
|
-
|
|
620
|
+
// Keep the RPC status alongside the message: an in-band quota failure
|
|
621
|
+
// is classified from this text, and `RESOURCE_EXHAUSTED` is the only
|
|
622
|
+
// account-exhaustion signal some of these chunks carry (#13090).
|
|
623
|
+
const detail =
|
|
624
|
+
chunk.error.message && chunk.error.status
|
|
625
|
+
? `${chunk.error.message} (${chunk.error.status})`
|
|
626
|
+
: chunk.error.message || chunk.error.status || "unknown error";
|
|
621
627
|
const message = `Google API stream error: ${detail}`;
|
|
622
628
|
throw typeof chunk.error.code === "number" && chunk.error.code >= 400
|
|
623
629
|
? new AIError.GoogleApiError(message, chunk.error.code)
|
|
@@ -969,7 +975,7 @@ export function streamGoogleGenAI<T extends "google-generative-ai" | "google-ver
|
|
|
969
975
|
if (!response.ok) {
|
|
970
976
|
const errorText = await response.text().catch(() => "");
|
|
971
977
|
throw new AIError.GoogleApiError(
|
|
972
|
-
`Google API error (${response.status}): ${extractGoogleErrorMessage(errorText)}`,
|
|
978
|
+
`Google API error (${response.status}): ${extractGoogleErrorMessage(errorText, response.status)}`,
|
|
973
979
|
response.status,
|
|
974
980
|
{ headers: response.headers },
|
|
975
981
|
);
|
|
@@ -1101,13 +1107,30 @@ function paramsToWireBody(params: GenerateContentParameters): Record<string, unk
|
|
|
1101
1107
|
return body;
|
|
1102
1108
|
}
|
|
1103
1109
|
|
|
1104
|
-
|
|
1110
|
+
/**
|
|
1111
|
+
* Human-readable message for a non-2xx Google response.
|
|
1112
|
+
*
|
|
1113
|
+
* On a usage-limit status the RPC `status`/`details` residue is kept after the
|
|
1114
|
+
* message: `parseGoogleRpcRateLimitReason` reads `RESOURCE_EXHAUSTED` plus the
|
|
1115
|
+
* `google.rpc.ErrorInfo` reason to tell an account billing cap (terminal) from
|
|
1116
|
+
* a per-minute throttle (retryable), and reducing the body to `error.message`
|
|
1117
|
+
* hid both, so every billing 429 replayed as a transient rate limit (#13090).
|
|
1118
|
+
* The Cloud Code Assist path keeps the whole raw body for the same reason.
|
|
1119
|
+
*/
|
|
1120
|
+
function extractGoogleErrorMessage(errorText: string, status: number): string {
|
|
1105
1121
|
if (!errorText) return "Unknown error";
|
|
1106
1122
|
try {
|
|
1107
|
-
const parsed = JSON.parse(errorText) as {
|
|
1108
|
-
|
|
1123
|
+
const parsed = JSON.parse(errorText) as {
|
|
1124
|
+
error?: { message?: string; status?: string; details?: unknown[] };
|
|
1125
|
+
};
|
|
1126
|
+
const error = parsed.error;
|
|
1127
|
+
if (!error?.message) return errorText;
|
|
1128
|
+
if (!AIError.isUsageLimitStatus(status)) return error.message;
|
|
1129
|
+
const residue = { error: { status: error.status, details: error.details } };
|
|
1130
|
+
return error.status === undefined && error.details === undefined
|
|
1131
|
+
? error.message
|
|
1132
|
+
: `${error.message} ${JSON.stringify(residue)}`;
|
|
1109
1133
|
} catch {
|
|
1110
|
-
|
|
1134
|
+
return errorText;
|
|
1111
1135
|
}
|
|
1112
|
-
return errorText;
|
|
1113
1136
|
}
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
/** Shared inference request identity headers. */
|
|
2
2
|
|
|
3
|
-
import { USER_AGENT } from "@oh-my-pi/pi-utils";
|
|
3
|
+
import { APP_NAME, APP_URL, USER_AGENT } from "@oh-my-pi/pi-utils";
|
|
4
4
|
|
|
5
5
|
/** Options controlling provider and protocol inference headers. */
|
|
6
6
|
export interface InferenceHeaderOptions {
|
|
@@ -33,6 +33,12 @@ function setHeader(headers: Record<string, string>, name: string, value: string)
|
|
|
33
33
|
* understood by the active inference protocol and host.
|
|
34
34
|
*/
|
|
35
35
|
export function applyInferenceHeaders(headers: Record<string, string>, options: InferenceHeaderOptions): void {
|
|
36
|
+
if (options.provider === "vercel-ai-gateway") {
|
|
37
|
+
// Vercel AI Gateway app attribution; caller/config headers take precedence.
|
|
38
|
+
setHeaderIfAbsent(headers, "http-referer", APP_URL);
|
|
39
|
+
setHeaderIfAbsent(headers, "x-title", APP_NAME);
|
|
40
|
+
}
|
|
41
|
+
|
|
36
42
|
const isOpenCode = options.provider === "opencode-go" || options.provider === "opencode-zen";
|
|
37
43
|
const sessionId = options.sessionId;
|
|
38
44
|
if (!sessionId) return;
|
package/src/providers/mock.ts
CHANGED
|
@@ -46,6 +46,7 @@ import { classifyModel } from "@oh-my-pi/pi-catalog/compat/taxonomy";
|
|
|
46
46
|
import { registerCustomApi } from "../api-registry";
|
|
47
47
|
import * as AIError from "../error";
|
|
48
48
|
import type {
|
|
49
|
+
AnthropicFallbackCreditHandle,
|
|
49
50
|
Api,
|
|
50
51
|
AssistantMessage,
|
|
51
52
|
Context,
|
|
@@ -86,6 +87,8 @@ export interface MockResponse {
|
|
|
86
87
|
stopReason?: StopReason;
|
|
87
88
|
/** Structured terminal stop classification, e.g. Anthropic refusal metadata. */
|
|
88
89
|
stopDetails?: StopDetails | null;
|
|
90
|
+
/** In-memory fallback credit handle attached when a refusal response carries a fallback credit token. */
|
|
91
|
+
fallbackCreditHandle?: AnthropicFallbackCreditHandle;
|
|
89
92
|
/** Error text paired with an explicit `"error"` stop reason. */
|
|
90
93
|
errorMessage?: string;
|
|
91
94
|
/** Usage stats. Missing fields default to 0; missing `cost.total` is recomputed from components. */
|
|
@@ -403,6 +406,7 @@ async function runMock(
|
|
|
403
406
|
|
|
404
407
|
partial.stopReason = reason;
|
|
405
408
|
partial.stopDetails = response.stopDetails;
|
|
409
|
+
partial.fallbackCreditHandle = response.fallbackCreditHandle;
|
|
406
410
|
partial.errorMessage = response.errorMessage;
|
|
407
411
|
partial.usage = mergeUsage(response.usage);
|
|
408
412
|
partial.duration = performance.now() - perfStart;
|