@oh-my-pi/pi-ai 18.3.0 → 18.3.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (45) hide show
  1. package/CHANGELOG.md +20 -0
  2. package/THIRD-PARTY-NOTICES.txt +2 -2
  3. package/dist/types/auth/policy.d.ts +8 -1
  4. package/dist/types/auth/pool.d.ts +8 -0
  5. package/dist/types/auth/types.d.ts +17 -7
  6. package/dist/types/auth/usage.d.ts +2 -0
  7. package/dist/types/auth-broker/discover.d.ts +22 -1
  8. package/dist/types/auth-storage.d.ts +35 -11
  9. package/dist/types/error/rate-limit.d.ts +3 -2
  10. package/dist/types/index.d.ts +1 -0
  11. package/dist/types/providers/anthropic-slow-mode.d.ts +109 -0
  12. package/dist/types/providers/anthropic-wire.d.ts +4 -0
  13. package/dist/types/providers/mock.d.ts +3 -1
  14. package/dist/types/providers/openai-codex/live-steering.d.ts +77 -0
  15. package/dist/types/providers/transform-messages.d.ts +9 -1
  16. package/dist/types/types.d.ts +56 -0
  17. package/dist/types/usage/xai-oauth.d.ts +6 -1
  18. package/dist/types/utils/http-inspector.d.ts +6 -0
  19. package/package.json +6 -6
  20. package/src/auth/policy.ts +27 -6
  21. package/src/auth/pool.ts +35 -2
  22. package/src/auth/refresh.ts +2 -2
  23. package/src/auth/types.ts +17 -7
  24. package/src/auth/usage.ts +5 -0
  25. package/src/auth-broker/discover.ts +57 -27
  26. package/src/auth-storage.ts +121 -35
  27. package/src/error/flags.ts +2 -1
  28. package/src/error/rate-limit.ts +8 -4
  29. package/src/index.ts +1 -0
  30. package/src/providers/anthropic-slow-mode.ts +232 -0
  31. package/src/providers/anthropic-wire.ts +4 -0
  32. package/src/providers/anthropic.ts +322 -22
  33. package/src/providers/cowork-fetch.ts +11 -4
  34. package/src/providers/google-shared.ts +30 -7
  35. package/src/providers/inference-headers.ts +7 -1
  36. package/src/providers/mock.ts +4 -0
  37. package/src/providers/openai-codex/live-steering.ts +237 -0
  38. package/src/providers/openai-codex-responses.ts +372 -73
  39. package/src/providers/transform-messages.ts +27 -7
  40. package/src/stream.ts +32 -2
  41. package/src/types.ts +59 -0
  42. package/src/usage/registry.ts +2 -1
  43. package/src/usage/xai-oauth.ts +31 -1
  44. package/src/utils/http-inspector.ts +21 -2
  45. package/src/utils/openrouter-headers.ts +3 -3
@@ -77,6 +77,11 @@ import { AssistantMessageEventStream } from "../utils/event-stream";
77
77
  import { isFoundryEnabled } from "../utils/foundry";
78
78
  import { finalizeErrorMessage, type RawHttpRequestDump } from "../utils/http-inspector";
79
79
  import { getStreamFirstEventTimeoutMs, getStreamIdleTimeoutMs, iterateWithIdleTimeout } from "../utils/idle-iterator";
80
+ import {
81
+ ANTHROPIC_SLOW_USAGE_LIMIT,
82
+ ANTHROPIC_USAGE_LIMIT_HEADER,
83
+ parseAnthropicSlowModeHeaders,
84
+ } from "./anthropic-slow-mode";
80
85
  import { notifyProviderResponse } from "../utils/provider-response";
81
86
  import { getHeadersFromError, getRetryAfterMsFromHeaders } from "../utils/retry-after";
82
87
  import { COMBINATOR_KEYS, NO_STRICT, toolWireSchema } from "../utils/schema";
@@ -1487,6 +1492,54 @@ function unwrapAnthropicThinkingEnvelope(text: string): string | undefined {
1487
1492
  return stripped ? current : undefined;
1488
1493
  }
1489
1494
 
1495
+ /**
1496
+ * The refused response's content as the continuation prefix: client tool
1497
+ * calls (no matching tool_result) are omitted and a trailing text block is
1498
+ * right-trimmed, per the fallback-credit continuation contract.
1499
+ */
1500
+ function refusalContinuationPrefix(content: AssistantMessage["content"]): AssistantMessage["content"] {
1501
+ const prefix = structuredClone(content.filter(block => block.type !== "toolCall"));
1502
+ const last = prefix.at(-1);
1503
+ if (last?.type === "text") last.text = last.text.trimEnd();
1504
+ return prefix;
1505
+ }
1506
+
1507
+ /** Wire form of {@link refusalContinuationPrefix} for the appended assistant message. */
1508
+ function formatEchoedRefusalContent(content: AssistantMessage["content"]): unknown[] {
1509
+ return refusalContinuationPrefix(content).map(block => {
1510
+ switch (block.type) {
1511
+ case "text":
1512
+ return { type: "text", text: block.text };
1513
+ case "thinking":
1514
+ return {
1515
+ type: "thinking",
1516
+ thinking: block.thinking,
1517
+ ...(block.thinkingSignature ? { signature: block.thinkingSignature } : {}),
1518
+ };
1519
+ case "redactedThinking":
1520
+ return { type: "redacted_thinking", data: block.data };
1521
+ case "fallback":
1522
+ return { type: "fallback", from: block.from, to: block.to };
1523
+ case "anthropicServerTool":
1524
+ return block.block;
1525
+ default:
1526
+ return block;
1527
+ }
1528
+ });
1529
+ }
1530
+
1531
+ function isAnthropicBadRequest(error: unknown): boolean {
1532
+ if (!error) return false;
1533
+ if (typeof error === "object") {
1534
+ const rec = error as Record<string, unknown>;
1535
+ if (rec.status === 400 || rec.statusCode === 400) return true;
1536
+ if (typeof rec.message === "string" && (/\b400\b/.test(rec.message) || rec.message.includes("BadRequestError"))) {
1537
+ return true;
1538
+ }
1539
+ }
1540
+ return false;
1541
+ }
1542
+
1490
1543
  function createEmptyUsage(premiumRequests?: number): Usage {
1491
1544
  return {
1492
1545
  input: 0,
@@ -2024,6 +2077,8 @@ const streamAnthropicOnce = (
2024
2077
  let isOAuthToken: boolean;
2025
2078
  // Retained so a Claude Code version bump can rebuild the client's fingerprint headers.
2026
2079
  let clientArgs: AnthropicClientOptionsArgs | undefined;
2080
+ let requestExtraBetas: readonly string[] = [];
2081
+ let clientDefaultHeaders: Record<string, string> | undefined;
2027
2082
 
2028
2083
  if (options?.client) {
2029
2084
  client = options.client;
@@ -2112,6 +2167,21 @@ const streamAnthropicOnce = (
2112
2167
  ) {
2113
2168
  extraBetas.push(extendedCacheTtlBeta);
2114
2169
  }
2170
+ if (!isOAuth && isOfficialAnthropicApiUrl(baseUrl) && !extraBetas.includes(fallbackCreditBeta)) {
2171
+ extraBetas.push(fallbackCreditBeta);
2172
+ }
2173
+ if (options?.fallbackCreditRedemption) {
2174
+ const frozenBetas = (options.fallbackCreditRedemption.betas ?? []).filter(
2175
+ b => !b.startsWith("server-side-fallback-"),
2176
+ );
2177
+ extraBetas.length = 0;
2178
+ for (const beta of frozenBetas) {
2179
+ extraBetas.push(beta);
2180
+ }
2181
+ if (!extraBetas.includes(fallbackCreditBeta) && !extraBetas.includes("fallback-credit-2026-06-01")) {
2182
+ extraBetas.push(fallbackCreditBeta);
2183
+ }
2184
+ }
2115
2185
  // Server-side fallback beta chain: opt-in via `options.fallbacks`.
2116
2186
  // Nested overrides (`speed`, `output_config.effort`,
2117
2187
  // `output_config.task_budget`) reuse the same top-level betas
@@ -2159,6 +2229,8 @@ const streamAnthropicOnce = (
2159
2229
  const created = createClient(model, clientArgs);
2160
2230
  client = created.client;
2161
2231
  isOAuthToken = created.isOAuthToken;
2232
+ clientDefaultHeaders = created.defaultHeaders;
2233
+ requestExtraBetas = extraBetas;
2162
2234
  }
2163
2235
  const preparedContext = await prepareAnthropicManyImageContext(context, model.input.includes("image"));
2164
2236
  const prepareParams = async (): Promise<MessageCreateParamsStreaming> => {
@@ -2200,7 +2272,50 @@ const streamAnthropicOnce = (
2200
2272
  };
2201
2273
  return nextParams;
2202
2274
  };
2203
- let params = await prepareParams();
2275
+ let usingFallbackCredit = false;
2276
+ let fallbackCreditShape: "continuation" | "unchanged" | undefined = undefined;
2277
+ let fallbackCreditTransientRetries = 0;
2278
+ let params: MessageCreateParamsStreaming;
2279
+ if (
2280
+ options?.fallbackCreditRedemption &&
2281
+ Date.now() <= options.fallbackCreditRedemption.expiresAt &&
2282
+ options.fallbackCreditRedemption.params
2283
+ ) {
2284
+ const redemption = options.fallbackCreditRedemption;
2285
+ usingFallbackCredit = true;
2286
+ const frozenParams = structuredClone(redemption.params as MessageCreateParamsStreaming);
2287
+ const targetModelId = options?.requestModelId ?? model.requestModelId ?? model.id;
2288
+ frozenParams.model = targetModelId;
2289
+ frozenParams.fallback_credit_token = redemption.token;
2290
+ delete frozenParams.fallbacks;
2291
+ if (
2292
+ redemption.prefillClaim !== false &&
2293
+ redemption.refusedContent &&
2294
+ redemption.refusedContent.length > 0
2295
+ ) {
2296
+ fallbackCreditShape = "continuation";
2297
+ const echoed = formatEchoedRefusalContent(redemption.refusedContent);
2298
+ if (echoed.length > 0) {
2299
+ frozenParams.messages = [
2300
+ ...frozenParams.messages,
2301
+ { role: "assistant", content: echoed } as unknown as (typeof frozenParams.messages)[number],
2302
+ ];
2303
+ }
2304
+ } else {
2305
+ fallbackCreditShape = "unchanged";
2306
+ }
2307
+ params = frozenParams;
2308
+ rawRequestDump = {
2309
+ provider: model.provider,
2310
+ api: output.api,
2311
+ model: model.id,
2312
+ method: "POST",
2313
+ url: `${baseUrl}/v1/messages${isOAuthToken ? "?beta=true" : ""}`,
2314
+ body: params,
2315
+ };
2316
+ } else {
2317
+ params = await prepareParams();
2318
+ }
2204
2319
  const seenInputTransformations = new Set<string>();
2205
2320
  const idleTimeoutMs = options?.streamIdleTimeoutMs ?? getStreamIdleTimeoutMs(model.compat.streamIdleTimeoutMs);
2206
2321
  const firstEventTimeoutMs = options?.streamFirstEventTimeoutMs ?? getStreamFirstEventTimeoutMs(idleTimeoutMs);
@@ -2377,6 +2492,59 @@ const streamAnthropicOnce = (
2377
2492
  const idleTimeoutAbortError = new AIError.StreamTimeoutError(
2378
2493
  "Anthropic stream stalled while waiting for the next event",
2379
2494
  );
2495
+ const resetStreamOutputState = (): void => {
2496
+ providerRetryAttempt = 0;
2497
+ output.content.length = 0;
2498
+ output.model = model.id;
2499
+ output.responseId = undefined;
2500
+ output.upstreamModel = undefined;
2501
+ output.errorMessage = undefined;
2502
+ output.stopDetails = undefined;
2503
+ output.inputTransformations = undefined;
2504
+ output.providerPayload = undefined;
2505
+ output.usage = createEmptyUsage(copilotDynamicHeaders?.premiumRequests);
2506
+ output.stopReason = "stop";
2507
+ firstTokenTime = undefined;
2508
+ };
2509
+ // A rebuilt body no longer matches the refused request, so it cannot carry
2510
+ // the credit token. When the refused turn already ran server tools, a
2511
+ // tokenless retry would re-run and re-bill them: surface the failure instead.
2512
+ const forfeitFallbackCredit = (streamFailure: unknown): void => {
2513
+ if (
2514
+ options?.fallbackCreditRedemption?.refusedContent?.some(block => block.type === "anthropicServerTool")
2515
+ ) {
2516
+ logger.warn("anthropic: fallback credit cannot be forfeited after server tools ran; surfacing error", {
2517
+ model: model.id,
2518
+ });
2519
+ throw streamFailure;
2520
+ }
2521
+ usingFallbackCredit = false;
2522
+ fallbackCreditShape = undefined;
2523
+ };
2524
+ const rebuildParams = async (streamFailure: unknown): Promise<MessageCreateParamsStreaming> => {
2525
+ if (usingFallbackCredit) forfeitFallbackCredit(streamFailure);
2526
+ return prepareParams();
2527
+ };
2528
+ // Subscription slow mode (Claude Code `/low-priority`): first-party OAuth
2529
+ // requests only. Capacity waits are tracked per request so the caller's
2530
+ // max-wait budget covers the whole wait, not one attempt.
2531
+ const slowMode =
2532
+ options?.anthropicSlowMode !== undefined &&
2533
+ model.provider === "anthropic" &&
2534
+ isOAuthToken &&
2535
+ !options.client &&
2536
+ isOfficialAnthropicApiUrl(baseUrl)
2537
+ ? options.anthropicSlowMode
2538
+ : undefined;
2539
+ // Slow-lane state is per Claude account: key it by the stored credential
2540
+ // that served this request, else by a digest of the bearer.
2541
+ const slowLane =
2542
+ options?.credentialId !== undefined
2543
+ ? `cred:${options.credentialId}`
2544
+ : `key:${Bun.hash(apiKey).toString(36)}`;
2545
+ let slowWaitSinceMs: number | undefined;
2546
+ let slowWaitAttempts = 0;
2547
+ let sentSlow = false;
2380
2548
  while (true) {
2381
2549
  activeAbortTracker = createAbortSourceTracker(options?.signal);
2382
2550
  const { requestSignal } = activeAbortTracker;
@@ -2418,14 +2586,29 @@ const streamAnthropicOnce = (
2418
2586
  );
2419
2587
  }
2420
2588
  }
2421
- const perRequestHeaders =
2422
- umansGatewayWebSearchHeader || injectedClientBetaHeaders || options?.userProfileId
2589
+ sentSlow = slowMode?.isActive(slowLane) === true;
2590
+ let perRequestHeaders: Record<string, string> | undefined =
2591
+ umansGatewayWebSearchHeader || injectedClientBetaHeaders || options?.userProfileId || sentSlow
2423
2592
  ? {
2424
2593
  ...umansGatewayWebSearchHeader,
2425
2594
  ...injectedClientBetaHeaders,
2426
2595
  ...(options?.userProfileId ? { "anthropic-user-profile-id": options.userProfileId } : {}),
2596
+ ...(sentSlow ? { [ANTHROPIC_USAGE_LIMIT_HEADER]: ANTHROPIC_SLOW_USAGE_LIMIT } : {}),
2427
2597
  }
2428
2598
  : undefined;
2599
+ if (usingFallbackCredit && options?.fallbackCreditRedemption?.betaHeader) {
2600
+ const rawBetas = options.fallbackCreditRedemption.betaHeader
2601
+ .split(",")
2602
+ .map(b => b.trim())
2603
+ .filter(b => b && !b.startsWith("server-side-fallback-"));
2604
+ if (!rawBetas.includes(fallbackCreditBeta) && !rawBetas.includes("fallback-credit-2026-06-01")) {
2605
+ rawBetas.push(fallbackCreditBeta);
2606
+ }
2607
+ perRequestHeaders = {
2608
+ ...perRequestHeaders,
2609
+ "anthropic-beta": rawBetas.join(","),
2610
+ };
2611
+ }
2429
2612
  const requestOptions = {
2430
2613
  ...createSdkStreamRequestOptions(requestSignal, requestTimeoutMs),
2431
2614
  maxRetries: 0,
@@ -2465,6 +2648,10 @@ const streamAnthropicOnce = (
2465
2648
  if (requestTimeout !== undefined) clearTimeout(requestTimeout);
2466
2649
  }
2467
2650
  await notifyProviderResponse(options, response, model, requestId);
2651
+ if (slowMode) {
2652
+ const slowSignal = parseAnthropicSlowModeHeaders(response.headers);
2653
+ if (slowSignal) slowMode.observe(slowSignal, slowLane);
2654
+ }
2468
2655
  let sawEvent = false;
2469
2656
  let sawMessageStart = false;
2470
2657
  let sawTerminalEnvelope = false;
@@ -2890,7 +3077,23 @@ const streamAnthropicOnce = (
2890
3077
  const category = stopDetails.category;
2891
3078
  const label = category ? `Refusal (${category})` : "Refusal";
2892
3079
  output.errorMessage = explanation ? `${label}: ${explanation}` : label;
2893
- } else if (!output.errorMessage) {
3080
+ }
3081
+ if (stopDetails?.fallback_credit_token) {
3082
+ const sentBetaHeader =
3083
+ getHeaderCaseInsensitive(perRequestHeaders ?? {}, "anthropic-beta") ??
3084
+ getHeaderCaseInsensitive(clientDefaultHeaders ?? {}, "anthropic-beta") ??
3085
+ requestExtraBetas.join(",");
3086
+ output.fallbackCreditHandle = {
3087
+ token: stopDetails.fallback_credit_token,
3088
+ prefillClaim: stopDetails.fallback_has_prefill_claim,
3089
+ params: structuredClone(params),
3090
+ betas: Array.from(requestExtraBetas),
3091
+ betaHeader: sentBetaHeader,
3092
+ expiresAt: Date.now() + 5 * 60 * 1000,
3093
+ refusedContent: output.content ? structuredClone(output.content) : undefined,
3094
+ };
3095
+ }
3096
+ if (!output.errorMessage) {
2894
3097
  // Anthropic flagged an error-class stop (refusal / sensitive) without
2895
3098
  // populating stop_details. Surface the raw reason instead of falling
2896
3099
  // through to the generic "unknown error" string when we throw below.
@@ -3004,19 +3207,66 @@ const streamAnthropicOnce = (
3004
3207
  providerSessionState.strictToolsDisabled = true;
3005
3208
  }
3006
3209
  disableStrictTools = true;
3007
- params = await prepareParams();
3008
- providerRetryAttempt = 0;
3009
- output.content.length = 0;
3010
- output.model = model.id;
3011
- output.responseId = undefined;
3012
- output.upstreamModel = undefined;
3013
- output.errorMessage = undefined;
3014
- output.providerPayload = undefined;
3015
- output.usage = createEmptyUsage(copilotDynamicHeaders?.premiumRequests);
3016
- output.stopReason = "stop";
3017
- firstTokenTime = undefined;
3210
+ params = await rebuildParams(streamFailure);
3211
+ resetStreamOutputState();
3018
3212
  continue;
3019
3213
  }
3214
+ if (usingFallbackCredit && firstTokenTime === undefined && isAnthropicBadRequest(streamFailure)) {
3215
+ const errMessage = streamFailure instanceof Error ? streamFailure.message : String(streamFailure);
3216
+ const redemption = options!.fallbackCreditRedemption!;
3217
+ if (errMessage.includes("redemption temporarily unavailable")) {
3218
+ if (Date.now() < redemption.expiresAt && fallbackCreditTransientRetries < 2) {
3219
+ fallbackCreditTransientRetries++;
3220
+ logger.warn(
3221
+ "anthropic: fallback credit redemption temporarily unavailable, retrying same shape",
3222
+ {
3223
+ model: model.id,
3224
+ attempt: fallbackCreditTransientRetries,
3225
+ },
3226
+ );
3227
+ if (options?.providerRetryWait) {
3228
+ await options.providerRetryWait(500, options.signal);
3229
+ } else {
3230
+ await scheduler.wait(500, { signal: options?.signal });
3231
+ }
3232
+ resetStreamOutputState();
3233
+ continue;
3234
+ }
3235
+ throw streamFailure;
3236
+ }
3237
+ if (fallbackCreditShape === "continuation") {
3238
+ logger.warn(
3239
+ "anthropic: fallback credit continuation shape rejected, retrying with unchanged body",
3240
+ {
3241
+ model: model.id,
3242
+ error: errMessage,
3243
+ },
3244
+ );
3245
+ fallbackCreditShape = "unchanged";
3246
+ const frozenParams = structuredClone(redemption.params as MessageCreateParamsStreaming);
3247
+ const targetModelId = options?.requestModelId ?? model.requestModelId ?? model.id;
3248
+ frozenParams.model = targetModelId;
3249
+ frozenParams.fallback_credit_token = redemption.token;
3250
+ delete frozenParams.fallbacks;
3251
+ params = frozenParams;
3252
+ resetStreamOutputState();
3253
+ continue;
3254
+ }
3255
+ if (errMessage.includes("fallback_credit_token")) {
3256
+ forfeitFallbackCredit(streamFailure);
3257
+ logger.warn(
3258
+ "anthropic: fallback credit token rejected, falling back to standard request without token",
3259
+ {
3260
+ model: model.id,
3261
+ error: errMessage,
3262
+ },
3263
+ );
3264
+ dropAllThinking = true;
3265
+ params = await prepareParams();
3266
+ resetStreamOutputState();
3267
+ continue;
3268
+ }
3269
+ }
3020
3270
  const streamFailureMessage =
3021
3271
  streamFailure instanceof Error ? streamFailure.message : String(streamFailure);
3022
3272
  if (
@@ -3030,7 +3280,8 @@ const streamAnthropicOnce = (
3030
3280
  version: getClaudeCodeVersion(),
3031
3281
  });
3032
3282
  client = createClient(model, { ...clientArgs, disableStrictTools }).client;
3033
- params = await prepareParams();
3283
+ // The version only changes client headers; a redemption keeps its frozen body.
3284
+ if (!usingFallbackCredit) params = await prepareParams();
3034
3285
  providerRetryAttempt = 0;
3035
3286
  output.content.length = 0;
3036
3287
  output.model = model.id;
@@ -3058,7 +3309,7 @@ const streamAnthropicOnce = (
3058
3309
  prefixBindingRetryAttempted = true;
3059
3310
  prefixMismatchBehavior = undefined;
3060
3311
  dropAllThinking = !rememberPrefixBindingFailure(params, streamFailureMessage, providerSessionState);
3061
- params = await prepareParams();
3312
+ params = await rebuildParams(streamFailure);
3062
3313
  providerRetryAttempt = 0;
3063
3314
  output.content.length = 0;
3064
3315
  output.model = model.id;
@@ -3092,7 +3343,7 @@ const streamAnthropicOnce = (
3092
3343
  providerSessionState.replayUnsignedThinkingDisabled = true;
3093
3344
  }
3094
3345
  forceDemoteUnsignedThinking = true;
3095
- params = await prepareParams();
3346
+ params = await rebuildParams(streamFailure);
3096
3347
  providerRetryAttempt = 0;
3097
3348
  output.content.length = 0;
3098
3349
  output.model = model.id;
@@ -3134,7 +3385,7 @@ const streamAnthropicOnce = (
3134
3385
  }
3135
3386
  droppedAllThinkingForSignature = true;
3136
3387
  dropAllThinking = true;
3137
- params = await prepareParams();
3388
+ params = await rebuildParams(streamFailure);
3138
3389
  providerRetryAttempt = 0;
3139
3390
  output.content.length = 0;
3140
3391
  output.model = model.id;
@@ -3162,7 +3413,7 @@ const streamAnthropicOnce = (
3162
3413
  providerSessionState.fastModeDisabled = true;
3163
3414
  }
3164
3415
  dropFastMode = true;
3165
- params = await prepareParams();
3416
+ params = await rebuildParams(streamFailure);
3166
3417
  providerRetryAttempt = 0;
3167
3418
  output.content.length = 0;
3168
3419
  output.model = model.id;
@@ -3175,6 +3426,46 @@ const streamAnthropicOnce = (
3175
3426
  firstTokenTime = undefined;
3176
3427
  continue;
3177
3428
  }
3429
+ if (
3430
+ slowMode &&
3431
+ firstTokenTime === undefined &&
3432
+ !streamedReplayUnsafeContent &&
3433
+ !activeAbortTracker.wasCallerAbort()
3434
+ ) {
3435
+ const failureStatus = (streamFailure as { status?: unknown } | null)?.status;
3436
+ const httpStatus = typeof failureStatus === "number" ? failureStatus : undefined;
3437
+ const failureText = streamFailure instanceof Error ? streamFailure.message : String(streamFailure);
3438
+ const slowRetry = await slowMode.onFailure({
3439
+ lane: slowLane,
3440
+ httpStatus,
3441
+ overloaded: httpStatus === 529 || failureText.includes("overloaded_error"),
3442
+ signal: parseAnthropicSlowModeHeaders(getHeadersFromError(streamFailure)),
3443
+ sentSlow,
3444
+ waitedMs: slowWaitSinceMs === undefined ? 0 : Date.now() - slowWaitSinceMs,
3445
+ attempts: slowWaitAttempts,
3446
+ });
3447
+ if (slowRetry) {
3448
+ if (slowRetry.capacityWait) {
3449
+ slowWaitSinceMs ??= Date.now();
3450
+ slowWaitAttempts++;
3451
+ }
3452
+ logger.debug("anthropic: slow mode retry", {
3453
+ model: model.id,
3454
+ status: httpStatus,
3455
+ delayMs: slowRetry.delayMs,
3456
+ attempt: slowWaitAttempts,
3457
+ });
3458
+ if (slowRetry.delayMs > 0) {
3459
+ if (options?.providerRetryWait) {
3460
+ await options.providerRetryWait(slowRetry.delayMs, options.signal);
3461
+ } else {
3462
+ await scheduler.wait(slowRetry.delayMs, { signal: options?.signal });
3463
+ }
3464
+ }
3465
+ resetStreamOutputState();
3466
+ continue;
3467
+ }
3468
+ }
3178
3469
  const isTransientEnvelopeFailure =
3179
3470
  AIError.isTransientStreamParseError(streamFailure) || AIError.isStreamEnvelopeError(streamFailure);
3180
3471
  const isLocalIdleTimeout =
@@ -3223,6 +3514,15 @@ const streamAnthropicOnce = (
3223
3514
  firstTokenTime = undefined;
3224
3515
  }
3225
3516
  }
3517
+ if (
3518
+ usingFallbackCredit &&
3519
+ fallbackCreditShape === "continuation" &&
3520
+ options?.fallbackCreditRedemption?.refusedContent
3521
+ ) {
3522
+ // The response continues the echoed prefix; keep it in the stored turn in
3523
+ // AssistantMessage form so later replays keep signatures and server tools.
3524
+ output.content.unshift(...refusalContinuationPrefix(options.fallbackCreditRedemption.refusedContent));
3525
+ }
3226
3526
  output.duration = performance.now() - startTime;
3227
3527
  if (firstTokenTime) output.ttft = firstTokenTime - startTime;
3228
3528
  if (dropFastMode && model.provider === "anthropic" && options?.serviceTier === "priority") {
@@ -3530,10 +3830,10 @@ export function buildAnthropicClientOptions(args: AnthropicClientOptionsArgs): A
3530
3830
  function createClient(
3531
3831
  model: Model<"anthropic-messages">,
3532
3832
  args: AnthropicClientOptionsArgs,
3533
- ): { client: AnthropicMessagesClient; isOAuthToken: boolean } {
3833
+ ): { client: AnthropicMessagesClient; isOAuthToken: boolean; defaultHeaders?: Record<string, string> } {
3534
3834
  const { isOAuthToken: oauthToken, ...clientOptions } = buildAnthropicClientOptions({ ...args, model });
3535
3835
  const client = new AnthropicMessagesClient(clientOptions);
3536
- return { client, isOAuthToken: oauthToken };
3836
+ return { client, isOAuthToken: oauthToken, defaultHeaders: clientOptions.defaultHeaders };
3537
3837
  }
3538
3838
 
3539
3839
  /** The compaction request is a standalone summary call, not a generation turn. */
@@ -104,18 +104,25 @@ function responseHeaders(message: IncomingMessage): Headers {
104
104
  function decodedResponseStream(message: IncomingMessage): stream.Readable {
105
105
  const rawEncoding = message.headers["content-encoding"];
106
106
  const encoding = (Array.isArray(rawEncoding) ? rawEncoding[0] : rawEncoding)?.trim().toLowerCase();
107
+ let decoder: stream.Transform;
107
108
  switch (encoding) {
108
109
  case "gzip":
109
- return message.pipe(zlib.createGunzip());
110
+ decoder = zlib.createGunzip();
111
+ break;
110
112
  case "deflate":
111
- return message.pipe(zlib.createInflate());
113
+ decoder = zlib.createInflate();
114
+ break;
112
115
  case "br":
113
- return message.pipe(zlib.createBrotliDecompress());
116
+ decoder = zlib.createBrotliDecompress();
117
+ break;
114
118
  case "zstd":
115
- return message.pipe(zlib.createZstdDecompress());
119
+ decoder = zlib.createZstdDecompress();
120
+ break;
116
121
  default:
117
122
  return message;
118
123
  }
124
+ // Couple decoded-body cancellation to the source so its keep-alive socket cannot be stranded.
125
+ return stream.pipeline(message, decoder, () => {});
119
126
  }
120
127
 
121
128
  function createResponse(message: IncomingMessage, method: string): Response {
@@ -617,7 +617,13 @@ export async function consumeGoogleStream<T extends GoogleApiType>(args: {
617
617
 
618
618
  for await (const chunk of googleStream) {
619
619
  if (chunk.error) {
620
- const detail = chunk.error.message || chunk.error.status || "unknown error";
620
+ // Keep the RPC status alongside the message: an in-band quota failure
621
+ // is classified from this text, and `RESOURCE_EXHAUSTED` is the only
622
+ // account-exhaustion signal some of these chunks carry (#13090).
623
+ const detail =
624
+ chunk.error.message && chunk.error.status
625
+ ? `${chunk.error.message} (${chunk.error.status})`
626
+ : chunk.error.message || chunk.error.status || "unknown error";
621
627
  const message = `Google API stream error: ${detail}`;
622
628
  throw typeof chunk.error.code === "number" && chunk.error.code >= 400
623
629
  ? new AIError.GoogleApiError(message, chunk.error.code)
@@ -969,7 +975,7 @@ export function streamGoogleGenAI<T extends "google-generative-ai" | "google-ver
969
975
  if (!response.ok) {
970
976
  const errorText = await response.text().catch(() => "");
971
977
  throw new AIError.GoogleApiError(
972
- `Google API error (${response.status}): ${extractGoogleErrorMessage(errorText)}`,
978
+ `Google API error (${response.status}): ${extractGoogleErrorMessage(errorText, response.status)}`,
973
979
  response.status,
974
980
  { headers: response.headers },
975
981
  );
@@ -1101,13 +1107,30 @@ function paramsToWireBody(params: GenerateContentParameters): Record<string, unk
1101
1107
  return body;
1102
1108
  }
1103
1109
 
1104
- function extractGoogleErrorMessage(errorText: string): string {
1110
+ /**
1111
+ * Human-readable message for a non-2xx Google response.
1112
+ *
1113
+ * On a usage-limit status the RPC `status`/`details` residue is kept after the
1114
+ * message: `parseGoogleRpcRateLimitReason` reads `RESOURCE_EXHAUSTED` plus the
1115
+ * `google.rpc.ErrorInfo` reason to tell an account billing cap (terminal) from
1116
+ * a per-minute throttle (retryable), and reducing the body to `error.message`
1117
+ * hid both, so every billing 429 replayed as a transient rate limit (#13090).
1118
+ * The Cloud Code Assist path keeps the whole raw body for the same reason.
1119
+ */
1120
+ function extractGoogleErrorMessage(errorText: string, status: number): string {
1105
1121
  if (!errorText) return "Unknown error";
1106
1122
  try {
1107
- const parsed = JSON.parse(errorText) as { error?: { message?: string } };
1108
- if (parsed.error?.message) return parsed.error.message;
1123
+ const parsed = JSON.parse(errorText) as {
1124
+ error?: { message?: string; status?: string; details?: unknown[] };
1125
+ };
1126
+ const error = parsed.error;
1127
+ if (!error?.message) return errorText;
1128
+ if (!AIError.isUsageLimitStatus(status)) return error.message;
1129
+ const residue = { error: { status: error.status, details: error.details } };
1130
+ return error.status === undefined && error.details === undefined
1131
+ ? error.message
1132
+ : `${error.message} ${JSON.stringify(residue)}`;
1109
1133
  } catch {
1110
- // fall through to raw text
1134
+ return errorText;
1111
1135
  }
1112
- return errorText;
1113
1136
  }
@@ -1,6 +1,6 @@
1
1
  /** Shared inference request identity headers. */
2
2
 
3
- import { USER_AGENT } from "@oh-my-pi/pi-utils";
3
+ import { APP_NAME, APP_URL, USER_AGENT } from "@oh-my-pi/pi-utils";
4
4
 
5
5
  /** Options controlling provider and protocol inference headers. */
6
6
  export interface InferenceHeaderOptions {
@@ -33,6 +33,12 @@ function setHeader(headers: Record<string, string>, name: string, value: string)
33
33
  * understood by the active inference protocol and host.
34
34
  */
35
35
  export function applyInferenceHeaders(headers: Record<string, string>, options: InferenceHeaderOptions): void {
36
+ if (options.provider === "vercel-ai-gateway") {
37
+ // Vercel AI Gateway app attribution; caller/config headers take precedence.
38
+ setHeaderIfAbsent(headers, "http-referer", APP_URL);
39
+ setHeaderIfAbsent(headers, "x-title", APP_NAME);
40
+ }
41
+
36
42
  const isOpenCode = options.provider === "opencode-go" || options.provider === "opencode-zen";
37
43
  const sessionId = options.sessionId;
38
44
  if (!sessionId) return;
@@ -46,6 +46,7 @@ import { classifyModel } from "@oh-my-pi/pi-catalog/compat/taxonomy";
46
46
  import { registerCustomApi } from "../api-registry";
47
47
  import * as AIError from "../error";
48
48
  import type {
49
+ AnthropicFallbackCreditHandle,
49
50
  Api,
50
51
  AssistantMessage,
51
52
  Context,
@@ -86,6 +87,8 @@ export interface MockResponse {
86
87
  stopReason?: StopReason;
87
88
  /** Structured terminal stop classification, e.g. Anthropic refusal metadata. */
88
89
  stopDetails?: StopDetails | null;
90
+ /** In-memory fallback credit handle attached when a refusal response carries a fallback credit token. */
91
+ fallbackCreditHandle?: AnthropicFallbackCreditHandle;
89
92
  /** Error text paired with an explicit `"error"` stop reason. */
90
93
  errorMessage?: string;
91
94
  /** Usage stats. Missing fields default to 0; missing `cost.total` is recomputed from components. */
@@ -403,6 +406,7 @@ async function runMock(
403
406
 
404
407
  partial.stopReason = reason;
405
408
  partial.stopDetails = response.stopDetails;
409
+ partial.fallbackCreditHandle = response.fallbackCreditHandle;
406
410
  partial.errorMessage = response.errorMessage;
407
411
  partial.usage = mergeUsage(response.usage);
408
412
  partial.duration = performance.now() - perfStart;