@oh-my-pi/pi-ai 18.2.8 → 18.2.9

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (45) hide show
  1. package/CHANGELOG.md +33 -18
  2. package/dist/types/auth-gateway/types.d.ts +5 -0
  3. package/dist/types/auth-storage.d.ts +35 -32
  4. package/dist/types/index.d.ts +1 -0
  5. package/dist/types/providers/amazon-bedrock.d.ts +7 -0
  6. package/dist/types/providers/claude-code-fingerprint.d.ts +22 -3
  7. package/dist/types/providers/google-gemini-cli.d.ts +0 -2
  8. package/dist/types/providers/openai-chat-server-schema.d.ts +2 -2
  9. package/dist/types/usage/claude-api.d.ts +22 -0
  10. package/dist/types/usage/claude-reset.d.ts +44 -0
  11. package/dist/types/usage.d.ts +111 -5
  12. package/dist/types/utils/schema/json-schema-validator.d.ts +5 -2
  13. package/dist/types/utils/tool-call-loop-guard.d.ts +1 -1
  14. package/package.json +6 -6
  15. package/src/auth/sqlite-credential-store.ts +44 -1
  16. package/src/auth-broker/remote-store.ts +6 -6
  17. package/src/auth-broker/wire-schemas.ts +14 -0
  18. package/src/auth-gateway/server.ts +4 -0
  19. package/src/auth-gateway/types.ts +5 -0
  20. package/src/auth-storage.ts +263 -140
  21. package/src/error/flags.ts +10 -0
  22. package/src/index.ts +1 -0
  23. package/src/providers/amazon-bedrock.ts +55 -5
  24. package/src/providers/anthropic.ts +45 -11
  25. package/src/providers/aws-credentials.ts +124 -11
  26. package/src/providers/claude-code-fingerprint.ts +55 -3
  27. package/src/providers/gitlab-duo.ts +20 -4
  28. package/src/providers/google-gemini-cli.ts +0 -8
  29. package/src/providers/google-shared.ts +1 -18
  30. package/src/providers/openai-chat-server-schema.ts +1 -1
  31. package/src/providers/openai-chat-server.ts +3 -1
  32. package/src/providers/openai-codex-responses.ts +27 -5
  33. package/src/providers/openai-completions.ts +122 -19
  34. package/src/providers/pi-native-server.ts +1 -0
  35. package/src/registry/oauth/anthropic.ts +2 -3
  36. package/src/stream.ts +13 -4
  37. package/src/usage/alibaba-token-plan.ts +7 -1
  38. package/src/usage/claude-api.ts +66 -0
  39. package/src/usage/claude-reset.ts +638 -0
  40. package/src/usage/claude.ts +37 -59
  41. package/src/usage/kimi.ts +32 -1
  42. package/src/usage.ts +52 -5
  43. package/src/utils/schema/json-schema-validator.ts +23 -10
  44. package/src/utils/tool-call-loop-guard.ts +2 -2
  45. package/src/utils/validation.ts +145 -50
@@ -71,6 +71,43 @@ const SIGNER_OWNED_HEADERS = new Set(["host", "x-amz-date", "x-amz-content-sha25
71
71
  // mismatch.
72
72
  const BEDROCK_RESERVED_HEADERS = new Set(["content-type", "accept", "authorization", "content-length"]);
73
73
 
74
+ /**
75
+ * HTTP status for a ConverseStream in-stream failure, keyed by lowercased
76
+ * exception shape name. Exception and error frames ride inside an HTTP 200
77
+ * event stream, so the shape name in `:exception-type` / `:error-code` is the
78
+ * only evidence of what actually failed upstream. Values come from the
79
+ * bedrock-runtime service model (the same source the AWS SDKs deserialize
80
+ * against): without them every in-stream failure would be stamped 400, which
81
+ * the retry classifier reads as a deterministic client rejection and refuses
82
+ * to replay — making a transient `internalServerException` (500) terminal.
83
+ * Shapes absent from the map keep 400 so an unrecognized rejection is never
84
+ * retried by accident.
85
+ */
86
+ const BEDROCK_STREAM_EXCEPTION_STATUS: Record<string, number> = {
87
+ accessdeniedexception: 403,
88
+ conflictexception: 400,
89
+ internalserverexception: 500,
90
+ modelerrorexception: 424,
91
+ modelnotreadyexception: 429,
92
+ modelstreamerrorexception: 424,
93
+ modeltimeoutexception: 408,
94
+ resourcenotfoundexception: 404,
95
+ servicequotaexceededexception: 400,
96
+ serviceunavailableexception: 503,
97
+ throttlingexception: 429,
98
+ validationexception: 400,
99
+ };
100
+
101
+ /**
102
+ * Resolve the service-model status for an in-stream exception/error code.
103
+ * Frame headers carry the bare shape name in either camelCase
104
+ * (`internalServerException`) or PascalCase (`InternalServerException`); both
105
+ * normalize to the same map key. Unknown shapes default to 400.
106
+ */
107
+ export function bedrockStreamExceptionStatus(code: string): number {
108
+ return BEDROCK_STREAM_EXCEPTION_STATUS[code.trim().toLowerCase()] ?? 400;
109
+ }
110
+
74
111
  export type BedrockThinkingDisplay = "summarized" | "omitted";
75
112
 
76
113
  /** Bedrock guardrail trace verbosity, mirrors the Converse `guardrailConfig.trace` values. */
@@ -260,7 +297,7 @@ interface WireMessage {
260
297
  }
261
298
 
262
299
  interface WireToolSpec {
263
- toolSpec: { name: string; description: string; inputSchema: { json: unknown } };
300
+ toolSpec: { name: string; description?: string; inputSchema: { json: unknown } };
264
301
  }
265
302
  interface WireToolChoice {
266
303
  auto?: Record<string, never>;
@@ -605,12 +642,16 @@ export const streamBedrock: StreamFunction<"bedrock-converse-stream"> = (
605
642
  const payload = safeParsePayload(message.payload) as { message?: string } | undefined;
606
643
  const errorMessage = payload?.message || new TextDecoder().decode(message.payload);
607
644
  const text = `${exceptionType}: ${errorMessage}`;
608
- throw new AIError.BedrockApiError(text, 400, { code: exceptionType });
645
+ throw new AIError.BedrockApiError(text, bedrockStreamExceptionStatus(exceptionType), {
646
+ code: exceptionType,
647
+ });
609
648
  }
610
649
  if (messageType === "error") {
611
650
  const code = message.headers[":error-code"] || "UnknownError";
612
651
  const errorMessage = message.headers[":error-message"] || new TextDecoder().decode(message.payload);
613
- throw new AIError.BedrockApiError(`${code}: ${errorMessage}`, 400, { code });
652
+ throw new AIError.BedrockApiError(`${code}: ${errorMessage}`, bedrockStreamExceptionStatus(code), {
653
+ code,
654
+ });
614
655
  }
615
656
  if (messageType !== "event") continue;
616
657
 
@@ -947,10 +988,17 @@ function buildToolResultBlock(
947
988
  hoistedImages: ImageBlockWire[],
948
989
  ): ToolResultBlockWire {
949
990
  const content: Array<TextBlockWire | ImageBlockWire> = [];
991
+ // Bedrock's Anthropic Claude models reject an error toolResult that carries a
992
+ // non-text block ("all content must be type `text` if `is_error` is true"),
993
+ // so images inside an error result must always be hoisted out regardless of
994
+ // the model's requiresToolResultImageHoisting flag (no `class "anthropic"`
995
+ // rule sets it). Mirrors anthropic.ts buildToolResultBlock. Re-serializing a
996
+ // previously-persisted poisoned result on a later turn repairs it in place.
997
+ const hoistImages = message.isError || model.requiresToolResultImageHoisting;
950
998
  for (const block of message.content) {
951
999
  if (block.type === "image") {
952
1000
  const image: ImageBlockWire = { image: createImageBlock(block.mimeType, block.data) };
953
- if (model.requiresToolResultImageHoisting) {
1001
+ if (hoistImages) {
954
1002
  content.push({ text: "(see attached image)" });
955
1003
  hoistedImages.push(image);
956
1004
  } else {
@@ -1114,7 +1162,9 @@ function convertToolSpec(tool: Tool): WireToolSpec {
1114
1162
  return {
1115
1163
  toolSpec: {
1116
1164
  name: tool.name,
1117
- description: tool.description || "",
1165
+ // Descriptions may be pruned into the system prompt. Bedrock permits
1166
+ // omission, but rejects an explicitly empty description (minLength: 1).
1167
+ description: tool.description || undefined,
1118
1168
  inputSchema: { json: toolWireSchema(tool) },
1119
1169
  },
1120
1170
  };
@@ -110,9 +110,10 @@ import {
110
110
  CLAUDE_CODE_MAX_OUTPUT_TOKENS,
111
111
  claudeCodeSdkVersion,
112
112
  claudeCodeSystemInstruction,
113
- claudeCodeVersion,
113
+ adoptRequiredClaudeCodeVersion,
114
114
  claudeToolPrefix,
115
- claudeCodeUserAgent,
115
+ getClaudeCodeUserAgent,
116
+ getClaudeCodeVersion,
116
117
  } from "./claude-code-fingerprint";
117
118
  import {
118
119
  buildCopilotDynamicHeaders,
@@ -376,7 +377,7 @@ export function buildAnthropicHeaders(options: AnthropicHeaderOptions): Record<s
376
377
  }
377
378
 
378
379
  if (oauthToken) {
379
- const userAgent = isClaudeCodeClientUserAgent(incomingUserAgent) ? incomingUserAgent : claudeCodeUserAgent;
380
+ const userAgent = isClaudeCodeClientUserAgent(incomingUserAgent) ? incomingUserAgent : getClaudeCodeUserAgent();
380
381
  const headers = {
381
382
  ...modelHeaders,
382
383
  Accept: acceptHeader,
@@ -665,14 +666,11 @@ function createClaudeBillingHeader(firstUserMessageText: string): string {
665
666
  // Matches CC's computeFingerprint in utils/fingerprint.ts.
666
667
  // Uses chars from the first user message (not the system prompt).
667
668
  const k = [4, 7, 20].map(i => firstUserMessageText[i] ?? "0").join("");
668
- const versionSuffix = nodeCrypto
669
- .createHash("sha256")
670
- .update(`59cf53e54c78${k}${claudeCodeVersion}`)
671
- .digest("hex")
672
- .slice(0, 3);
669
+ const version = getClaudeCodeVersion();
670
+ const versionSuffix = nodeCrypto.createHash("sha256").update(`59cf53e54c78${k}${version}`).digest("hex").slice(0, 3);
673
671
  // cch=00000: placeholder replaced with the real attestation hash by wrapFetchForCch
674
672
  // before the request hits the wire (see below).
675
- return `${CLAUDE_BILLING_HEADER_PREFIX} cc_version=${claudeCodeVersion}.${versionSuffix}; cc_entrypoint=cli; ${CCH_PLACEHOLDER_STR};`;
673
+ return `${CLAUDE_BILLING_HEADER_PREFIX} cc_version=${version}.${versionSuffix}; cc_entrypoint=cli; ${CCH_PLACEHOLDER_STR};`;
676
674
  }
677
675
 
678
676
  // cch attestation: XXHash64(body_with_placeholder, seed) low-20-bits, 5 hex chars.
@@ -2085,6 +2083,8 @@ const streamAnthropicOnce = (
2085
2083
  const zeroOutputCacheRefresh = options?.anthropicCacheRefreshRequest === true;
2086
2084
  let client: AnthropicMessagesClientLike;
2087
2085
  let isOAuthToken: boolean;
2086
+ // Retained so a Claude Code version bump can rebuild the client's fingerprint headers.
2087
+ let clientArgs: AnthropicClientOptionsArgs | undefined;
2088
2088
 
2089
2089
  if (options?.client) {
2090
2090
  client = options.client;
@@ -2190,7 +2190,7 @@ const streamAnthropicOnce = (
2190
2190
  }
2191
2191
  }
2192
2192
 
2193
- const created = createClient(model, {
2193
+ clientArgs = {
2194
2194
  model,
2195
2195
  apiKey,
2196
2196
  extraBetas,
@@ -2211,7 +2211,8 @@ const streamAnthropicOnce = (
2211
2211
  extractClaudeMetadataSessionId(options?.metadata?.user_id) ??
2212
2212
  options?.promptCacheKey,
2213
2213
  disableStrictTools,
2214
- });
2214
+ };
2215
+ const created = createClient(model, clientArgs);
2215
2216
  client = created.client;
2216
2217
  isOAuthToken = created.isOAuthToken;
2217
2218
  }
@@ -2260,6 +2261,15 @@ const streamAnthropicOnce = (
2260
2261
 
2261
2262
  if (zeroOutputCacheRefresh) {
2262
2263
  const refreshParams: MessageCreateParams = { ...params, max_tokens: 0, stream: false };
2264
+ // Anthropic rejects `tool_choice: {type:"tool"|"any"}` with `max_tokens: 0`
2265
+ // ("tool_choice ... cannot be used when max_tokens is 0", #12597). A refresh
2266
+ // replays the captured turn's payload, which can carry a forced selector
2267
+ // (e.g. a forced yield). A zero-output keep-alive produces no tokens, so the
2268
+ // forced choice is meaningless here — drop it so the request is accepted.
2269
+ const refreshChoiceType = refreshParams.tool_choice?.type;
2270
+ if (refreshChoiceType === "tool" || refreshChoiceType === "any") {
2271
+ delete refreshParams.tool_choice;
2272
+ }
2263
2273
  rawRequestDump = {
2264
2274
  provider: model.provider,
2265
2275
  api: output.api,
@@ -3064,6 +3074,30 @@ const streamAnthropicOnce = (
3064
3074
  }
3065
3075
  const streamFailureMessage =
3066
3076
  streamFailure instanceof Error ? streamFailure.message : String(streamFailure);
3077
+ if (
3078
+ isOAuthToken &&
3079
+ clientArgs &&
3080
+ firstTokenTime === undefined &&
3081
+ adoptRequiredClaudeCodeVersion(streamFailure)
3082
+ ) {
3083
+ logger.warn("anthropic: Claude Code version rejected as too old, retrying with required version", {
3084
+ model: model.id,
3085
+ version: getClaudeCodeVersion(),
3086
+ });
3087
+ client = createClient(model, { ...clientArgs, disableStrictTools }).client;
3088
+ params = await prepareParams();
3089
+ providerRetryAttempt = 0;
3090
+ output.content.length = 0;
3091
+ output.model = model.id;
3092
+ output.responseId = undefined;
3093
+ output.upstreamModel = undefined;
3094
+ output.errorMessage = undefined;
3095
+ output.providerPayload = undefined;
3096
+ output.usage = createEmptyUsage(copilotDynamicHeaders?.premiumRequests);
3097
+ output.stopReason = "stop";
3098
+ firstTokenTime = undefined;
3099
+ continue;
3100
+ }
3067
3101
  if (
3068
3102
  !prefixBindingRetryAttempted &&
3069
3103
  options?.anthropicPrefixMismatchBehavior !== "error" &&
@@ -407,8 +407,23 @@ interface SsoCachedToken {
407
407
  expiresAt?: string;
408
408
  startUrl?: string;
409
409
  region?: string;
410
+ /** Present when the token was minted with the refresh_token grant enabled. */
411
+ refreshToken?: string;
412
+ clientId?: string;
413
+ clientSecret?: string;
414
+ /** Client registration expiry; refresh is impossible once this passes. */
415
+ registrationExpiresAt?: string;
410
416
  }
411
417
 
418
+ /** A cache hit plus the filename it came from, so a refresh can be written back. */
419
+ interface SsoCacheEntry {
420
+ token: SsoCachedToken;
421
+ file: string;
422
+ }
423
+
424
+ /** Refresh this long before `expiresAt` so a request in flight cannot age out. */
425
+ const SSO_TOKEN_REFRESH_SKEW_MS = 60_000;
426
+
412
427
  async function readSsoCredentials(
413
428
  profileCfg: Record<string, string>,
414
429
  configIni: AwsIniFile | undefined,
@@ -431,19 +446,29 @@ async function readSsoCredentials(
431
446
  }
432
447
  if (!startUrl || !ssoRegion) return undefined;
433
448
 
434
- const token = await loadSsoCachedToken(startUrl, sessionName);
435
- if (!token?.accessToken) {
449
+ const cached = await loadSsoCachedToken(startUrl, sessionName);
450
+ if (!cached?.token.accessToken) {
436
451
  throw new AIError.AwsCredentialsError(
437
452
  `AWS SSO token for ${startUrl} not found in ~/.aws/sso/cache. Run 'aws sso login' first.`,
438
453
  "sso-token-missing",
439
454
  );
440
455
  }
441
- const expiresAt = token.expiresAt ? Date.parse(token.expiresAt) : Number.POSITIVE_INFINITY;
442
- if (Number.isFinite(expiresAt) && expiresAt <= Date.now()) {
443
- throw new AIError.AwsCredentialsError(
444
- `AWS SSO token for ${startUrl} has expired. Run 'aws sso login' to refresh.`,
445
- "sso-token-expired",
446
- );
456
+ let accessToken = cached.token.accessToken;
457
+ const expiresAt = cached.token.expiresAt ? Date.parse(cached.token.expiresAt) : Number.POSITIVE_INFINITY;
458
+ const expired = Number.isFinite(expiresAt) && expiresAt <= Date.now();
459
+ // Access tokens are short-lived (often 1 h) but ship with a refresh token whose
460
+ // client registration lasts weeks. The AWS CLI refreshes transparently, so a
461
+ // profile that works under `aws` must not fail here; only a genuinely
462
+ // unrefreshable token warrants sending the user back to `aws sso login`.
463
+ if (Number.isFinite(expiresAt) && expiresAt - SSO_TOKEN_REFRESH_SKEW_MS <= Date.now()) {
464
+ const refreshed = await refreshSsoToken(cached.token, cached.file, ssoRegion, signal, fetchImpl);
465
+ if (refreshed?.accessToken) accessToken = refreshed.accessToken;
466
+ else if (expired) {
467
+ throw new AIError.AwsCredentialsError(
468
+ `AWS SSO token for ${startUrl} has expired. Run 'aws sso login' to refresh.`,
469
+ "sso-token-expired",
470
+ );
471
+ }
447
472
  }
448
473
 
449
474
  const url =
@@ -452,7 +477,7 @@ async function readSsoCredentials(
452
477
  `&role_name=${encodeURIComponent(profileCfg.sso_role_name)}`;
453
478
  const response = await fetchImpl(url, {
454
479
  method: "GET",
455
- headers: { "x-amz-sso_bearer_token": token.accessToken },
480
+ headers: { "x-amz-sso_bearer_token": accessToken },
456
481
  signal,
457
482
  });
458
483
  if (!response.ok) {
@@ -487,7 +512,7 @@ async function readSsoCredentials(
487
512
  async function loadSsoCachedToken(
488
513
  startUrl: string,
489
514
  sessionName: string | undefined,
490
- ): Promise<SsoCachedToken | undefined> {
515
+ ): Promise<SsoCacheEntry | undefined> {
491
516
  const cacheDir = path.join(os.homedir(), ".aws", "sso", "cache");
492
517
  let entries: string[];
493
518
  try {
@@ -510,7 +535,7 @@ async function loadSsoCachedToken(
510
535
  const text = await fs.promises.readFile(path.join(cacheDir, file), "utf8");
511
536
  const parsed = JSON.parse(text) as SsoCachedToken;
512
537
  if (parsed.startUrl === startUrl || (sessionName && file === `${hash}.json`)) {
513
- return parsed;
538
+ return { token: parsed, file };
514
539
  }
515
540
  } catch (err) {
516
541
  logger.debug("aws-credentials: failed to read SSO cache", { file, err: String(err) });
@@ -519,6 +544,94 @@ async function loadSsoCachedToken(
519
544
  return undefined;
520
545
  }
521
546
 
547
+ /**
548
+ * Exchange the cached refresh token for a fresh access token via SSO OIDC
549
+ * `CreateToken`, which is what the AWS CLI does transparently on every command.
550
+ *
551
+ * Returns `undefined` when refresh is impossible (no refresh grant material, or
552
+ * the client registration itself has expired) or when the exchange fails, so a
553
+ * broken refresh surfaces the existing "run aws sso login" remedy rather than an
554
+ * opaque network error.
555
+ */
556
+ async function refreshSsoToken(
557
+ token: SsoCachedToken,
558
+ file: string,
559
+ ssoRegion: string,
560
+ signal: AbortSignal | undefined,
561
+ fetchImpl: FetchImpl,
562
+ ): Promise<SsoCachedToken | undefined> {
563
+ if (!token.refreshToken || !token.clientId || !token.clientSecret) return undefined;
564
+ const registrationExpiresAt = token.registrationExpiresAt ? Date.parse(token.registrationExpiresAt) : Number.NaN;
565
+ if (Number.isFinite(registrationExpiresAt) && registrationExpiresAt <= Date.now()) {
566
+ logger.debug("aws-credentials: SSO client registration expired; cannot refresh");
567
+ return undefined;
568
+ }
569
+
570
+ let response: Response;
571
+ try {
572
+ response = await fetchImpl(`https://oidc.${ssoRegion}.amazonaws.com/token`, {
573
+ method: "POST",
574
+ headers: { "content-type": "application/json" },
575
+ body: JSON.stringify({
576
+ clientId: token.clientId,
577
+ clientSecret: token.clientSecret,
578
+ grantType: "refresh_token",
579
+ refreshToken: token.refreshToken,
580
+ }),
581
+ signal,
582
+ });
583
+ } catch (err) {
584
+ logger.debug("aws-credentials: SSO token refresh request failed", { err: String(err) });
585
+ return undefined;
586
+ }
587
+ if (!response.ok) {
588
+ const body = await response.text().catch(() => "");
589
+ logger.debug("aws-credentials: SSO token refresh rejected", {
590
+ status: response.status,
591
+ body: body.slice(0, 200),
592
+ });
593
+ return undefined;
594
+ }
595
+ const json = (await response.json().catch(() => undefined)) as
596
+ | { accessToken?: string; expiresIn?: number; refreshToken?: string }
597
+ | undefined;
598
+ if (!json?.accessToken) {
599
+ logger.debug("aws-credentials: SSO token refresh returned no accessToken");
600
+ return undefined;
601
+ }
602
+
603
+ const updated: SsoCachedToken = {
604
+ ...token,
605
+ accessToken: json.accessToken,
606
+ // `expiresIn` is seconds from now; the cache records an absolute instant.
607
+ // Trim milliseconds to match the format the AWS CLI writes.
608
+ expiresAt: new Date(Date.now() + (json.expiresIn ?? 0) * 1000).toISOString().replace(/\.\d{3}Z$/, "Z"),
609
+ // The service may rotate the refresh token; persisting the new one keeps
610
+ // the following refresh working.
611
+ refreshToken: json.refreshToken ?? token.refreshToken,
612
+ };
613
+ await writeSsoCachedToken(file, updated);
614
+ return updated;
615
+ }
616
+
617
+ /**
618
+ * Persist a refreshed token so the AWS CLI, other SDKs, and the next OMP process
619
+ * all start from a live token. Written via temp file + rename so a concurrent
620
+ * reader never observes a half-written cache entry; a failure here is logged and
621
+ * ignored, since the in-memory token is still usable for this run.
622
+ */
623
+ async function writeSsoCachedToken(file: string, token: SsoCachedToken): Promise<void> {
624
+ const target = path.join(os.homedir(), ".aws", "sso", "cache", file);
625
+ const tmp = `${target}.${process.pid}.tmp`;
626
+ try {
627
+ await fs.promises.writeFile(tmp, JSON.stringify(token), { mode: 0o600 });
628
+ await fs.promises.rename(tmp, target);
629
+ } catch (err) {
630
+ logger.debug("aws-credentials: failed to persist refreshed SSO token", { file, err: String(err) });
631
+ await fs.promises.rm(tmp, { force: true }).catch(() => {});
632
+ }
633
+ }
634
+
522
635
  async function sha1Hex(input: string): Promise<string> {
523
636
  const digest = await globalThis.crypto.subtle.digest("SHA-1", new TextEncoder().encode(input));
524
637
  const bytes = new Uint8Array(digest);
@@ -8,12 +8,64 @@
8
8
  * provider module.
9
9
  */
10
10
 
11
- /** Current Claude Code CLI version represented on the Anthropic wire. */
12
- export const claudeCodeVersion = "2.1.257";
11
+ /**
12
+ * Pinned Claude Code CLI version: the offline fallback for {@link getClaudeCodeVersion}.
13
+ * Bumped to the latest npm release by `bun run check-spoofed-versions --update`.
14
+ */
15
+ export const DEFAULT_CLAUDE_CODE_VERSION = "2.1.280";
13
16
  /** `@anthropic-ai/sdk` version bundled by the current Claude Code release. */
14
17
  export const claudeCodeSdkVersion = "0.112.1";
18
+
19
+ const SEMVER_PATTERN = /^(\d+)\.(\d+)\.(\d+)$/;
20
+ const VERSION_TOO_OLD_CODE = "claude_code_version_too_old";
21
+ const REQUIRED_VERSION_PATTERN = /version (\d+\.\d+\.\d+) or newer is required/i;
22
+
23
+ let adoptedClaudeCodeVersion: string | null = null;
24
+
25
+ /**
26
+ * Claude Code CLI version represented on the Anthropic wire (User-Agent, billing
27
+ * header): `PI_AI_CLAUDE_CODE_VERSION` → version adopted from a server rejection
28
+ * (see {@link adoptRequiredClaudeCodeVersion}) → {@link DEFAULT_CLAUDE_CODE_VERSION}.
29
+ */
30
+ export function getClaudeCodeVersion(): string {
31
+ return process.env.PI_AI_CLAUDE_CODE_VERSION || adoptedClaudeCodeVersion || DEFAULT_CLAUDE_CODE_VERSION;
32
+ }
33
+
15
34
  /** User-Agent emitted by Claude Code's CLI inference entrypoint. */
16
- export const claudeCodeUserAgent = `claude-cli/${claudeCodeVersion} (external, cli)`;
35
+ export function getClaudeCodeUserAgent(): string {
36
+ return `claude-cli/${getClaudeCodeVersion()} (external, cli)`;
37
+ }
38
+
39
+ function compareSemver(a: string, b: string): number {
40
+ const pa = SEMVER_PATTERN.exec(a);
41
+ const pb = SEMVER_PATTERN.exec(b);
42
+ if (!pa || !pb) return 0;
43
+ for (let i = 1; i <= 3; i++) {
44
+ const diff = Number(pa[i]) - Number(pb[i]);
45
+ if (diff !== 0) return diff;
46
+ }
47
+ return 0;
48
+ }
49
+
50
+ /**
51
+ * Adopts the minimum version named by an Anthropic `claude_code_version_too_old`
52
+ * rejection for the rest of the process, so the pinned fallback going stale costs
53
+ * one rejected request instead of a hard failure.
54
+ *
55
+ * Returns true only when the wire version actually increased — callers retry on
56
+ * true, so a repeated rejection at the same version cannot loop. Always false
57
+ * while `PI_AI_CLAUDE_CODE_VERSION` pins the version explicitly.
58
+ */
59
+ export function adoptRequiredClaudeCodeVersion(error: unknown): boolean {
60
+ if (process.env.PI_AI_CLAUDE_CODE_VERSION) return false;
61
+ const message = error instanceof Error ? error.message : String(error);
62
+ if (!message.includes(VERSION_TOO_OLD_CODE)) return false;
63
+ const required = REQUIRED_VERSION_PATTERN.exec(message)?.[1];
64
+ if (!required || compareSemver(required, getClaudeCodeVersion()) <= 0) return false;
65
+ adoptedClaudeCodeVersion = required;
66
+ return true;
67
+ }
68
+
17
69
  /** Prefix used to isolate custom Anthropic OAuth tools from built-in tools. */
18
70
  export const claudeToolPrefix: string = "_";
19
71
  /** Identity block prepended by Claude Code's CLI runtime. */
@@ -123,7 +123,17 @@ export function streamGitLabDuo(
123
123
  ...options.headers,
124
124
  };
125
125
 
126
+ // This wrapper dispatches directly to the routed provider and bypasses
127
+ // mapOptionsForApi(), so preserve the shared reasoning contracts here as
128
+ // well. The anthropic-messages route derives thinking on/off from the
129
+ // effort itself, so fold the explicit off into a cleared effort — capped
130
+ // side turns rely on this to keep Anthropic from raising max_tokens for a
131
+ // thinking budget. The OpenAI routes take the flags themselves, mirroring
132
+ // mapOptionsForApi(), and keep the requested effort for their own
133
+ // off/fallback handling.
126
134
  const reasoningEffort = options.reasoning;
135
+ const anthropicReasoningEffort =
136
+ options.disableReasoning || options.forceReasoningOff ? undefined : options.reasoning;
127
137
 
128
138
  const inner =
129
139
  route.api === "anthropic-messages"
@@ -158,11 +168,12 @@ export function streamGitLabDuo(
158
168
  onResponse: options.onResponse,
159
169
  onSseEvent: options.onSseEvent,
160
170
  fetch: options.fetch,
161
- thinkingEnabled: Boolean(reasoningEffort) && model.reasoning,
162
- thinkingBudgetTokens: reasoningEffort
163
- ? (options.thinkingBudgets?.[reasoningEffort] ?? ANTHROPIC_THINKING[reasoningEffort])
171
+ thinkingEnabled: Boolean(anthropicReasoningEffort) && model.reasoning,
172
+ thinkingBudgetTokens: anthropicReasoningEffort
173
+ ? (options.thinkingBudgets?.[anthropicReasoningEffort] ??
174
+ ANTHROPIC_THINKING[anthropicReasoningEffort])
164
175
  : undefined,
165
- reasoning: reasoningEffort,
176
+ reasoning: anthropicReasoningEffort,
166
177
  toolChoice: mapAnthropicToolChoice(options.toolChoice),
167
178
  },
168
179
  )
@@ -199,6 +210,8 @@ export function streamGitLabDuo(
199
210
  onSseEvent: options.onSseEvent,
200
211
  fetch: options.fetch,
201
212
  reasoning: reasoningEffort,
213
+ disableReasoning: options.disableReasoning,
214
+ forceReasoningOff: options.forceReasoningOff,
202
215
  toolChoice: options.toolChoice,
203
216
  } satisfies OpenAIResponsesOptions,
204
217
  )
@@ -233,6 +246,9 @@ export function streamGitLabDuo(
233
246
  onSseEvent: options.onSseEvent,
234
247
  fetch: options.fetch,
235
248
  reasoning: reasoningEffort,
249
+ // OpenAICompletionsOptions carries no forceReasoningOff; fold it
250
+ // like the azure-openai-responses mapping does.
251
+ disableReasoning: options.disableReasoning || options.forceReasoningOff,
236
252
  toolChoice: options.toolChoice,
237
253
  } satisfies OpenAICompletionsOptions,
238
254
  );
@@ -420,9 +420,7 @@ interface CloudCodeAssistRequest {
420
420
  temperature?: number;
421
421
  topP?: number;
422
422
  topK?: number;
423
- minP?: number;
424
423
  presencePenalty?: number;
425
- repetitionPenalty?: number;
426
424
  thinkingConfig?: ThinkingConfig;
427
425
  };
428
426
  tools?: { functionDeclarations: Record<string, unknown>[] }[] | undefined;
@@ -1275,15 +1273,9 @@ export function buildRequest(
1275
1273
  if (options.topK !== undefined) {
1276
1274
  generationConfig.topK = options.topK;
1277
1275
  }
1278
- if (options.minP !== undefined) {
1279
- generationConfig.minP = options.minP;
1280
- }
1281
1276
  if (options.presencePenalty !== undefined) {
1282
1277
  generationConfig.presencePenalty = options.presencePenalty;
1283
1278
  }
1284
- if (options.repetitionPenalty !== undefined) {
1285
- generationConfig.repetitionPenalty = options.repetitionPenalty;
1286
- }
1287
1279
 
1288
1280
  // Thinking config
1289
1281
  if (options.thinking?.enabled && model.reasoning) {
@@ -780,18 +780,6 @@ export async function consumeGoogleStream<T extends GoogleApiType>(args: {
780
780
  }
781
781
  }
782
782
 
783
- /**
784
- * Generation/sampling fields that map directly onto Gemini's `GenerateContentConfig`.
785
- * Excludes any provider-specific extensions (`topP`/`topK`/etc are all forwarded as-is).
786
- */
787
- interface GoogleGenerationConfig extends GenerateContentConfig {
788
- topP?: number;
789
- topK?: number;
790
- minP?: number;
791
- presencePenalty?: number;
792
- repetitionPenalty?: number;
793
- }
794
-
795
783
  /**
796
784
  * Build the `GenerateContentParameters` payload for the public Gemini API and Vertex AI.
797
785
  * Both surfaces accept the same `GenerateContentConfig` shape — every numeric/string knob,
@@ -809,14 +797,12 @@ export function buildGoogleGenerateContentParams<T extends "google-generative-ai
809
797
  const systemPrompts = normalizeSystemPrompts(context.systemPrompt);
810
798
  const contents = convertMessages(model, context);
811
799
 
812
- const generationConfig: GoogleGenerationConfig = {};
800
+ const generationConfig: GenerateContentConfig = {};
813
801
  if (options.temperature !== undefined) generationConfig.temperature = options.temperature;
814
802
  if (options.maxTokens !== undefined) generationConfig.maxOutputTokens = options.maxTokens;
815
803
  if (options.topP !== undefined) generationConfig.topP = options.topP;
816
804
  if (options.topK !== undefined) generationConfig.topK = options.topK;
817
- if (options.minP !== undefined) generationConfig.minP = options.minP;
818
805
  if (options.presencePenalty !== undefined) generationConfig.presencePenalty = options.presencePenalty;
819
- if (options.repetitionPenalty !== undefined) generationConfig.repetitionPenalty = options.repetitionPenalty;
820
806
 
821
807
  const config: GenerateContentConfig = {
822
808
  ...(Object.keys(generationConfig).length > 0 && generationConfig),
@@ -1111,9 +1097,6 @@ function paramsToWireBody(params: GenerateContentParameters): Record<string, unk
1111
1097
  if (config.responseJsonSchema !== undefined) gen.responseJsonSchema = config.responseJsonSchema;
1112
1098
  if (config.responseModalities !== undefined) gen.responseModalities = config.responseModalities;
1113
1099
  if (config.thinkingConfig !== undefined) gen.thinkingConfig = config.thinkingConfig;
1114
- const generationConfig = config as unknown as { minP?: number; repetitionPenalty?: number };
1115
- if (generationConfig.minP !== undefined) gen.minP = generationConfig.minP;
1116
- if (generationConfig.repetitionPenalty !== undefined) gen.repetitionPenalty = generationConfig.repetitionPenalty;
1117
1100
  if (Object.keys(gen).length > 0) body.generationConfig = gen;
1118
1101
  return body;
1119
1102
  }
@@ -210,7 +210,7 @@ export const openaiChatRequestSchema = type({
210
210
  "frequency_penalty?": "number",
211
211
  "logit_bias?": type({ "[string]": "number" }),
212
212
  "user?": "string",
213
- "reasoning_effort?": "'minimal' | 'low' | 'medium' | 'high' | 'xhigh' | 'max'",
213
+ "reasoning_effort?": "'none' | 'minimal' | 'low' | 'medium' | 'high' | 'xhigh' | 'max'",
214
214
  "parallel_tool_calls?": "boolean",
215
215
  "service_tier?": "'auto' | 'default' | 'flex' | 'scale' | 'priority'",
216
216
  "metadata?": type({ "[string]": "unknown" }),
@@ -200,7 +200,9 @@ export function parseRequest(body: unknown, headers?: Headers): ParsedRequest {
200
200
  if (data.user !== undefined) options.user = data.user;
201
201
  if (data.response_format !== undefined) options.responseFormat = data.response_format;
202
202
  if (data.parallel_tool_calls !== undefined) options.parallelToolCalls = data.parallel_tool_calls;
203
- if (data.reasoning_effort !== undefined && isReasoningEffort(data.reasoning_effort)) {
203
+ if (data.reasoning_effort === "none") {
204
+ options.forceReasoningOff = true;
205
+ } else if (data.reasoning_effort !== undefined && isReasoningEffort(data.reasoning_effort)) {
204
206
  options.reasoning = data.reasoning_effort;
205
207
  }
206
208
  if (data.service_tier !== undefined && isServiceTier(data.service_tier)) {