@oh-my-pi/pi-ai 18.4.3 → 18.4.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (58) hide show
  1. package/CHANGELOG.md +20 -17
  2. package/dist/types/auth-broker/protocol.d.ts +12 -0
  3. package/dist/types/dialect/rendering.d.ts +4 -0
  4. package/dist/types/images/shared.d.ts +5 -2
  5. package/dist/types/providers/anthropic-wire.d.ts +9 -1
  6. package/dist/types/providers/anthropic.d.ts +17 -0
  7. package/dist/types/providers/aws-sigv4.d.ts +5 -0
  8. package/dist/types/providers/bedrock-anthropic.d.ts +9 -0
  9. package/dist/types/providers/bedrock-request-metadata.d.ts +2 -0
  10. package/dist/types/providers/cursor/interaction-query.d.ts +10 -0
  11. package/dist/types/providers/openai-chat-wire.d.ts +2 -2
  12. package/dist/types/providers/openai-codex/request-transformer.d.ts +1 -1
  13. package/dist/types/providers/openai-responses-wire.d.ts +2 -2
  14. package/dist/types/providers/xai-base-url.d.ts +17 -0
  15. package/dist/types/types.d.ts +14 -3
  16. package/dist/types/usage/shared.d.ts +13 -1
  17. package/package.json +6 -6
  18. package/src/auth-broker/client.ts +1 -12
  19. package/src/auth-broker/protocol.ts +32 -0
  20. package/src/auth-broker/remote-store.ts +4 -40
  21. package/src/auth-broker/server.ts +1 -20
  22. package/src/auth-broker/snapshot-cache.ts +1 -9
  23. package/src/dialect/anthropic.ts +3 -25
  24. package/src/dialect/minimax.ts +3 -24
  25. package/src/dialect/rendering.ts +18 -0
  26. package/src/dialect/xml.ts +3 -19
  27. package/src/images/openai-images.ts +10 -4
  28. package/src/images/shared.ts +9 -4
  29. package/src/providers/amazon-bedrock.ts +3 -6
  30. package/src/providers/anthropic-compaction.ts +10 -1
  31. package/src/providers/anthropic-wire.ts +12 -1
  32. package/src/providers/anthropic.ts +58 -15
  33. package/src/providers/aws-sigv4.ts +1 -1
  34. package/src/providers/bedrock-anthropic.ts +30 -0
  35. package/src/providers/bedrock-request-metadata.ts +6 -0
  36. package/src/providers/connect-error-detail.ts +1 -5
  37. package/src/providers/cursor/interaction-query.ts +4 -2
  38. package/src/providers/cursor.ts +30 -26
  39. package/src/providers/google-shared.ts +8 -2
  40. package/src/providers/openai-chat-wire.ts +2 -2
  41. package/src/providers/openai-codex/request-transformer.ts +1 -1
  42. package/src/providers/openai-codex-responses.ts +19 -14
  43. package/src/providers/openai-responses-wire.ts +2 -2
  44. package/src/providers/openai-shared.ts +4 -0
  45. package/src/providers/xai-base-url.ts +32 -0
  46. package/src/types.ts +35 -4
  47. package/src/usage/claude.ts +4 -11
  48. package/src/usage/cline-pass.ts +2 -14
  49. package/src/usage/cursor.ts +9 -1
  50. package/src/usage/openai-codex.ts +3 -5
  51. package/src/usage/shared.ts +28 -1
  52. package/src/usage/synthetic.ts +4 -40
  53. package/src/usage/umans.ts +8 -36
  54. package/src/usage/zai.ts +10 -38
  55. package/src/utils/http-inspector.ts +4 -8
  56. package/src/utils/schema/json-schema-validator.ts +23 -26
  57. package/src/utils/schema/meta-validator.ts +4 -7
  58. package/src/utils/schema/wire.ts +17 -21
@@ -1,18 +1,13 @@
1
+ import { escapeXmlText } from "@oh-my-pi/pi-utils";
1
2
  import type { Message, ToolCall } from "../types";
2
3
  import {
3
4
  ANTHROPIC_THINKING_TAG_PREFIXES,
4
5
  AnthropicInbandScanner,
5
6
  type AnthropicInbandScannerConfig,
6
7
  } from "./anthropic";
7
- import { buildArgShapes, type ToolArgShape } from "./coercion";
8
+ import { buildArgShapes } from "./coercion";
8
9
  import dialectPrompt from "./minimax.md" with { type: "text" };
9
- import {
10
- escapeXmlAttr,
11
- escapeXmlText,
12
- renderDelimitedThinking,
13
- renderLegacyTextTranscript,
14
- stringifyJson,
15
- } from "./rendering";
10
+ import { renderDelimitedThinking, renderInvoke, renderInvokes, renderLegacyTextTranscript } from "./rendering";
16
11
  import type { DialectDefinition, DialectRenderOptions, DialectToolResult } from "./types";
17
12
 
18
13
  const MINIMAX_WRAPPER_TAGS: Readonly<Record<string, true>> = { tool_call: true };
@@ -65,22 +60,6 @@ function renderTranscript(messages: readonly Message[], options: DialectRenderOp
65
60
  });
66
61
  }
67
62
 
68
- function renderInvoke(call: ToolCall, shape: ToolArgShape | undefined): string {
69
- let body = `<invoke name="${escapeXmlAttr(call.name)}">`;
70
- for (const key in call.arguments) {
71
- const value = call.arguments[key];
72
- const isString = shape?.stringArgs.has(key) === true;
73
- const rendered = isString && typeof value === "string" ? value : stringifyJson(value);
74
- body += `<parameter name="${escapeXmlAttr(key)}">${rendered}</parameter>`;
75
- }
76
- return `${body}</invoke>`;
77
- }
78
-
79
- function renderInvokes(calls: readonly ToolCall[], tools: NonNullable<DialectRenderOptions["tools"]>): string {
80
- const shapes = buildArgShapes(tools);
81
- return calls.map(call => renderInvoke(call, shapes.get(call.name))).join("\n");
82
- }
83
-
84
63
  const definition: DialectDefinition = {
85
64
  dialect: "minimax",
86
65
  prompt: dialectPrompt,
@@ -1,5 +1,6 @@
1
1
  import { stringifyJson as stringifyJsonValue } from "@oh-my-pi/pi-utils";
2
2
  import type { AssistantMessage, Message, ToolCall, ToolResultMessage } from "../types";
3
+ import { buildArgShapes, type ToolArgShape } from "./coercion";
3
4
  import type { DialectRenderOptions, DialectToolResult } from "./types";
4
5
 
5
6
  export function renderToolResponseResults(results: readonly DialectToolResult[]): string {
@@ -81,6 +82,23 @@ export function escapeXmlText(value: string): string {
81
82
  return value.replaceAll("&", "&amp;").replaceAll("<", "&lt;").replaceAll(">", "&gt;");
82
83
  }
83
84
 
85
+ /** Render one Anthropic-style `<invoke>`; declared string args stay raw, everything else is JSON. */
86
+ export function renderInvoke(call: ToolCall, shape: ToolArgShape | undefined): string {
87
+ let body = `<invoke name="${escapeXmlAttr(call.name)}">`;
88
+ for (const key in call.arguments) {
89
+ const value = call.arguments[key];
90
+ const isString = shape?.stringArgs.has(key) === true;
91
+ const rendered = isString && typeof value === "string" ? value : stringifyJson(value);
92
+ body += `<parameter name="${escapeXmlAttr(key)}">${rendered}</parameter>`;
93
+ }
94
+ return `${body}</invoke>`;
95
+ }
96
+
97
+ export function renderInvokes(calls: readonly ToolCall[], tools: NonNullable<DialectRenderOptions["tools"]>): string {
98
+ const shapes = buildArgShapes(tools);
99
+ return calls.map(call => renderInvoke(call, shapes.get(call.name))).join("\n");
100
+ }
101
+
84
102
  export type AssistantTranscriptParts = {
85
103
  readonly text: string;
86
104
  readonly thinking: string;
@@ -1,13 +1,13 @@
1
1
  import type { Message, ToolCall } from "../types";
2
2
  import { AnthropicInbandScanner } from "./anthropic";
3
- import { buildArgShapes, type ToolArgShape } from "./coercion";
3
+ import { buildArgShapes } from "./coercion";
4
4
  import { DeepSeekInbandScanner } from "./deepseek";
5
5
  import {
6
- escapeXmlAttr,
7
6
  renderDelimitedThinking,
7
+ renderInvoke,
8
+ renderInvokes,
8
9
  renderLegacyTextTranscript,
9
10
  renderToolResponseResults,
10
- stringifyJson,
11
11
  } from "./rendering";
12
12
  import type {
13
13
  DialectDefinition,
@@ -60,22 +60,6 @@ function renderTranscript(messages: readonly Message[], options: DialectRenderOp
60
60
  });
61
61
  }
62
62
 
63
- function renderInvoke(call: ToolCall, shape: ToolArgShape | undefined): string {
64
- let body = `<invoke name="${escapeXmlAttr(call.name)}">`;
65
- for (const key in call.arguments) {
66
- const value = call.arguments[key];
67
- const isString = shape?.stringArgs.has(key) === true;
68
- const rendered = isString && typeof value === "string" ? value : stringifyJson(value);
69
- body += `<parameter name="${escapeXmlAttr(key)}">${rendered}</parameter>`;
70
- }
71
- return `${body}</invoke>`;
72
- }
73
-
74
- function renderInvokes(calls: readonly ToolCall[], tools: NonNullable<DialectRenderOptions["tools"]>): string {
75
- const shapes = buildArgShapes(tools);
76
- return calls.map(call => renderInvoke(call, shapes.get(call.name))).join("\n");
77
- }
78
-
79
63
  const definition: DialectDefinition = {
80
64
  dialect: "xml",
81
65
  prompt: dialectPrompt,
@@ -1,5 +1,6 @@
1
1
  import type { Model } from "@oh-my-pi/pi-catalog/types";
2
2
  import * as AIError from "../error";
3
+ import { resolveXaiBaseUrl } from "../providers/xai-base-url";
3
4
  import {
4
5
  decodeImageResponse,
5
6
  imageBaseUrl,
@@ -54,11 +55,16 @@ export async function generateOpenAIImage(
54
55
  : { ...generationBody, images: references }
55
56
  : { ...generationBody, input_references: references };
56
57
  const baseUrl = imageBaseUrl(model);
58
+ // xAI resolves the endpoint per bearer: XAI_BASE_URL never receives an xai-oauth OAuth access token.
59
+ const endpoint = (path: string) =>
60
+ isXAI
61
+ ? (bearer: string) => `${resolveXaiBaseUrl(model.provider, baseUrl, bearer) ?? baseUrl}${path}`
62
+ : `${baseUrl}${path}`;
57
63
  let response: unknown;
58
64
  if (references.length === 0) {
59
65
  response = await postJson({
60
66
  model,
61
- url: `${baseUrl}/images/generations`,
67
+ url: endpoint("/images/generations"),
62
68
  body: generationBody,
63
69
  apiKey: options.apiKey,
64
70
  fetch: fetchImpl,
@@ -78,7 +84,7 @@ export async function generateOpenAIImage(
78
84
  }
79
85
  response = await postMultipart({
80
86
  model,
81
- url: `${baseUrl}/images/edits`,
87
+ url: endpoint("/images/edits"),
82
88
  body: form,
83
89
  apiKey: options.apiKey,
84
90
  fetch: fetchImpl,
@@ -87,7 +93,7 @@ export async function generateOpenAIImage(
87
93
  } else {
88
94
  response = await postJson({
89
95
  model,
90
- url: `${baseUrl}/images/edits`,
96
+ url: endpoint("/images/edits"),
91
97
  body,
92
98
  apiKey: options.apiKey,
93
99
  fetch: fetchImpl,
@@ -98,7 +104,7 @@ export async function generateOpenAIImage(
98
104
  if (!(error instanceof AIError.ProviderHttpError) || error.status !== 404) throw error;
99
105
  response = await postJson({
100
106
  model,
101
- url: `${baseUrl}/images/generations`,
107
+ url: endpoint("/images/generations"),
102
108
  body,
103
109
  apiKey: options.apiKey,
104
110
  fetch: fetchImpl,
@@ -69,9 +69,12 @@ async function parseImageApiResponse(model: Model, response: Response): Promise<
69
69
  }
70
70
  }
71
71
 
72
+ /** Request URL, or a builder for routes that depend on the bearer (xAI's `XAI_BASE_URL` rule). */
73
+ type ImageRequestUrl = string | ((bearer: string) => string);
74
+
72
75
  export async function postJson(options: {
73
76
  model: Model;
74
- url: string;
77
+ url: ImageRequestUrl;
75
78
  body: unknown;
76
79
  apiKey: ApiKey;
77
80
  fetch: FetchImpl;
@@ -80,7 +83,8 @@ export async function postJson(options: {
80
83
  return withAuth(
81
84
  options.apiKey,
82
85
  async key => {
83
- const response = await options.fetch(options.url, {
86
+ const url = typeof options.url === "string" ? options.url : options.url(key);
87
+ const response = await options.fetch(url, {
84
88
  method: "POST",
85
89
  headers: {
86
90
  ...(await modelHeaders(options.model, options.signal)),
@@ -99,7 +103,7 @@ export async function postJson(options: {
99
103
 
100
104
  export async function postMultipart(options: {
101
105
  model: Model;
102
- url: string;
106
+ url: ImageRequestUrl;
103
107
  body: FormData;
104
108
  apiKey: ApiKey;
105
109
  fetch: FetchImpl;
@@ -108,7 +112,8 @@ export async function postMultipart(options: {
108
112
  return withAuth(
109
113
  options.apiKey,
110
114
  async key => {
111
- const response = await options.fetch(options.url, {
115
+ const url = typeof options.url === "string" ? options.url : options.url(key);
116
+ const response = await options.fetch(url, {
112
117
  method: "POST",
113
118
  headers: {
114
119
  ...(await modelHeaders(options.model, options.signal)),
@@ -56,6 +56,7 @@ import { invalidateAwsCredentialCache, resolveAwsCredentials } from "./aws-crede
56
56
  import { decodeEventStream } from "./aws-eventstream";
57
57
  import { signRequest } from "./aws-sigv4";
58
58
  import { parseAnthropicInputTransformations, THINKING_BINDING_CONTROLS_BETA } from "./anthropic-wire";
59
+ import { isBedrockRequestMetadataValue } from "./bedrock-request-metadata";
59
60
  import { transformMessages } from "./transform-messages";
60
61
 
61
62
  /**
@@ -381,9 +382,7 @@ interface MetadataEvent {
381
382
  };
382
383
  }
383
384
 
384
- const REQUEST_METADATA_PATTERN = /^[a-zA-Z0-9\s:_@$#=/+,\-.]*$/;
385
385
  const REQUEST_METADATA_MAX_ENTRIES = 16;
386
- const REQUEST_METADATA_MAX_LENGTH = 256;
387
386
 
388
387
  /**
389
388
  * Bedrock rejects the whole invocation on a malformed `requestMetadata` entry.
@@ -400,10 +399,8 @@ function sanitizeRequestMetadata(raw: unknown): Record<string, string> | undefin
400
399
  if (
401
400
  typeof value !== "string" ||
402
401
  key.length < 1 ||
403
- key.length > REQUEST_METADATA_MAX_LENGTH ||
404
- !REQUEST_METADATA_PATTERN.test(key) ||
405
- value.length > REQUEST_METADATA_MAX_LENGTH ||
406
- !REQUEST_METADATA_PATTERN.test(value) ||
402
+ !isBedrockRequestMetadataValue(key) ||
403
+ !isBedrockRequestMetadataValue(value) ||
407
404
  kept >= REQUEST_METADATA_MAX_ENTRIES
408
405
  ) {
409
406
  dropped.push(key);
@@ -1,4 +1,4 @@
1
- import { isOfficialAnthropicApiUrl } from "@oh-my-pi/pi-catalog/compat/anthropic";
1
+ import { isBedrockAnthropicRoute, isOfficialAnthropicApiUrl } from "@oh-my-pi/pi-catalog/compat/anthropic";
2
2
  import type { Model } from "../types";
3
3
  import type { AnthropicMessagesClientLike } from "./anthropic-client";
4
4
  import { normalizeAnthropicBaseUrl, resolveDirectAnthropicBaseUrl } from "./anthropic-state";
@@ -46,6 +46,15 @@ export function supportsAnthropicCompaction(model: Model<"anthropic-messages">,
46
46
  (model.provider === "anthropic"
47
47
  ? resolveDirectAnthropicBaseUrl(model)
48
48
  : normalizeAnthropicBaseUrl(model.baseUrl));
49
+ // Bedrock's Anthropic Messages API implements on-demand compaction. The flag is detected
50
+ // from a Bedrock `/anthropic` baseUrl, or set in models.yml for a proxy or a reroute; it
51
+ // applies to the model's own endpoint or a Bedrock `/anthropic` route it reaches.
52
+ if (
53
+ model.compat.bedrockMessagesApi === true &&
54
+ (isBedrockAnthropicRoute(route) || route === normalizeAnthropicBaseUrl(model.baseUrl))
55
+ ) {
56
+ return true;
57
+ }
49
58
  return (
50
59
  isSupportedCompactionEndpoint(route) &&
51
60
  (model.compat.firstPartyProvider === true ||
@@ -272,7 +272,18 @@ export type ThinkingConfigAdaptive = {
272
272
  block_binding?: ThinkingBlockBinding;
273
273
  };
274
274
 
275
- export type ThinkingConfigParam = ThinkingConfigEnabled | ThinkingConfigDisabled | ThinkingConfigAdaptive;
275
+ /**
276
+ * Sonnet 5.5's replacement for `disabled`: no up-front thinking, progress
277
+ * updates between tool calls only. Takes no other field, and effort above
278
+ * `high` is rejected alongside it.
279
+ */
280
+ export type ThinkingConfigBetweenTools = { type: "between_tools" };
281
+
282
+ export type ThinkingConfigParam =
283
+ | ThinkingConfigEnabled
284
+ | ThinkingConfigDisabled
285
+ | ThinkingConfigAdaptive
286
+ | ThinkingConfigBetweenTools;
276
287
 
277
288
  export type OutputConfig = {
278
289
  /** Adaptive-thinking effort level (effort beta). */
@@ -155,6 +155,7 @@ import {
155
155
  resolveAnthropicMetadataUserId,
156
156
  stripClaudeToolPrefix,
157
157
  } from "./anthropic-identity";
158
+ import { fitBedrockAnthropicPayload } from "./bedrock-anthropic";
158
159
  import {
159
160
  anthropicProviderSessionStateKey,
160
161
  clearAnthropicFastModeFallback,
@@ -2264,6 +2265,8 @@ const streamAnthropicOnce = (
2264
2265
  nextParams = replacementPayload as typeof nextParams;
2265
2266
  }
2266
2267
  if (nextParams.compaction) stripCompactionIncompatibleParams(nextParams);
2268
+ // After `onPayload`, so a hook cannot restore a field Bedrock rejects.
2269
+ if (model.compat.bedrockMessagesApi) fitBedrockAnthropicPayload(nextParams);
2267
2270
  nextParams = toWellFormedDeep(nextParams) as typeof nextParams;
2268
2271
  rawRequestDump = {
2269
2272
  provider: model.provider,
@@ -3827,22 +3830,17 @@ function ensureMaxTokensForThinking(params: MessageCreateParamsStreaming, maxAll
3827
3830
  const budgetTokens = thinking.budget_tokens ?? 0;
3828
3831
  if (budgetTokens <= 0) return;
3829
3832
 
3830
- const currentMaxTokens = Math.min(params.max_tokens ?? maxAllowedTokens, maxAllowedTokens);
3831
- const raisedMaxTokens = Math.min(
3832
- Math.max(currentMaxTokens, budgetTokens + OUTPUT_FALLBACK_BUFFER),
3833
- maxAllowedTokens,
3834
- );
3835
- params.max_tokens = raisedMaxTokens;
3833
+ const output = budgetThinkingOutput(params.max_tokens, budgetTokens, maxAllowedTokens);
3834
+ params.max_tokens = output.maxTokens;
3836
3835
 
3837
- if (budgetTokens + OUTPUT_FALLBACK_BUFFER <= raisedMaxTokens) return;
3836
+ if (output.budgetTokens === budgetTokens) return;
3838
3837
 
3839
- const clampedBudget = raisedMaxTokens - OUTPUT_FALLBACK_BUFFER;
3840
- if (clampedBudget <= 0) {
3838
+ if (output.budgetTokens <= 0) {
3841
3839
  throw new AIError.ConfigurationError(
3842
- `Anthropic thinking budget requires max_tokens greater than ${OUTPUT_FALLBACK_BUFFER}; got ${raisedMaxTokens}`,
3840
+ `Anthropic thinking budget requires max_tokens greater than ${OUTPUT_FALLBACK_BUFFER}; got ${output.maxTokens}`,
3843
3841
  );
3844
3842
  }
3845
- thinking.budget_tokens = clampedBudget;
3843
+ thinking.budget_tokens = output.budgetTokens;
3846
3844
  }
3847
3845
 
3848
3846
  function applyCacheControlToLastBlock(blocks: ContentBlockParam[], cacheControl: AnthropicCacheControl): boolean {
@@ -4128,6 +4126,41 @@ function usesAdaptiveThinkingTagOnly(model: Model<"anthropic-messages">): boolea
4128
4126
  return thinking.efforts.length > 0;
4129
4127
  }
4130
4128
 
4129
+ /**
4130
+ * True when enabled thinking on `model` is budget thinking
4131
+ * (`thinking.type: "enabled"` with `budget_tokens`) rather than adaptive.
4132
+ */
4133
+ export function usesBudgetThinking(model: Model<"anthropic-messages">): boolean {
4134
+ return model.thinking?.mode !== "anthropic-adaptive" || model.compat.disableAdaptiveThinking === true;
4135
+ }
4136
+
4137
+ /** The most output tokens a request to `model` may ask for (`max_tokens` ceiling). */
4138
+ export function anthropicOutputLimit(model: Model<"anthropic-messages">): number {
4139
+ return model.maxTokens ?? UNKNOWN_MODEL_MAX_OUTPUT_TOKENS;
4140
+ }
4141
+
4142
+ /**
4143
+ * The `max_tokens` and thinking budget of budget thinking: `max_tokens`
4144
+ * rises to leave {@link OUTPUT_FALLBACK_BUFFER} visible output tokens after
4145
+ * the budget, within `maxAllowedTokens`, and the budget shrinks when that
4146
+ * ceiling leaves less (a non-positive budget means the ceiling is too low).
4147
+ */
4148
+ export function budgetThinkingOutput(
4149
+ maxTokens: number | undefined,
4150
+ budgetTokens: number,
4151
+ maxAllowedTokens: number,
4152
+ ): { maxTokens: number; budgetTokens: number } {
4153
+ const currentMaxTokens = Math.min(maxTokens ?? maxAllowedTokens, maxAllowedTokens);
4154
+ const raisedMaxTokens = Math.min(
4155
+ Math.max(currentMaxTokens, budgetTokens + OUTPUT_FALLBACK_BUFFER),
4156
+ maxAllowedTokens,
4157
+ );
4158
+ return {
4159
+ maxTokens: raisedMaxTokens,
4160
+ budgetTokens: Math.min(budgetTokens, raisedMaxTokens - OUTPUT_FALLBACK_BUFFER),
4161
+ };
4162
+ }
4163
+
4131
4164
  /**
4132
4165
  * True for adaptive-only Claude models (Opus 4.6+, Sonnet 4.6+, Fable/Mythos 5)
4133
4166
  * that reject `thinking.type: "disabled"`. Turning thinking off on these models
@@ -4550,8 +4583,7 @@ function buildParams(
4550
4583
  const thinkingOptions = options ?? {};
4551
4584
  const mode = model.thinking?.mode;
4552
4585
  const effort = resolveAnthropicAdaptiveEffort(model, thinkingOptions);
4553
- const compat = model.compat;
4554
- if (mode === "anthropic-adaptive" && !compat.disableAdaptiveThinking) {
4586
+ if (!usesBudgetThinking(model)) {
4555
4587
  const adaptive: { type: "adaptive"; display?: AnthropicThinkingDisplay } = { type: "adaptive" };
4556
4588
  // Starting with Claude Opus 4.7 and Claude Fable/Mythos 5, adaptive thinking
4557
4589
  // content is omitted from the response by default. Opt into summarized
@@ -4573,7 +4605,12 @@ function buildParams(
4573
4605
  if (mode === "anthropic-budget-effort" && effort && effort !== "adaptive") outputConfigEffort = effort;
4574
4606
  }
4575
4607
  } else if (options?.thinkingEnabled === false) {
4576
- if (isAdaptiveOnlyThinking(model)) {
4608
+ if (model.compat.supportsBetweenToolsThinking) {
4609
+ // Sonnet 5.5 rejects `disabled` with a 400; `between_tools` is its lowest
4610
+ // thinking setting. It takes no other field and leaves effort untouched:
4611
+ // pinning `low` here would cap the whole turn's quality, not only thinking.
4612
+ thinking = { type: "between_tools" };
4613
+ } else if (isAdaptiveOnlyThinking(model)) {
4577
4614
  // Adaptive-only Claude models (Opus 4.6+, Sonnet 4.6+, Fable/Mythos 5) reject
4578
4615
  // `thinking.type: "disabled"` — adaptive thinking cannot be switched off.
4579
4616
  // Omit the thinking field (the API defaults to adaptive) and pin the
@@ -4643,6 +4680,12 @@ function buildParams(
4643
4680
  model.compat.supportsPerMessageEffort === true,
4644
4681
  compactionReplay,
4645
4682
  );
4683
+ // `between_tools` returns a 400 at `xhigh`/`max` effort, and the effort in
4684
+ // force from earlier turns outlives a thinking toggle. Fall back to the
4685
+ // default adaptive request, which accepts every effort level.
4686
+ if (thinking?.type === "between_tools" && (effortPlan.topLevel === "xhigh" || effortPlan.topLevel === "max")) {
4687
+ thinking = undefined;
4688
+ }
4646
4689
  const wireMessages = convertAnthropicMessages(
4647
4690
  insertAnthropicControlMarkers(context.messages, [...toolPlan.inserts, ...effortPlan.inserts]),
4648
4691
  effectiveModel,
@@ -4683,7 +4726,7 @@ function buildParams(
4683
4726
 
4684
4727
  // OAuth and API-key requests alike get the full model ceiling; Claude Code
4685
4728
  // itself requests 128k on Opus 5.5.
4686
- const maxOutputTokens = model.maxTokens ?? UNKNOWN_MODEL_MAX_OUTPUT_TOKENS;
4729
+ const maxOutputTokens = anthropicOutputLimit(model);
4687
4730
 
4688
4731
  // A caller-owned client targets its own endpoint: route body betas by the
4689
4732
  // client's URL when it exposes one, not the model's routing. Otherwise the
@@ -61,7 +61,7 @@ const UNSIGNABLE: Record<string, true> = {
61
61
  * `ArrayBuffer`, which is what `crypto.subtle.{digest,sign,importKey}` requires
62
62
  * under the strict TS DOM typings. No-op when already strict.
63
63
  */
64
- function asStrict(bytes: Uint8Array): Uint8Array<ArrayBuffer> {
64
+ export function asStrict(bytes: Uint8Array): Uint8Array<ArrayBuffer> {
65
65
  if (bytes.buffer instanceof ArrayBuffer && bytes.byteOffset === 0 && bytes.byteLength === bytes.buffer.byteLength) {
66
66
  return bytes as Uint8Array<ArrayBuffer>;
67
67
  }
@@ -0,0 +1,30 @@
1
+ import { isRecord } from "@oh-my-pi/pi-utils";
2
+ import { extractClaudeMetadataSessionId } from "./anthropic-identity";
3
+ import { isBedrockRequestMetadataValue } from "./bedrock-request-metadata";
4
+
5
+ /**
6
+ * Fit an Anthropic request body to Bedrock's Anthropic Messages API
7
+ * (`compat.bedrockMessagesApi`): both `/anthropic` routes reject the tool
8
+ * `strict` field, and bedrock-runtime rejects a `metadata.user_id` outside
9
+ * Bedrock's request-metadata pattern. A user id that fits is kept, otherwise
10
+ * its embedded session id, otherwise the metadata is dropped. Mutates and
11
+ * returns `payload`.
12
+ */
13
+ export function fitBedrockAnthropicPayload<T>(payload: T): T {
14
+ if (!isRecord(payload)) return payload;
15
+ const body: Record<string, unknown> = payload;
16
+ if (Array.isArray(body.tools)) {
17
+ for (const tool of body.tools) {
18
+ if (isRecord(tool)) delete tool.strict;
19
+ }
20
+ }
21
+ if (body.metadata === undefined) return payload;
22
+ const userId = isRecord(body.metadata) ? body.metadata.user_id : undefined;
23
+ const fitted =
24
+ typeof userId === "string" && isBedrockRequestMetadataValue(userId)
25
+ ? userId
26
+ : extractClaudeMetadataSessionId(userId);
27
+ if (fitted && isBedrockRequestMetadataValue(fitted)) body.metadata = { user_id: fitted };
28
+ else delete body.metadata;
29
+ return payload;
30
+ }
@@ -0,0 +1,6 @@
1
+ const BEDROCK_REQUEST_METADATA_PATTERN = /^[a-zA-Z0-9\s:_@$#=/+,\-.]*$/;
2
+
3
+ /** Check Bedrock's request-metadata character and length limits. Keys must also be nonempty. */
4
+ export function isBedrockRequestMetadataValue(value: string): boolean {
5
+ return value.length <= 256 && BEDROCK_REQUEST_METADATA_PATTERN.test(value);
6
+ }
@@ -1,4 +1,4 @@
1
- import { truncate } from "@oh-my-pi/pi-utils";
1
+ import { isRecord, truncate } from "@oh-my-pi/pi-utils";
2
2
 
3
3
  /**
4
4
  * Connect-protocol end-stream error formatting.
@@ -21,10 +21,6 @@ const GENERIC_CONNECT_ERROR_MESSAGES = new Set(["", "error", "unknown", "unknown
21
21
  /** Upper bound for appended trailer context so errors stay log-line sized. */
22
22
  const MAX_EXTRA_DETAIL_CHARS = 400;
23
23
 
24
- function isRecord(value: unknown): value is Record<string, unknown> {
25
- return typeof value === "object" && value !== null && !Array.isArray(value);
26
- }
27
-
28
24
  function safeJson(value: unknown): string | undefined {
29
25
  try {
30
26
  const text = typeof value === "string" ? value : JSON.stringify(value);
@@ -32,7 +32,8 @@ type ProtoUnknownBag = { $unknown?: ProtoUnknownField[] };
32
32
  type InteractionQueryCase = NonNullable<InteractionQuery["query"]["case"]>;
33
33
  type InteractionResult = Exclude<InteractionResponse["result"], { case: undefined; value?: undefined }>;
34
34
 
35
- function frameConnectMessage(data: Uint8Array, flags = 0): Buffer {
35
+ /** Wrap one Connect-protocol message: 1 flag byte + 4-byte big-endian length + payload. */
36
+ export function frameConnectMessage(data: Uint8Array, flags = 0): Buffer {
36
37
  const frame = Buffer.alloc(5 + data.length);
37
38
  frame[0] = flags;
38
39
  frame.writeUInt32BE(data.length, 1);
@@ -46,7 +47,8 @@ function isProtoUnknownField(value: unknown): value is ProtoUnknownField {
46
47
  return typeof value.no === "number" && typeof value.wireType === "number" && value.data instanceof Uint8Array;
47
48
  }
48
49
 
49
- function protoUnknownFields(message: object): ProtoUnknownField[] {
50
+ /** Well-formed protobuf-es `$unknown` entries on `message`; anything else on the bag is ignored. */
51
+ export function protoUnknownFields(message: object): ProtoUnknownField[] {
50
52
  if (!("$unknown" in message) || !Array.isArray(message.$unknown)) return [];
51
53
  return message.$unknown.filter(isProtoUnknownField);
52
54
  }
@@ -1,5 +1,6 @@
1
1
  import * as fs from "node:fs/promises";
2
2
  import http2 from "node:http2";
3
+ import { cursorModelParameters } from "@oh-my-pi/pi-catalog/compat/behavior";
3
4
  import { isCursorMaxModeWireId } from "@oh-my-pi/pi-catalog/compat/collapse";
4
5
  import { classifyModel, collapseVariantId } from "@oh-my-pi/pi-catalog/compat/taxonomy";
5
6
  import type {
@@ -246,7 +247,7 @@ import {
246
247
  piTimeout,
247
248
  shellTimeoutSeconds,
248
249
  } from "./cursor/exec-modern";
249
- import { handleInteractionQuery } from "./cursor/interaction-query";
250
+ import { frameConnectMessage, handleInteractionQuery, protoUnknownFields } from "./cursor/interaction-query";
250
251
 
251
252
  export const CURSOR_API_URL = "https://api2.cursor.sh";
252
253
  export const CURSOR_CLIENT_VERSION = "cli-2026.07.23-e383d2b";
@@ -407,14 +408,14 @@ function log(type: string, subtype?: string, data?: unknown): void {
407
408
  void appendCursorDebugLog(entry);
408
409
  }
409
410
 
410
- function frameConnectMessage(data: Uint8Array, flags = 0): Buffer {
411
- const frame = Buffer.alloc(5 + data.length);
412
- frame[0] = flags;
413
- frame.writeUInt32BE(data.length, 1);
414
- frame.set(data, 5);
415
- return frame;
411
+ /**
412
+ * Write one client message. Once the server's end frame has half-closed our
413
+ * side, late writes (heartbeats, exec replies from a handler still running)
414
+ * are dropped: writing after `end()` would error the stream.
415
+ */
416
+ function writeClientMessage(h2Request: http2.ClientHttp2Stream, data: Uint8Array): void {
417
+ if (!h2Request.writableEnded) h2Request.write(frameConnectMessage(data));
416
418
  }
417
-
418
419
  class ConnectEndStreamError extends AIError.ProviderResponseError {
419
420
  readonly diagnosticMessage: string;
420
421
 
@@ -845,6 +846,11 @@ function streamCursorWithWireMode(
845
846
  if (endError) {
846
847
  endStreamError = endError;
847
848
  h2Request?.close();
849
+ } else {
850
+ // The end frame is the server's last message. Half-close our
851
+ // side so the stream can finish: a CONNECT proxy holds the
852
+ // HTTP/2 stream open until the client ends its request.
853
+ h2Request?.end();
848
854
  }
849
855
  continue;
850
856
  }
@@ -903,7 +909,7 @@ function streamCursorWithWireMode(
903
909
  message: { case: "clientHeartbeat", value: create(ClientHeartbeatSchema, {}) },
904
910
  });
905
911
  const heartbeatBytes = toBinary(AgentClientMessageSchema, heartbeatMessage);
906
- h2Request.write(frameConnectMessage(heartbeatBytes));
912
+ writeClientMessage(h2Request, heartbeatBytes);
907
913
  };
908
914
 
909
915
  const closeDebugLog = async (): Promise<void> => {
@@ -1207,8 +1213,6 @@ export async function handleServerMessage(
1207
1213
  }
1208
1214
  }
1209
1215
 
1210
- type ProtoUnknownField = { no: number; wireType: number; data: Uint8Array };
1211
-
1212
1216
  type HostedFetchCall = {
1213
1217
  args?: { url?: string; toolCallId?: string };
1214
1218
  result?: { result?: { case?: string; value?: { content?: string; error?: string; url?: string } } };
@@ -1250,11 +1254,6 @@ function describeHostedFetchResult(call: HostedFetchCall | undefined): { text: s
1250
1254
  return { text: "Fetch completed", isError: false };
1251
1255
  }
1252
1256
 
1253
- function protoUnknownFields(message: object): ProtoUnknownField[] {
1254
- const raw = (message as { $unknown?: ProtoUnknownField[] }).$unknown;
1255
- return Array.isArray(raw) ? raw : [];
1256
- }
1257
-
1258
1257
  function handleKvServerMessage(
1259
1258
  kvMsg: KvServerMessage,
1260
1259
  blobStore: Map<string, Uint8Array>,
@@ -1281,7 +1280,7 @@ function handleKvServerMessage(
1281
1280
  });
1282
1281
 
1283
1282
  const responseBytes = toBinary(AgentClientMessageSchema, kvClientMessage);
1284
- h2Request.write(frameConnectMessage(responseBytes));
1283
+ writeClientMessage(h2Request, responseBytes);
1285
1284
 
1286
1285
  log("kvClient", "getBlobResult", { blobId: blobIdKey.slice(0, 40) });
1287
1286
  } else if (kvCase === "setBlobArgs") {
@@ -1302,7 +1301,7 @@ function handleKvServerMessage(
1302
1301
  });
1303
1302
 
1304
1303
  const responseBytes = toBinary(AgentClientMessageSchema, kvClientMessage);
1305
- h2Request.write(frameConnectMessage(responseBytes));
1304
+ writeClientMessage(h2Request, responseBytes);
1306
1305
 
1307
1306
  log("kvClient", "setBlobResult", { blobId: blobIdKey.slice(0, 40) });
1308
1307
  }
@@ -2604,7 +2603,7 @@ function sendExecClientMessage<TCase extends NonNullable<ExecClientMessage["mess
2604
2603
  });
2605
2604
 
2606
2605
  const responseBytes = toBinary(AgentClientMessageSchema, clientMessage);
2607
- h2Request.write(frameConnectMessage(responseBytes));
2606
+ writeClientMessage(h2Request, responseBytes);
2608
2607
 
2609
2608
  log("execClientMessage", messageCase, value);
2610
2609
  }
@@ -2640,7 +2639,7 @@ function sendExecClientThrow(
2640
2639
  const clientMessage = create(AgentClientMessageSchema, {
2641
2640
  message: { case: "execClientControlMessage", value: controlMessage },
2642
2641
  });
2643
- h2Request.write(frameConnectMessage(toBinary(AgentClientMessageSchema, clientMessage)));
2642
+ writeClientMessage(h2Request, toBinary(AgentClientMessageSchema, clientMessage));
2644
2643
  log("execClientControl", "throw", { id: execMsg.id, execId: execMsg.execId, error, errorCode });
2645
2644
  sendExecClientStreamClose(h2Request, execMsg);
2646
2645
  }
@@ -2658,7 +2657,7 @@ function sendExecClientStreamClose(h2Request: http2.ClientHttp2Stream, execMsg:
2658
2657
  message: { case: "execClientControlMessage", value: closeMessage },
2659
2658
  });
2660
2659
  const responseBytes = toBinary(AgentClientMessageSchema, clientMessage);
2661
- h2Request.write(frameConnectMessage(responseBytes));
2660
+ writeClientMessage(h2Request, responseBytes);
2662
2661
  log("execClientControl", "streamClose", { id: execMsg.id, execId: execMsg.execId });
2663
2662
  }
2664
2663
 
@@ -5487,13 +5486,18 @@ function resolveCursorWireModel(
5487
5486
  };
5488
5487
  }
5489
5488
  }
5490
- // A bare `composer-2.5` id resolves to the Fast variant server-side
5491
- // (can1357/oh-my-pi#9012). Pin the Standard tier explicitly; `-fast`
5492
- // selections keep the Fast lane by omitting the parameter.
5493
- if (wireModelId === "composer-2.5") {
5489
+ // Fixed per-model parameters come from catalog KDL (`cursor-model-parameter`
5490
+ // in `runtime/behavior.kdl`). A bare `composer-2.5` id resolves to the Fast
5491
+ // variant server-side (can1357/oh-my-pi#9012), so the catalog pins the
5492
+ // Standard tier with `fast=false`; `-fast` selections keep the Fast lane by
5493
+ // declaring no parameter.
5494
+ const fixedParameters = cursorModelParameters(wireModelId);
5495
+ if (fixedParameters.length > 0) {
5494
5496
  return {
5495
5497
  modelId: wireModelId,
5496
- parameters: [create(RequestedModel_ModelParameterbytesSchema, { id: "fast", value: "false" })],
5498
+ parameters: fixedParameters.map(({ id, value }) =>
5499
+ create(RequestedModel_ModelParameterbytesSchema, { id, value }),
5500
+ ),
5497
5501
  maxMode,
5498
5502
  };
5499
5503
  }