@oh-my-pi/pi-ai 18.4.3 → 18.4.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +20 -17
- package/dist/types/auth-broker/protocol.d.ts +12 -0
- package/dist/types/dialect/rendering.d.ts +4 -0
- package/dist/types/images/shared.d.ts +5 -2
- package/dist/types/providers/anthropic-wire.d.ts +9 -1
- package/dist/types/providers/anthropic.d.ts +17 -0
- package/dist/types/providers/aws-sigv4.d.ts +5 -0
- package/dist/types/providers/bedrock-anthropic.d.ts +9 -0
- package/dist/types/providers/bedrock-request-metadata.d.ts +2 -0
- package/dist/types/providers/cursor/interaction-query.d.ts +10 -0
- package/dist/types/providers/openai-chat-wire.d.ts +2 -2
- package/dist/types/providers/openai-codex/request-transformer.d.ts +1 -1
- package/dist/types/providers/openai-responses-wire.d.ts +2 -2
- package/dist/types/providers/xai-base-url.d.ts +17 -0
- package/dist/types/types.d.ts +14 -3
- package/dist/types/usage/shared.d.ts +13 -1
- package/package.json +6 -6
- package/src/auth-broker/client.ts +1 -12
- package/src/auth-broker/protocol.ts +32 -0
- package/src/auth-broker/remote-store.ts +4 -40
- package/src/auth-broker/server.ts +1 -20
- package/src/auth-broker/snapshot-cache.ts +1 -9
- package/src/dialect/anthropic.ts +3 -25
- package/src/dialect/minimax.ts +3 -24
- package/src/dialect/rendering.ts +18 -0
- package/src/dialect/xml.ts +3 -19
- package/src/images/openai-images.ts +10 -4
- package/src/images/shared.ts +9 -4
- package/src/providers/amazon-bedrock.ts +3 -6
- package/src/providers/anthropic-compaction.ts +10 -1
- package/src/providers/anthropic-wire.ts +12 -1
- package/src/providers/anthropic.ts +58 -15
- package/src/providers/aws-sigv4.ts +1 -1
- package/src/providers/bedrock-anthropic.ts +30 -0
- package/src/providers/bedrock-request-metadata.ts +6 -0
- package/src/providers/connect-error-detail.ts +1 -5
- package/src/providers/cursor/interaction-query.ts +4 -2
- package/src/providers/cursor.ts +30 -26
- package/src/providers/google-shared.ts +8 -2
- package/src/providers/openai-chat-wire.ts +2 -2
- package/src/providers/openai-codex/request-transformer.ts +1 -1
- package/src/providers/openai-codex-responses.ts +19 -14
- package/src/providers/openai-responses-wire.ts +2 -2
- package/src/providers/openai-shared.ts +4 -0
- package/src/providers/xai-base-url.ts +32 -0
- package/src/types.ts +35 -4
- package/src/usage/claude.ts +4 -11
- package/src/usage/cline-pass.ts +2 -14
- package/src/usage/cursor.ts +9 -1
- package/src/usage/openai-codex.ts +3 -5
- package/src/usage/shared.ts +28 -1
- package/src/usage/synthetic.ts +4 -40
- package/src/usage/umans.ts +8 -36
- package/src/usage/zai.ts +10 -38
- package/src/utils/http-inspector.ts +4 -8
- package/src/utils/schema/json-schema-validator.ts +23 -26
- package/src/utils/schema/meta-validator.ts +4 -7
- package/src/utils/schema/wire.ts +17 -21
package/src/dialect/minimax.ts
CHANGED
|
@@ -1,18 +1,13 @@
|
|
|
1
|
+
import { escapeXmlText } from "@oh-my-pi/pi-utils";
|
|
1
2
|
import type { Message, ToolCall } from "../types";
|
|
2
3
|
import {
|
|
3
4
|
ANTHROPIC_THINKING_TAG_PREFIXES,
|
|
4
5
|
AnthropicInbandScanner,
|
|
5
6
|
type AnthropicInbandScannerConfig,
|
|
6
7
|
} from "./anthropic";
|
|
7
|
-
import { buildArgShapes
|
|
8
|
+
import { buildArgShapes } from "./coercion";
|
|
8
9
|
import dialectPrompt from "./minimax.md" with { type: "text" };
|
|
9
|
-
import {
|
|
10
|
-
escapeXmlAttr,
|
|
11
|
-
escapeXmlText,
|
|
12
|
-
renderDelimitedThinking,
|
|
13
|
-
renderLegacyTextTranscript,
|
|
14
|
-
stringifyJson,
|
|
15
|
-
} from "./rendering";
|
|
10
|
+
import { renderDelimitedThinking, renderInvoke, renderInvokes, renderLegacyTextTranscript } from "./rendering";
|
|
16
11
|
import type { DialectDefinition, DialectRenderOptions, DialectToolResult } from "./types";
|
|
17
12
|
|
|
18
13
|
const MINIMAX_WRAPPER_TAGS: Readonly<Record<string, true>> = { tool_call: true };
|
|
@@ -65,22 +60,6 @@ function renderTranscript(messages: readonly Message[], options: DialectRenderOp
|
|
|
65
60
|
});
|
|
66
61
|
}
|
|
67
62
|
|
|
68
|
-
function renderInvoke(call: ToolCall, shape: ToolArgShape | undefined): string {
|
|
69
|
-
let body = `<invoke name="${escapeXmlAttr(call.name)}">`;
|
|
70
|
-
for (const key in call.arguments) {
|
|
71
|
-
const value = call.arguments[key];
|
|
72
|
-
const isString = shape?.stringArgs.has(key) === true;
|
|
73
|
-
const rendered = isString && typeof value === "string" ? value : stringifyJson(value);
|
|
74
|
-
body += `<parameter name="${escapeXmlAttr(key)}">${rendered}</parameter>`;
|
|
75
|
-
}
|
|
76
|
-
return `${body}</invoke>`;
|
|
77
|
-
}
|
|
78
|
-
|
|
79
|
-
function renderInvokes(calls: readonly ToolCall[], tools: NonNullable<DialectRenderOptions["tools"]>): string {
|
|
80
|
-
const shapes = buildArgShapes(tools);
|
|
81
|
-
return calls.map(call => renderInvoke(call, shapes.get(call.name))).join("\n");
|
|
82
|
-
}
|
|
83
|
-
|
|
84
63
|
const definition: DialectDefinition = {
|
|
85
64
|
dialect: "minimax",
|
|
86
65
|
prompt: dialectPrompt,
|
package/src/dialect/rendering.ts
CHANGED
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
import { stringifyJson as stringifyJsonValue } from "@oh-my-pi/pi-utils";
|
|
2
2
|
import type { AssistantMessage, Message, ToolCall, ToolResultMessage } from "../types";
|
|
3
|
+
import { buildArgShapes, type ToolArgShape } from "./coercion";
|
|
3
4
|
import type { DialectRenderOptions, DialectToolResult } from "./types";
|
|
4
5
|
|
|
5
6
|
export function renderToolResponseResults(results: readonly DialectToolResult[]): string {
|
|
@@ -81,6 +82,23 @@ export function escapeXmlText(value: string): string {
|
|
|
81
82
|
return value.replaceAll("&", "&").replaceAll("<", "<").replaceAll(">", ">");
|
|
82
83
|
}
|
|
83
84
|
|
|
85
|
+
/** Render one Anthropic-style `<invoke>`; declared string args stay raw, everything else is JSON. */
|
|
86
|
+
export function renderInvoke(call: ToolCall, shape: ToolArgShape | undefined): string {
|
|
87
|
+
let body = `<invoke name="${escapeXmlAttr(call.name)}">`;
|
|
88
|
+
for (const key in call.arguments) {
|
|
89
|
+
const value = call.arguments[key];
|
|
90
|
+
const isString = shape?.stringArgs.has(key) === true;
|
|
91
|
+
const rendered = isString && typeof value === "string" ? value : stringifyJson(value);
|
|
92
|
+
body += `<parameter name="${escapeXmlAttr(key)}">${rendered}</parameter>`;
|
|
93
|
+
}
|
|
94
|
+
return `${body}</invoke>`;
|
|
95
|
+
}
|
|
96
|
+
|
|
97
|
+
export function renderInvokes(calls: readonly ToolCall[], tools: NonNullable<DialectRenderOptions["tools"]>): string {
|
|
98
|
+
const shapes = buildArgShapes(tools);
|
|
99
|
+
return calls.map(call => renderInvoke(call, shapes.get(call.name))).join("\n");
|
|
100
|
+
}
|
|
101
|
+
|
|
84
102
|
export type AssistantTranscriptParts = {
|
|
85
103
|
readonly text: string;
|
|
86
104
|
readonly thinking: string;
|
package/src/dialect/xml.ts
CHANGED
|
@@ -1,13 +1,13 @@
|
|
|
1
1
|
import type { Message, ToolCall } from "../types";
|
|
2
2
|
import { AnthropicInbandScanner } from "./anthropic";
|
|
3
|
-
import { buildArgShapes
|
|
3
|
+
import { buildArgShapes } from "./coercion";
|
|
4
4
|
import { DeepSeekInbandScanner } from "./deepseek";
|
|
5
5
|
import {
|
|
6
|
-
escapeXmlAttr,
|
|
7
6
|
renderDelimitedThinking,
|
|
7
|
+
renderInvoke,
|
|
8
|
+
renderInvokes,
|
|
8
9
|
renderLegacyTextTranscript,
|
|
9
10
|
renderToolResponseResults,
|
|
10
|
-
stringifyJson,
|
|
11
11
|
} from "./rendering";
|
|
12
12
|
import type {
|
|
13
13
|
DialectDefinition,
|
|
@@ -60,22 +60,6 @@ function renderTranscript(messages: readonly Message[], options: DialectRenderOp
|
|
|
60
60
|
});
|
|
61
61
|
}
|
|
62
62
|
|
|
63
|
-
function renderInvoke(call: ToolCall, shape: ToolArgShape | undefined): string {
|
|
64
|
-
let body = `<invoke name="${escapeXmlAttr(call.name)}">`;
|
|
65
|
-
for (const key in call.arguments) {
|
|
66
|
-
const value = call.arguments[key];
|
|
67
|
-
const isString = shape?.stringArgs.has(key) === true;
|
|
68
|
-
const rendered = isString && typeof value === "string" ? value : stringifyJson(value);
|
|
69
|
-
body += `<parameter name="${escapeXmlAttr(key)}">${rendered}</parameter>`;
|
|
70
|
-
}
|
|
71
|
-
return `${body}</invoke>`;
|
|
72
|
-
}
|
|
73
|
-
|
|
74
|
-
function renderInvokes(calls: readonly ToolCall[], tools: NonNullable<DialectRenderOptions["tools"]>): string {
|
|
75
|
-
const shapes = buildArgShapes(tools);
|
|
76
|
-
return calls.map(call => renderInvoke(call, shapes.get(call.name))).join("\n");
|
|
77
|
-
}
|
|
78
|
-
|
|
79
63
|
const definition: DialectDefinition = {
|
|
80
64
|
dialect: "xml",
|
|
81
65
|
prompt: dialectPrompt,
|
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
import type { Model } from "@oh-my-pi/pi-catalog/types";
|
|
2
2
|
import * as AIError from "../error";
|
|
3
|
+
import { resolveXaiBaseUrl } from "../providers/xai-base-url";
|
|
3
4
|
import {
|
|
4
5
|
decodeImageResponse,
|
|
5
6
|
imageBaseUrl,
|
|
@@ -54,11 +55,16 @@ export async function generateOpenAIImage(
|
|
|
54
55
|
: { ...generationBody, images: references }
|
|
55
56
|
: { ...generationBody, input_references: references };
|
|
56
57
|
const baseUrl = imageBaseUrl(model);
|
|
58
|
+
// xAI resolves the endpoint per bearer: XAI_BASE_URL never receives an xai-oauth OAuth access token.
|
|
59
|
+
const endpoint = (path: string) =>
|
|
60
|
+
isXAI
|
|
61
|
+
? (bearer: string) => `${resolveXaiBaseUrl(model.provider, baseUrl, bearer) ?? baseUrl}${path}`
|
|
62
|
+
: `${baseUrl}${path}`;
|
|
57
63
|
let response: unknown;
|
|
58
64
|
if (references.length === 0) {
|
|
59
65
|
response = await postJson({
|
|
60
66
|
model,
|
|
61
|
-
url:
|
|
67
|
+
url: endpoint("/images/generations"),
|
|
62
68
|
body: generationBody,
|
|
63
69
|
apiKey: options.apiKey,
|
|
64
70
|
fetch: fetchImpl,
|
|
@@ -78,7 +84,7 @@ export async function generateOpenAIImage(
|
|
|
78
84
|
}
|
|
79
85
|
response = await postMultipart({
|
|
80
86
|
model,
|
|
81
|
-
url:
|
|
87
|
+
url: endpoint("/images/edits"),
|
|
82
88
|
body: form,
|
|
83
89
|
apiKey: options.apiKey,
|
|
84
90
|
fetch: fetchImpl,
|
|
@@ -87,7 +93,7 @@ export async function generateOpenAIImage(
|
|
|
87
93
|
} else {
|
|
88
94
|
response = await postJson({
|
|
89
95
|
model,
|
|
90
|
-
url:
|
|
96
|
+
url: endpoint("/images/edits"),
|
|
91
97
|
body,
|
|
92
98
|
apiKey: options.apiKey,
|
|
93
99
|
fetch: fetchImpl,
|
|
@@ -98,7 +104,7 @@ export async function generateOpenAIImage(
|
|
|
98
104
|
if (!(error instanceof AIError.ProviderHttpError) || error.status !== 404) throw error;
|
|
99
105
|
response = await postJson({
|
|
100
106
|
model,
|
|
101
|
-
url:
|
|
107
|
+
url: endpoint("/images/generations"),
|
|
102
108
|
body,
|
|
103
109
|
apiKey: options.apiKey,
|
|
104
110
|
fetch: fetchImpl,
|
package/src/images/shared.ts
CHANGED
|
@@ -69,9 +69,12 @@ async function parseImageApiResponse(model: Model, response: Response): Promise<
|
|
|
69
69
|
}
|
|
70
70
|
}
|
|
71
71
|
|
|
72
|
+
/** Request URL, or a builder for routes that depend on the bearer (xAI's `XAI_BASE_URL` rule). */
|
|
73
|
+
type ImageRequestUrl = string | ((bearer: string) => string);
|
|
74
|
+
|
|
72
75
|
export async function postJson(options: {
|
|
73
76
|
model: Model;
|
|
74
|
-
url:
|
|
77
|
+
url: ImageRequestUrl;
|
|
75
78
|
body: unknown;
|
|
76
79
|
apiKey: ApiKey;
|
|
77
80
|
fetch: FetchImpl;
|
|
@@ -80,7 +83,8 @@ export async function postJson(options: {
|
|
|
80
83
|
return withAuth(
|
|
81
84
|
options.apiKey,
|
|
82
85
|
async key => {
|
|
83
|
-
const
|
|
86
|
+
const url = typeof options.url === "string" ? options.url : options.url(key);
|
|
87
|
+
const response = await options.fetch(url, {
|
|
84
88
|
method: "POST",
|
|
85
89
|
headers: {
|
|
86
90
|
...(await modelHeaders(options.model, options.signal)),
|
|
@@ -99,7 +103,7 @@ export async function postJson(options: {
|
|
|
99
103
|
|
|
100
104
|
export async function postMultipart(options: {
|
|
101
105
|
model: Model;
|
|
102
|
-
url:
|
|
106
|
+
url: ImageRequestUrl;
|
|
103
107
|
body: FormData;
|
|
104
108
|
apiKey: ApiKey;
|
|
105
109
|
fetch: FetchImpl;
|
|
@@ -108,7 +112,8 @@ export async function postMultipart(options: {
|
|
|
108
112
|
return withAuth(
|
|
109
113
|
options.apiKey,
|
|
110
114
|
async key => {
|
|
111
|
-
const
|
|
115
|
+
const url = typeof options.url === "string" ? options.url : options.url(key);
|
|
116
|
+
const response = await options.fetch(url, {
|
|
112
117
|
method: "POST",
|
|
113
118
|
headers: {
|
|
114
119
|
...(await modelHeaders(options.model, options.signal)),
|
|
@@ -56,6 +56,7 @@ import { invalidateAwsCredentialCache, resolveAwsCredentials } from "./aws-crede
|
|
|
56
56
|
import { decodeEventStream } from "./aws-eventstream";
|
|
57
57
|
import { signRequest } from "./aws-sigv4";
|
|
58
58
|
import { parseAnthropicInputTransformations, THINKING_BINDING_CONTROLS_BETA } from "./anthropic-wire";
|
|
59
|
+
import { isBedrockRequestMetadataValue } from "./bedrock-request-metadata";
|
|
59
60
|
import { transformMessages } from "./transform-messages";
|
|
60
61
|
|
|
61
62
|
/**
|
|
@@ -381,9 +382,7 @@ interface MetadataEvent {
|
|
|
381
382
|
};
|
|
382
383
|
}
|
|
383
384
|
|
|
384
|
-
const REQUEST_METADATA_PATTERN = /^[a-zA-Z0-9\s:_@$#=/+,\-.]*$/;
|
|
385
385
|
const REQUEST_METADATA_MAX_ENTRIES = 16;
|
|
386
|
-
const REQUEST_METADATA_MAX_LENGTH = 256;
|
|
387
386
|
|
|
388
387
|
/**
|
|
389
388
|
* Bedrock rejects the whole invocation on a malformed `requestMetadata` entry.
|
|
@@ -400,10 +399,8 @@ function sanitizeRequestMetadata(raw: unknown): Record<string, string> | undefin
|
|
|
400
399
|
if (
|
|
401
400
|
typeof value !== "string" ||
|
|
402
401
|
key.length < 1 ||
|
|
403
|
-
key
|
|
404
|
-
!
|
|
405
|
-
value.length > REQUEST_METADATA_MAX_LENGTH ||
|
|
406
|
-
!REQUEST_METADATA_PATTERN.test(value) ||
|
|
402
|
+
!isBedrockRequestMetadataValue(key) ||
|
|
403
|
+
!isBedrockRequestMetadataValue(value) ||
|
|
407
404
|
kept >= REQUEST_METADATA_MAX_ENTRIES
|
|
408
405
|
) {
|
|
409
406
|
dropped.push(key);
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { isOfficialAnthropicApiUrl } from "@oh-my-pi/pi-catalog/compat/anthropic";
|
|
1
|
+
import { isBedrockAnthropicRoute, isOfficialAnthropicApiUrl } from "@oh-my-pi/pi-catalog/compat/anthropic";
|
|
2
2
|
import type { Model } from "../types";
|
|
3
3
|
import type { AnthropicMessagesClientLike } from "./anthropic-client";
|
|
4
4
|
import { normalizeAnthropicBaseUrl, resolveDirectAnthropicBaseUrl } from "./anthropic-state";
|
|
@@ -46,6 +46,15 @@ export function supportsAnthropicCompaction(model: Model<"anthropic-messages">,
|
|
|
46
46
|
(model.provider === "anthropic"
|
|
47
47
|
? resolveDirectAnthropicBaseUrl(model)
|
|
48
48
|
: normalizeAnthropicBaseUrl(model.baseUrl));
|
|
49
|
+
// Bedrock's Anthropic Messages API implements on-demand compaction. The flag is detected
|
|
50
|
+
// from a Bedrock `/anthropic` baseUrl, or set in models.yml for a proxy or a reroute; it
|
|
51
|
+
// applies to the model's own endpoint or a Bedrock `/anthropic` route it reaches.
|
|
52
|
+
if (
|
|
53
|
+
model.compat.bedrockMessagesApi === true &&
|
|
54
|
+
(isBedrockAnthropicRoute(route) || route === normalizeAnthropicBaseUrl(model.baseUrl))
|
|
55
|
+
) {
|
|
56
|
+
return true;
|
|
57
|
+
}
|
|
49
58
|
return (
|
|
50
59
|
isSupportedCompactionEndpoint(route) &&
|
|
51
60
|
(model.compat.firstPartyProvider === true ||
|
|
@@ -272,7 +272,18 @@ export type ThinkingConfigAdaptive = {
|
|
|
272
272
|
block_binding?: ThinkingBlockBinding;
|
|
273
273
|
};
|
|
274
274
|
|
|
275
|
-
|
|
275
|
+
/**
|
|
276
|
+
* Sonnet 5.5's replacement for `disabled`: no up-front thinking, progress
|
|
277
|
+
* updates between tool calls only. Takes no other field, and effort above
|
|
278
|
+
* `high` is rejected alongside it.
|
|
279
|
+
*/
|
|
280
|
+
export type ThinkingConfigBetweenTools = { type: "between_tools" };
|
|
281
|
+
|
|
282
|
+
export type ThinkingConfigParam =
|
|
283
|
+
| ThinkingConfigEnabled
|
|
284
|
+
| ThinkingConfigDisabled
|
|
285
|
+
| ThinkingConfigAdaptive
|
|
286
|
+
| ThinkingConfigBetweenTools;
|
|
276
287
|
|
|
277
288
|
export type OutputConfig = {
|
|
278
289
|
/** Adaptive-thinking effort level (effort beta). */
|
|
@@ -155,6 +155,7 @@ import {
|
|
|
155
155
|
resolveAnthropicMetadataUserId,
|
|
156
156
|
stripClaudeToolPrefix,
|
|
157
157
|
} from "./anthropic-identity";
|
|
158
|
+
import { fitBedrockAnthropicPayload } from "./bedrock-anthropic";
|
|
158
159
|
import {
|
|
159
160
|
anthropicProviderSessionStateKey,
|
|
160
161
|
clearAnthropicFastModeFallback,
|
|
@@ -2264,6 +2265,8 @@ const streamAnthropicOnce = (
|
|
|
2264
2265
|
nextParams = replacementPayload as typeof nextParams;
|
|
2265
2266
|
}
|
|
2266
2267
|
if (nextParams.compaction) stripCompactionIncompatibleParams(nextParams);
|
|
2268
|
+
// After `onPayload`, so a hook cannot restore a field Bedrock rejects.
|
|
2269
|
+
if (model.compat.bedrockMessagesApi) fitBedrockAnthropicPayload(nextParams);
|
|
2267
2270
|
nextParams = toWellFormedDeep(nextParams) as typeof nextParams;
|
|
2268
2271
|
rawRequestDump = {
|
|
2269
2272
|
provider: model.provider,
|
|
@@ -3827,22 +3830,17 @@ function ensureMaxTokensForThinking(params: MessageCreateParamsStreaming, maxAll
|
|
|
3827
3830
|
const budgetTokens = thinking.budget_tokens ?? 0;
|
|
3828
3831
|
if (budgetTokens <= 0) return;
|
|
3829
3832
|
|
|
3830
|
-
const
|
|
3831
|
-
|
|
3832
|
-
Math.max(currentMaxTokens, budgetTokens + OUTPUT_FALLBACK_BUFFER),
|
|
3833
|
-
maxAllowedTokens,
|
|
3834
|
-
);
|
|
3835
|
-
params.max_tokens = raisedMaxTokens;
|
|
3833
|
+
const output = budgetThinkingOutput(params.max_tokens, budgetTokens, maxAllowedTokens);
|
|
3834
|
+
params.max_tokens = output.maxTokens;
|
|
3836
3835
|
|
|
3837
|
-
if (budgetTokens
|
|
3836
|
+
if (output.budgetTokens === budgetTokens) return;
|
|
3838
3837
|
|
|
3839
|
-
|
|
3840
|
-
if (clampedBudget <= 0) {
|
|
3838
|
+
if (output.budgetTokens <= 0) {
|
|
3841
3839
|
throw new AIError.ConfigurationError(
|
|
3842
|
-
`Anthropic thinking budget requires max_tokens greater than ${OUTPUT_FALLBACK_BUFFER}; got ${
|
|
3840
|
+
`Anthropic thinking budget requires max_tokens greater than ${OUTPUT_FALLBACK_BUFFER}; got ${output.maxTokens}`,
|
|
3843
3841
|
);
|
|
3844
3842
|
}
|
|
3845
|
-
thinking.budget_tokens =
|
|
3843
|
+
thinking.budget_tokens = output.budgetTokens;
|
|
3846
3844
|
}
|
|
3847
3845
|
|
|
3848
3846
|
function applyCacheControlToLastBlock(blocks: ContentBlockParam[], cacheControl: AnthropicCacheControl): boolean {
|
|
@@ -4128,6 +4126,41 @@ function usesAdaptiveThinkingTagOnly(model: Model<"anthropic-messages">): boolea
|
|
|
4128
4126
|
return thinking.efforts.length > 0;
|
|
4129
4127
|
}
|
|
4130
4128
|
|
|
4129
|
+
/**
|
|
4130
|
+
* True when enabled thinking on `model` is budget thinking
|
|
4131
|
+
* (`thinking.type: "enabled"` with `budget_tokens`) rather than adaptive.
|
|
4132
|
+
*/
|
|
4133
|
+
export function usesBudgetThinking(model: Model<"anthropic-messages">): boolean {
|
|
4134
|
+
return model.thinking?.mode !== "anthropic-adaptive" || model.compat.disableAdaptiveThinking === true;
|
|
4135
|
+
}
|
|
4136
|
+
|
|
4137
|
+
/** The most output tokens a request to `model` may ask for (`max_tokens` ceiling). */
|
|
4138
|
+
export function anthropicOutputLimit(model: Model<"anthropic-messages">): number {
|
|
4139
|
+
return model.maxTokens ?? UNKNOWN_MODEL_MAX_OUTPUT_TOKENS;
|
|
4140
|
+
}
|
|
4141
|
+
|
|
4142
|
+
/**
|
|
4143
|
+
* The `max_tokens` and thinking budget of budget thinking: `max_tokens`
|
|
4144
|
+
* rises to leave {@link OUTPUT_FALLBACK_BUFFER} visible output tokens after
|
|
4145
|
+
* the budget, within `maxAllowedTokens`, and the budget shrinks when that
|
|
4146
|
+
* ceiling leaves less (a non-positive budget means the ceiling is too low).
|
|
4147
|
+
*/
|
|
4148
|
+
export function budgetThinkingOutput(
|
|
4149
|
+
maxTokens: number | undefined,
|
|
4150
|
+
budgetTokens: number,
|
|
4151
|
+
maxAllowedTokens: number,
|
|
4152
|
+
): { maxTokens: number; budgetTokens: number } {
|
|
4153
|
+
const currentMaxTokens = Math.min(maxTokens ?? maxAllowedTokens, maxAllowedTokens);
|
|
4154
|
+
const raisedMaxTokens = Math.min(
|
|
4155
|
+
Math.max(currentMaxTokens, budgetTokens + OUTPUT_FALLBACK_BUFFER),
|
|
4156
|
+
maxAllowedTokens,
|
|
4157
|
+
);
|
|
4158
|
+
return {
|
|
4159
|
+
maxTokens: raisedMaxTokens,
|
|
4160
|
+
budgetTokens: Math.min(budgetTokens, raisedMaxTokens - OUTPUT_FALLBACK_BUFFER),
|
|
4161
|
+
};
|
|
4162
|
+
}
|
|
4163
|
+
|
|
4131
4164
|
/**
|
|
4132
4165
|
* True for adaptive-only Claude models (Opus 4.6+, Sonnet 4.6+, Fable/Mythos 5)
|
|
4133
4166
|
* that reject `thinking.type: "disabled"`. Turning thinking off on these models
|
|
@@ -4550,8 +4583,7 @@ function buildParams(
|
|
|
4550
4583
|
const thinkingOptions = options ?? {};
|
|
4551
4584
|
const mode = model.thinking?.mode;
|
|
4552
4585
|
const effort = resolveAnthropicAdaptiveEffort(model, thinkingOptions);
|
|
4553
|
-
|
|
4554
|
-
if (mode === "anthropic-adaptive" && !compat.disableAdaptiveThinking) {
|
|
4586
|
+
if (!usesBudgetThinking(model)) {
|
|
4555
4587
|
const adaptive: { type: "adaptive"; display?: AnthropicThinkingDisplay } = { type: "adaptive" };
|
|
4556
4588
|
// Starting with Claude Opus 4.7 and Claude Fable/Mythos 5, adaptive thinking
|
|
4557
4589
|
// content is omitted from the response by default. Opt into summarized
|
|
@@ -4573,7 +4605,12 @@ function buildParams(
|
|
|
4573
4605
|
if (mode === "anthropic-budget-effort" && effort && effort !== "adaptive") outputConfigEffort = effort;
|
|
4574
4606
|
}
|
|
4575
4607
|
} else if (options?.thinkingEnabled === false) {
|
|
4576
|
-
if (
|
|
4608
|
+
if (model.compat.supportsBetweenToolsThinking) {
|
|
4609
|
+
// Sonnet 5.5 rejects `disabled` with a 400; `between_tools` is its lowest
|
|
4610
|
+
// thinking setting. It takes no other field and leaves effort untouched:
|
|
4611
|
+
// pinning `low` here would cap the whole turn's quality, not only thinking.
|
|
4612
|
+
thinking = { type: "between_tools" };
|
|
4613
|
+
} else if (isAdaptiveOnlyThinking(model)) {
|
|
4577
4614
|
// Adaptive-only Claude models (Opus 4.6+, Sonnet 4.6+, Fable/Mythos 5) reject
|
|
4578
4615
|
// `thinking.type: "disabled"` — adaptive thinking cannot be switched off.
|
|
4579
4616
|
// Omit the thinking field (the API defaults to adaptive) and pin the
|
|
@@ -4643,6 +4680,12 @@ function buildParams(
|
|
|
4643
4680
|
model.compat.supportsPerMessageEffort === true,
|
|
4644
4681
|
compactionReplay,
|
|
4645
4682
|
);
|
|
4683
|
+
// `between_tools` returns a 400 at `xhigh`/`max` effort, and the effort in
|
|
4684
|
+
// force from earlier turns outlives a thinking toggle. Fall back to the
|
|
4685
|
+
// default adaptive request, which accepts every effort level.
|
|
4686
|
+
if (thinking?.type === "between_tools" && (effortPlan.topLevel === "xhigh" || effortPlan.topLevel === "max")) {
|
|
4687
|
+
thinking = undefined;
|
|
4688
|
+
}
|
|
4646
4689
|
const wireMessages = convertAnthropicMessages(
|
|
4647
4690
|
insertAnthropicControlMarkers(context.messages, [...toolPlan.inserts, ...effortPlan.inserts]),
|
|
4648
4691
|
effectiveModel,
|
|
@@ -4683,7 +4726,7 @@ function buildParams(
|
|
|
4683
4726
|
|
|
4684
4727
|
// OAuth and API-key requests alike get the full model ceiling; Claude Code
|
|
4685
4728
|
// itself requests 128k on Opus 5.5.
|
|
4686
|
-
const maxOutputTokens = model
|
|
4729
|
+
const maxOutputTokens = anthropicOutputLimit(model);
|
|
4687
4730
|
|
|
4688
4731
|
// A caller-owned client targets its own endpoint: route body betas by the
|
|
4689
4732
|
// client's URL when it exposes one, not the model's routing. Otherwise the
|
|
@@ -61,7 +61,7 @@ const UNSIGNABLE: Record<string, true> = {
|
|
|
61
61
|
* `ArrayBuffer`, which is what `crypto.subtle.{digest,sign,importKey}` requires
|
|
62
62
|
* under the strict TS DOM typings. No-op when already strict.
|
|
63
63
|
*/
|
|
64
|
-
function asStrict(bytes: Uint8Array): Uint8Array<ArrayBuffer> {
|
|
64
|
+
export function asStrict(bytes: Uint8Array): Uint8Array<ArrayBuffer> {
|
|
65
65
|
if (bytes.buffer instanceof ArrayBuffer && bytes.byteOffset === 0 && bytes.byteLength === bytes.buffer.byteLength) {
|
|
66
66
|
return bytes as Uint8Array<ArrayBuffer>;
|
|
67
67
|
}
|
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
import { isRecord } from "@oh-my-pi/pi-utils";
|
|
2
|
+
import { extractClaudeMetadataSessionId } from "./anthropic-identity";
|
|
3
|
+
import { isBedrockRequestMetadataValue } from "./bedrock-request-metadata";
|
|
4
|
+
|
|
5
|
+
/**
|
|
6
|
+
* Fit an Anthropic request body to Bedrock's Anthropic Messages API
|
|
7
|
+
* (`compat.bedrockMessagesApi`): both `/anthropic` routes reject the tool
|
|
8
|
+
* `strict` field, and bedrock-runtime rejects a `metadata.user_id` outside
|
|
9
|
+
* Bedrock's request-metadata pattern. A user id that fits is kept, otherwise
|
|
10
|
+
* its embedded session id, otherwise the metadata is dropped. Mutates and
|
|
11
|
+
* returns `payload`.
|
|
12
|
+
*/
|
|
13
|
+
export function fitBedrockAnthropicPayload<T>(payload: T): T {
|
|
14
|
+
if (!isRecord(payload)) return payload;
|
|
15
|
+
const body: Record<string, unknown> = payload;
|
|
16
|
+
if (Array.isArray(body.tools)) {
|
|
17
|
+
for (const tool of body.tools) {
|
|
18
|
+
if (isRecord(tool)) delete tool.strict;
|
|
19
|
+
}
|
|
20
|
+
}
|
|
21
|
+
if (body.metadata === undefined) return payload;
|
|
22
|
+
const userId = isRecord(body.metadata) ? body.metadata.user_id : undefined;
|
|
23
|
+
const fitted =
|
|
24
|
+
typeof userId === "string" && isBedrockRequestMetadataValue(userId)
|
|
25
|
+
? userId
|
|
26
|
+
: extractClaudeMetadataSessionId(userId);
|
|
27
|
+
if (fitted && isBedrockRequestMetadataValue(fitted)) body.metadata = { user_id: fitted };
|
|
28
|
+
else delete body.metadata;
|
|
29
|
+
return payload;
|
|
30
|
+
}
|
|
@@ -0,0 +1,6 @@
|
|
|
1
|
+
const BEDROCK_REQUEST_METADATA_PATTERN = /^[a-zA-Z0-9\s:_@$#=/+,\-.]*$/;
|
|
2
|
+
|
|
3
|
+
/** Check Bedrock's request-metadata character and length limits. Keys must also be nonempty. */
|
|
4
|
+
export function isBedrockRequestMetadataValue(value: string): boolean {
|
|
5
|
+
return value.length <= 256 && BEDROCK_REQUEST_METADATA_PATTERN.test(value);
|
|
6
|
+
}
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { truncate } from "@oh-my-pi/pi-utils";
|
|
1
|
+
import { isRecord, truncate } from "@oh-my-pi/pi-utils";
|
|
2
2
|
|
|
3
3
|
/**
|
|
4
4
|
* Connect-protocol end-stream error formatting.
|
|
@@ -21,10 +21,6 @@ const GENERIC_CONNECT_ERROR_MESSAGES = new Set(["", "error", "unknown", "unknown
|
|
|
21
21
|
/** Upper bound for appended trailer context so errors stay log-line sized. */
|
|
22
22
|
const MAX_EXTRA_DETAIL_CHARS = 400;
|
|
23
23
|
|
|
24
|
-
function isRecord(value: unknown): value is Record<string, unknown> {
|
|
25
|
-
return typeof value === "object" && value !== null && !Array.isArray(value);
|
|
26
|
-
}
|
|
27
|
-
|
|
28
24
|
function safeJson(value: unknown): string | undefined {
|
|
29
25
|
try {
|
|
30
26
|
const text = typeof value === "string" ? value : JSON.stringify(value);
|
|
@@ -32,7 +32,8 @@ type ProtoUnknownBag = { $unknown?: ProtoUnknownField[] };
|
|
|
32
32
|
type InteractionQueryCase = NonNullable<InteractionQuery["query"]["case"]>;
|
|
33
33
|
type InteractionResult = Exclude<InteractionResponse["result"], { case: undefined; value?: undefined }>;
|
|
34
34
|
|
|
35
|
-
|
|
35
|
+
/** Wrap one Connect-protocol message: 1 flag byte + 4-byte big-endian length + payload. */
|
|
36
|
+
export function frameConnectMessage(data: Uint8Array, flags = 0): Buffer {
|
|
36
37
|
const frame = Buffer.alloc(5 + data.length);
|
|
37
38
|
frame[0] = flags;
|
|
38
39
|
frame.writeUInt32BE(data.length, 1);
|
|
@@ -46,7 +47,8 @@ function isProtoUnknownField(value: unknown): value is ProtoUnknownField {
|
|
|
46
47
|
return typeof value.no === "number" && typeof value.wireType === "number" && value.data instanceof Uint8Array;
|
|
47
48
|
}
|
|
48
49
|
|
|
49
|
-
|
|
50
|
+
/** Well-formed protobuf-es `$unknown` entries on `message`; anything else on the bag is ignored. */
|
|
51
|
+
export function protoUnknownFields(message: object): ProtoUnknownField[] {
|
|
50
52
|
if (!("$unknown" in message) || !Array.isArray(message.$unknown)) return [];
|
|
51
53
|
return message.$unknown.filter(isProtoUnknownField);
|
|
52
54
|
}
|
package/src/providers/cursor.ts
CHANGED
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
import * as fs from "node:fs/promises";
|
|
2
2
|
import http2 from "node:http2";
|
|
3
|
+
import { cursorModelParameters } from "@oh-my-pi/pi-catalog/compat/behavior";
|
|
3
4
|
import { isCursorMaxModeWireId } from "@oh-my-pi/pi-catalog/compat/collapse";
|
|
4
5
|
import { classifyModel, collapseVariantId } from "@oh-my-pi/pi-catalog/compat/taxonomy";
|
|
5
6
|
import type {
|
|
@@ -246,7 +247,7 @@ import {
|
|
|
246
247
|
piTimeout,
|
|
247
248
|
shellTimeoutSeconds,
|
|
248
249
|
} from "./cursor/exec-modern";
|
|
249
|
-
import { handleInteractionQuery } from "./cursor/interaction-query";
|
|
250
|
+
import { frameConnectMessage, handleInteractionQuery, protoUnknownFields } from "./cursor/interaction-query";
|
|
250
251
|
|
|
251
252
|
export const CURSOR_API_URL = "https://api2.cursor.sh";
|
|
252
253
|
export const CURSOR_CLIENT_VERSION = "cli-2026.07.23-e383d2b";
|
|
@@ -407,14 +408,14 @@ function log(type: string, subtype?: string, data?: unknown): void {
|
|
|
407
408
|
void appendCursorDebugLog(entry);
|
|
408
409
|
}
|
|
409
410
|
|
|
410
|
-
|
|
411
|
-
|
|
412
|
-
|
|
413
|
-
|
|
414
|
-
|
|
415
|
-
|
|
411
|
+
/**
|
|
412
|
+
* Write one client message. Once the server's end frame has half-closed our
|
|
413
|
+
* side, late writes (heartbeats, exec replies from a handler still running)
|
|
414
|
+
* are dropped: writing after `end()` would error the stream.
|
|
415
|
+
*/
|
|
416
|
+
function writeClientMessage(h2Request: http2.ClientHttp2Stream, data: Uint8Array): void {
|
|
417
|
+
if (!h2Request.writableEnded) h2Request.write(frameConnectMessage(data));
|
|
416
418
|
}
|
|
417
|
-
|
|
418
419
|
class ConnectEndStreamError extends AIError.ProviderResponseError {
|
|
419
420
|
readonly diagnosticMessage: string;
|
|
420
421
|
|
|
@@ -845,6 +846,11 @@ function streamCursorWithWireMode(
|
|
|
845
846
|
if (endError) {
|
|
846
847
|
endStreamError = endError;
|
|
847
848
|
h2Request?.close();
|
|
849
|
+
} else {
|
|
850
|
+
// The end frame is the server's last message. Half-close our
|
|
851
|
+
// side so the stream can finish: a CONNECT proxy holds the
|
|
852
|
+
// HTTP/2 stream open until the client ends its request.
|
|
853
|
+
h2Request?.end();
|
|
848
854
|
}
|
|
849
855
|
continue;
|
|
850
856
|
}
|
|
@@ -903,7 +909,7 @@ function streamCursorWithWireMode(
|
|
|
903
909
|
message: { case: "clientHeartbeat", value: create(ClientHeartbeatSchema, {}) },
|
|
904
910
|
});
|
|
905
911
|
const heartbeatBytes = toBinary(AgentClientMessageSchema, heartbeatMessage);
|
|
906
|
-
h2Request
|
|
912
|
+
writeClientMessage(h2Request, heartbeatBytes);
|
|
907
913
|
};
|
|
908
914
|
|
|
909
915
|
const closeDebugLog = async (): Promise<void> => {
|
|
@@ -1207,8 +1213,6 @@ export async function handleServerMessage(
|
|
|
1207
1213
|
}
|
|
1208
1214
|
}
|
|
1209
1215
|
|
|
1210
|
-
type ProtoUnknownField = { no: number; wireType: number; data: Uint8Array };
|
|
1211
|
-
|
|
1212
1216
|
type HostedFetchCall = {
|
|
1213
1217
|
args?: { url?: string; toolCallId?: string };
|
|
1214
1218
|
result?: { result?: { case?: string; value?: { content?: string; error?: string; url?: string } } };
|
|
@@ -1250,11 +1254,6 @@ function describeHostedFetchResult(call: HostedFetchCall | undefined): { text: s
|
|
|
1250
1254
|
return { text: "Fetch completed", isError: false };
|
|
1251
1255
|
}
|
|
1252
1256
|
|
|
1253
|
-
function protoUnknownFields(message: object): ProtoUnknownField[] {
|
|
1254
|
-
const raw = (message as { $unknown?: ProtoUnknownField[] }).$unknown;
|
|
1255
|
-
return Array.isArray(raw) ? raw : [];
|
|
1256
|
-
}
|
|
1257
|
-
|
|
1258
1257
|
function handleKvServerMessage(
|
|
1259
1258
|
kvMsg: KvServerMessage,
|
|
1260
1259
|
blobStore: Map<string, Uint8Array>,
|
|
@@ -1281,7 +1280,7 @@ function handleKvServerMessage(
|
|
|
1281
1280
|
});
|
|
1282
1281
|
|
|
1283
1282
|
const responseBytes = toBinary(AgentClientMessageSchema, kvClientMessage);
|
|
1284
|
-
h2Request
|
|
1283
|
+
writeClientMessage(h2Request, responseBytes);
|
|
1285
1284
|
|
|
1286
1285
|
log("kvClient", "getBlobResult", { blobId: blobIdKey.slice(0, 40) });
|
|
1287
1286
|
} else if (kvCase === "setBlobArgs") {
|
|
@@ -1302,7 +1301,7 @@ function handleKvServerMessage(
|
|
|
1302
1301
|
});
|
|
1303
1302
|
|
|
1304
1303
|
const responseBytes = toBinary(AgentClientMessageSchema, kvClientMessage);
|
|
1305
|
-
h2Request
|
|
1304
|
+
writeClientMessage(h2Request, responseBytes);
|
|
1306
1305
|
|
|
1307
1306
|
log("kvClient", "setBlobResult", { blobId: blobIdKey.slice(0, 40) });
|
|
1308
1307
|
}
|
|
@@ -2604,7 +2603,7 @@ function sendExecClientMessage<TCase extends NonNullable<ExecClientMessage["mess
|
|
|
2604
2603
|
});
|
|
2605
2604
|
|
|
2606
2605
|
const responseBytes = toBinary(AgentClientMessageSchema, clientMessage);
|
|
2607
|
-
h2Request
|
|
2606
|
+
writeClientMessage(h2Request, responseBytes);
|
|
2608
2607
|
|
|
2609
2608
|
log("execClientMessage", messageCase, value);
|
|
2610
2609
|
}
|
|
@@ -2640,7 +2639,7 @@ function sendExecClientThrow(
|
|
|
2640
2639
|
const clientMessage = create(AgentClientMessageSchema, {
|
|
2641
2640
|
message: { case: "execClientControlMessage", value: controlMessage },
|
|
2642
2641
|
});
|
|
2643
|
-
h2Request
|
|
2642
|
+
writeClientMessage(h2Request, toBinary(AgentClientMessageSchema, clientMessage));
|
|
2644
2643
|
log("execClientControl", "throw", { id: execMsg.id, execId: execMsg.execId, error, errorCode });
|
|
2645
2644
|
sendExecClientStreamClose(h2Request, execMsg);
|
|
2646
2645
|
}
|
|
@@ -2658,7 +2657,7 @@ function sendExecClientStreamClose(h2Request: http2.ClientHttp2Stream, execMsg:
|
|
|
2658
2657
|
message: { case: "execClientControlMessage", value: closeMessage },
|
|
2659
2658
|
});
|
|
2660
2659
|
const responseBytes = toBinary(AgentClientMessageSchema, clientMessage);
|
|
2661
|
-
h2Request
|
|
2660
|
+
writeClientMessage(h2Request, responseBytes);
|
|
2662
2661
|
log("execClientControl", "streamClose", { id: execMsg.id, execId: execMsg.execId });
|
|
2663
2662
|
}
|
|
2664
2663
|
|
|
@@ -5487,13 +5486,18 @@ function resolveCursorWireModel(
|
|
|
5487
5486
|
};
|
|
5488
5487
|
}
|
|
5489
5488
|
}
|
|
5490
|
-
//
|
|
5491
|
-
//
|
|
5492
|
-
//
|
|
5493
|
-
|
|
5489
|
+
// Fixed per-model parameters come from catalog KDL (`cursor-model-parameter`
|
|
5490
|
+
// in `runtime/behavior.kdl`). A bare `composer-2.5` id resolves to the Fast
|
|
5491
|
+
// variant server-side (can1357/oh-my-pi#9012), so the catalog pins the
|
|
5492
|
+
// Standard tier with `fast=false`; `-fast` selections keep the Fast lane by
|
|
5493
|
+
// declaring no parameter.
|
|
5494
|
+
const fixedParameters = cursorModelParameters(wireModelId);
|
|
5495
|
+
if (fixedParameters.length > 0) {
|
|
5494
5496
|
return {
|
|
5495
5497
|
modelId: wireModelId,
|
|
5496
|
-
parameters:
|
|
5498
|
+
parameters: fixedParameters.map(({ id, value }) =>
|
|
5499
|
+
create(RequestedModel_ModelParameterbytesSchema, { id, value }),
|
|
5500
|
+
),
|
|
5497
5501
|
maxMode,
|
|
5498
5502
|
};
|
|
5499
5503
|
}
|