@oh-my-pi/pi-ai 18.4.2 → 18.4.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +30 -41
- package/dist/types/auth-broker/protocol.d.ts +12 -0
- package/dist/types/dialect/rendering.d.ts +4 -0
- package/dist/types/images/openai-hosted.d.ts +1 -1
- package/dist/types/images/shared.d.ts +5 -2
- package/dist/types/providers/anthropic-wire.d.ts +9 -1
- package/dist/types/providers/anthropic.d.ts +17 -0
- package/dist/types/providers/aws-sigv4.d.ts +5 -0
- package/dist/types/providers/bedrock-anthropic.d.ts +9 -0
- package/dist/types/providers/bedrock-request-metadata.d.ts +2 -0
- package/dist/types/providers/cursor/interaction-query.d.ts +10 -0
- package/dist/types/providers/openai-chat-wire.d.ts +2 -2
- package/dist/types/providers/openai-codex/request-transformer.d.ts +1 -1
- package/dist/types/providers/openai-responses-wire.d.ts +2 -2
- package/dist/types/providers/xai-base-url.d.ts +17 -0
- package/dist/types/types.d.ts +14 -3
- package/dist/types/usage/commandcode.d.ts +4 -0
- package/dist/types/usage/shared.d.ts +13 -1
- package/dist/types/utils/event-stream.d.ts +7 -0
- package/dist/types/utils/openai-http.d.ts +7 -1
- package/package.json +6 -6
- package/src/auth-broker/client.ts +1 -12
- package/src/auth-broker/protocol.ts +32 -0
- package/src/auth-broker/remote-store.ts +4 -40
- package/src/auth-broker/server.ts +1 -20
- package/src/auth-broker/snapshot-cache.ts +1 -9
- package/src/dialect/anthropic.ts +3 -25
- package/src/dialect/minimax.ts +3 -24
- package/src/dialect/rendering.ts +18 -0
- package/src/dialect/thinking.ts +24 -6
- package/src/dialect/xml.ts +3 -19
- package/src/images/openai-hosted.ts +5 -2
- package/src/images/openai-images.ts +10 -4
- package/src/images/shared.ts +9 -4
- package/src/providers/amazon-bedrock.ts +61 -22
- package/src/providers/anthropic-compaction.ts +10 -1
- package/src/providers/anthropic-wire.ts +12 -1
- package/src/providers/anthropic.ts +58 -15
- package/src/providers/aws-eventstream.ts +4 -3
- package/src/providers/aws-sigv4.ts +1 -1
- package/src/providers/azure-openai-responses.ts +3 -4
- package/src/providers/bedrock-anthropic.ts +30 -0
- package/src/providers/bedrock-request-metadata.ts +6 -0
- package/src/providers/connect-error-detail.ts +1 -5
- package/src/providers/cursor/interaction-query.ts +4 -2
- package/src/providers/cursor.ts +30 -26
- package/src/providers/google-gemini-cli.ts +9 -2
- package/src/providers/google-shared.ts +16 -6
- package/src/providers/openai-chat-wire.ts +2 -2
- package/src/providers/openai-codex/request-transformer.ts +1 -1
- package/src/providers/openai-codex-responses.ts +27 -16
- package/src/providers/openai-completions.ts +51 -36
- package/src/providers/openai-responses-wire.ts +2 -2
- package/src/providers/openai-responses.ts +3 -4
- package/src/providers/openai-shared.ts +130 -18
- package/src/providers/xai-base-url.ts +32 -0
- package/src/registry/engine/api-key.ts +8 -3
- package/src/types.ts +35 -4
- package/src/usage/claude.ts +4 -11
- package/src/usage/cline-pass.ts +2 -14
- package/src/usage/commandcode.ts +209 -0
- package/src/usage/cursor.ts +9 -1
- package/src/usage/openai-codex.ts +3 -5
- package/src/usage/registry.ts +3 -0
- package/src/usage/shared.ts +28 -1
- package/src/usage/synthetic.ts +4 -40
- package/src/usage/umans.ts +8 -36
- package/src/usage/zai.ts +10 -38
- package/src/utils/event-stream.ts +38 -2
- package/src/utils/http-inspector.ts +4 -8
- package/src/utils/openai-http.ts +10 -3
- package/src/utils/schema/json-schema-validator.ts +23 -26
- package/src/utils/schema/meta-validator.ts +4 -7
- package/src/utils/schema/wire.ts +17 -21
|
@@ -155,6 +155,7 @@ import {
|
|
|
155
155
|
resolveAnthropicMetadataUserId,
|
|
156
156
|
stripClaudeToolPrefix,
|
|
157
157
|
} from "./anthropic-identity";
|
|
158
|
+
import { fitBedrockAnthropicPayload } from "./bedrock-anthropic";
|
|
158
159
|
import {
|
|
159
160
|
anthropicProviderSessionStateKey,
|
|
160
161
|
clearAnthropicFastModeFallback,
|
|
@@ -2264,6 +2265,8 @@ const streamAnthropicOnce = (
|
|
|
2264
2265
|
nextParams = replacementPayload as typeof nextParams;
|
|
2265
2266
|
}
|
|
2266
2267
|
if (nextParams.compaction) stripCompactionIncompatibleParams(nextParams);
|
|
2268
|
+
// After `onPayload`, so a hook cannot restore a field Bedrock rejects.
|
|
2269
|
+
if (model.compat.bedrockMessagesApi) fitBedrockAnthropicPayload(nextParams);
|
|
2267
2270
|
nextParams = toWellFormedDeep(nextParams) as typeof nextParams;
|
|
2268
2271
|
rawRequestDump = {
|
|
2269
2272
|
provider: model.provider,
|
|
@@ -3827,22 +3830,17 @@ function ensureMaxTokensForThinking(params: MessageCreateParamsStreaming, maxAll
|
|
|
3827
3830
|
const budgetTokens = thinking.budget_tokens ?? 0;
|
|
3828
3831
|
if (budgetTokens <= 0) return;
|
|
3829
3832
|
|
|
3830
|
-
const
|
|
3831
|
-
|
|
3832
|
-
Math.max(currentMaxTokens, budgetTokens + OUTPUT_FALLBACK_BUFFER),
|
|
3833
|
-
maxAllowedTokens,
|
|
3834
|
-
);
|
|
3835
|
-
params.max_tokens = raisedMaxTokens;
|
|
3833
|
+
const output = budgetThinkingOutput(params.max_tokens, budgetTokens, maxAllowedTokens);
|
|
3834
|
+
params.max_tokens = output.maxTokens;
|
|
3836
3835
|
|
|
3837
|
-
if (budgetTokens
|
|
3836
|
+
if (output.budgetTokens === budgetTokens) return;
|
|
3838
3837
|
|
|
3839
|
-
|
|
3840
|
-
if (clampedBudget <= 0) {
|
|
3838
|
+
if (output.budgetTokens <= 0) {
|
|
3841
3839
|
throw new AIError.ConfigurationError(
|
|
3842
|
-
`Anthropic thinking budget requires max_tokens greater than ${OUTPUT_FALLBACK_BUFFER}; got ${
|
|
3840
|
+
`Anthropic thinking budget requires max_tokens greater than ${OUTPUT_FALLBACK_BUFFER}; got ${output.maxTokens}`,
|
|
3843
3841
|
);
|
|
3844
3842
|
}
|
|
3845
|
-
thinking.budget_tokens =
|
|
3843
|
+
thinking.budget_tokens = output.budgetTokens;
|
|
3846
3844
|
}
|
|
3847
3845
|
|
|
3848
3846
|
function applyCacheControlToLastBlock(blocks: ContentBlockParam[], cacheControl: AnthropicCacheControl): boolean {
|
|
@@ -4128,6 +4126,41 @@ function usesAdaptiveThinkingTagOnly(model: Model<"anthropic-messages">): boolea
|
|
|
4128
4126
|
return thinking.efforts.length > 0;
|
|
4129
4127
|
}
|
|
4130
4128
|
|
|
4129
|
+
/**
|
|
4130
|
+
* True when enabled thinking on `model` is budget thinking
|
|
4131
|
+
* (`thinking.type: "enabled"` with `budget_tokens`) rather than adaptive.
|
|
4132
|
+
*/
|
|
4133
|
+
export function usesBudgetThinking(model: Model<"anthropic-messages">): boolean {
|
|
4134
|
+
return model.thinking?.mode !== "anthropic-adaptive" || model.compat.disableAdaptiveThinking === true;
|
|
4135
|
+
}
|
|
4136
|
+
|
|
4137
|
+
/** The most output tokens a request to `model` may ask for (`max_tokens` ceiling). */
|
|
4138
|
+
export function anthropicOutputLimit(model: Model<"anthropic-messages">): number {
|
|
4139
|
+
return model.maxTokens ?? UNKNOWN_MODEL_MAX_OUTPUT_TOKENS;
|
|
4140
|
+
}
|
|
4141
|
+
|
|
4142
|
+
/**
|
|
4143
|
+
* The `max_tokens` and thinking budget of budget thinking: `max_tokens`
|
|
4144
|
+
* rises to leave {@link OUTPUT_FALLBACK_BUFFER} visible output tokens after
|
|
4145
|
+
* the budget, within `maxAllowedTokens`, and the budget shrinks when that
|
|
4146
|
+
* ceiling leaves less (a non-positive budget means the ceiling is too low).
|
|
4147
|
+
*/
|
|
4148
|
+
export function budgetThinkingOutput(
|
|
4149
|
+
maxTokens: number | undefined,
|
|
4150
|
+
budgetTokens: number,
|
|
4151
|
+
maxAllowedTokens: number,
|
|
4152
|
+
): { maxTokens: number; budgetTokens: number } {
|
|
4153
|
+
const currentMaxTokens = Math.min(maxTokens ?? maxAllowedTokens, maxAllowedTokens);
|
|
4154
|
+
const raisedMaxTokens = Math.min(
|
|
4155
|
+
Math.max(currentMaxTokens, budgetTokens + OUTPUT_FALLBACK_BUFFER),
|
|
4156
|
+
maxAllowedTokens,
|
|
4157
|
+
);
|
|
4158
|
+
return {
|
|
4159
|
+
maxTokens: raisedMaxTokens,
|
|
4160
|
+
budgetTokens: Math.min(budgetTokens, raisedMaxTokens - OUTPUT_FALLBACK_BUFFER),
|
|
4161
|
+
};
|
|
4162
|
+
}
|
|
4163
|
+
|
|
4131
4164
|
/**
|
|
4132
4165
|
* True for adaptive-only Claude models (Opus 4.6+, Sonnet 4.6+, Fable/Mythos 5)
|
|
4133
4166
|
* that reject `thinking.type: "disabled"`. Turning thinking off on these models
|
|
@@ -4550,8 +4583,7 @@ function buildParams(
|
|
|
4550
4583
|
const thinkingOptions = options ?? {};
|
|
4551
4584
|
const mode = model.thinking?.mode;
|
|
4552
4585
|
const effort = resolveAnthropicAdaptiveEffort(model, thinkingOptions);
|
|
4553
|
-
|
|
4554
|
-
if (mode === "anthropic-adaptive" && !compat.disableAdaptiveThinking) {
|
|
4586
|
+
if (!usesBudgetThinking(model)) {
|
|
4555
4587
|
const adaptive: { type: "adaptive"; display?: AnthropicThinkingDisplay } = { type: "adaptive" };
|
|
4556
4588
|
// Starting with Claude Opus 4.7 and Claude Fable/Mythos 5, adaptive thinking
|
|
4557
4589
|
// content is omitted from the response by default. Opt into summarized
|
|
@@ -4573,7 +4605,12 @@ function buildParams(
|
|
|
4573
4605
|
if (mode === "anthropic-budget-effort" && effort && effort !== "adaptive") outputConfigEffort = effort;
|
|
4574
4606
|
}
|
|
4575
4607
|
} else if (options?.thinkingEnabled === false) {
|
|
4576
|
-
if (
|
|
4608
|
+
if (model.compat.supportsBetweenToolsThinking) {
|
|
4609
|
+
// Sonnet 5.5 rejects `disabled` with a 400; `between_tools` is its lowest
|
|
4610
|
+
// thinking setting. It takes no other field and leaves effort untouched:
|
|
4611
|
+
// pinning `low` here would cap the whole turn's quality, not only thinking.
|
|
4612
|
+
thinking = { type: "between_tools" };
|
|
4613
|
+
} else if (isAdaptiveOnlyThinking(model)) {
|
|
4577
4614
|
// Adaptive-only Claude models (Opus 4.6+, Sonnet 4.6+, Fable/Mythos 5) reject
|
|
4578
4615
|
// `thinking.type: "disabled"` — adaptive thinking cannot be switched off.
|
|
4579
4616
|
// Omit the thinking field (the API defaults to adaptive) and pin the
|
|
@@ -4643,6 +4680,12 @@ function buildParams(
|
|
|
4643
4680
|
model.compat.supportsPerMessageEffort === true,
|
|
4644
4681
|
compactionReplay,
|
|
4645
4682
|
);
|
|
4683
|
+
// `between_tools` returns a 400 at `xhigh`/`max` effort, and the effort in
|
|
4684
|
+
// force from earlier turns outlives a thinking toggle. Fall back to the
|
|
4685
|
+
// default adaptive request, which accepts every effort level.
|
|
4686
|
+
if (thinking?.type === "between_tools" && (effortPlan.topLevel === "xhigh" || effortPlan.topLevel === "max")) {
|
|
4687
|
+
thinking = undefined;
|
|
4688
|
+
}
|
|
4646
4689
|
const wireMessages = convertAnthropicMessages(
|
|
4647
4690
|
insertAnthropicControlMarkers(context.messages, [...toolPlan.inserts, ...effortPlan.inserts]),
|
|
4648
4691
|
effectiveModel,
|
|
@@ -4683,7 +4726,7 @@ function buildParams(
|
|
|
4683
4726
|
|
|
4684
4727
|
// OAuth and API-key requests alike get the full model ceiling; Claude Code
|
|
4685
4728
|
// itself requests 128k on Opus 5.5.
|
|
4686
|
-
const maxOutputTokens = model
|
|
4729
|
+
const maxOutputTokens = anthropicOutputLimit(model);
|
|
4687
4730
|
|
|
4688
4731
|
// A caller-owned client targets its own endpoint: route body betas by the
|
|
4689
4732
|
// client's URL when it exposes one, not the model's routing. Otherwise the
|
|
@@ -24,6 +24,8 @@ const PRELUDE_CRC_LEN = 4;
|
|
|
24
24
|
const MESSAGE_CRC_LEN = 4;
|
|
25
25
|
const HEADER_BLOCK_OFFSET = PRELUDE_LEN + PRELUDE_CRC_LEN;
|
|
26
26
|
const MIN_MESSAGE_LEN = HEADER_BLOCK_OFFSET + MESSAGE_CRC_LEN;
|
|
27
|
+
/** Shared across messages: every header decode is a complete, non-streaming call. */
|
|
28
|
+
const HEADER_DECODER = new TextDecoder();
|
|
27
29
|
|
|
28
30
|
export interface EventStreamMessage {
|
|
29
31
|
/** Lower-cased copy is *not* applied — Bedrock uses casing like `:event-type` verbatim. */
|
|
@@ -64,12 +66,11 @@ export function decodeMessage(frame: Uint8Array): EventStreamMessage {
|
|
|
64
66
|
function parseHeaders(buf: Uint8Array): Record<string, string> {
|
|
65
67
|
const out: Record<string, string> = {};
|
|
66
68
|
const view = new DataView(buf.buffer, buf.byteOffset, buf.byteLength);
|
|
67
|
-
const decoder = new TextDecoder();
|
|
68
69
|
let p = 0;
|
|
69
70
|
while (p < buf.length) {
|
|
70
71
|
const nameLen = view.getUint8(p);
|
|
71
72
|
p += 1;
|
|
72
|
-
const name =
|
|
73
|
+
const name = HEADER_DECODER.decode(buf.subarray(p, p + nameLen));
|
|
73
74
|
p += nameLen;
|
|
74
75
|
const type = view.getUint8(p);
|
|
75
76
|
p += 1;
|
|
@@ -108,7 +109,7 @@ function parseHeaders(buf: Uint8Array): Record<string, string> {
|
|
|
108
109
|
// string
|
|
109
110
|
const len = view.getUint16(p, false);
|
|
110
111
|
p += 2;
|
|
111
|
-
out[name] =
|
|
112
|
+
out[name] = HEADER_DECODER.decode(buf.subarray(p, p + len));
|
|
112
113
|
p += len;
|
|
113
114
|
break;
|
|
114
115
|
}
|
|
@@ -61,7 +61,7 @@ const UNSIGNABLE: Record<string, true> = {
|
|
|
61
61
|
* `ArrayBuffer`, which is what `crypto.subtle.{digest,sign,importKey}` requires
|
|
62
62
|
* under the strict TS DOM typings. No-op when already strict.
|
|
63
63
|
*/
|
|
64
|
-
function asStrict(bytes: Uint8Array): Uint8Array<ArrayBuffer> {
|
|
64
|
+
export function asStrict(bytes: Uint8Array): Uint8Array<ArrayBuffer> {
|
|
65
65
|
if (bytes.buffer instanceof ArrayBuffer && bytes.byteOffset === 0 && bytes.byteLength === bytes.buffer.byteLength) {
|
|
66
66
|
return bytes as Uint8Array<ArrayBuffer>;
|
|
67
67
|
}
|
|
@@ -1,11 +1,10 @@
|
|
|
1
|
-
import { $env } from "@oh-my-pi/pi-utils";
|
|
1
|
+
import { $env, type ServerSentEvent } from "@oh-my-pi/pi-utils";
|
|
2
2
|
import * as AIError from "../error";
|
|
3
3
|
import { getEnvApiKey } from "../stream";
|
|
4
4
|
import type {
|
|
5
5
|
AssistantMessage,
|
|
6
6
|
Context,
|
|
7
7
|
Model,
|
|
8
|
-
RawSseEvent,
|
|
9
8
|
ServiceTier,
|
|
10
9
|
StreamFunction,
|
|
11
10
|
StreamOptions,
|
|
@@ -104,7 +103,7 @@ const streamAzureOpenAIResponsesOnce = (
|
|
|
104
103
|
const { requestAbortController, requestSignal } = abortTracker;
|
|
105
104
|
const onSseEvent = options?.onSseEvent;
|
|
106
105
|
const rawSseObserver = onSseEvent
|
|
107
|
-
? (event:
|
|
106
|
+
? (event: ServerSentEvent) => {
|
|
108
107
|
if (!event.event && event.data && event.data !== "[DONE]") {
|
|
109
108
|
try {
|
|
110
109
|
const parsed = JSON.parse(event.data);
|
|
@@ -120,7 +119,7 @@ const streamAzureOpenAIResponsesOnce = (
|
|
|
120
119
|
}
|
|
121
120
|
} catch {}
|
|
122
121
|
}
|
|
123
|
-
onSseEvent(event, model);
|
|
122
|
+
onSseEvent({ event: event.event, data: event.data, raw: [...event.raw] }, model);
|
|
124
123
|
}
|
|
125
124
|
: undefined;
|
|
126
125
|
|
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
import { isRecord } from "@oh-my-pi/pi-utils";
|
|
2
|
+
import { extractClaudeMetadataSessionId } from "./anthropic-identity";
|
|
3
|
+
import { isBedrockRequestMetadataValue } from "./bedrock-request-metadata";
|
|
4
|
+
|
|
5
|
+
/**
|
|
6
|
+
* Fit an Anthropic request body to Bedrock's Anthropic Messages API
|
|
7
|
+
* (`compat.bedrockMessagesApi`): both `/anthropic` routes reject the tool
|
|
8
|
+
* `strict` field, and bedrock-runtime rejects a `metadata.user_id` outside
|
|
9
|
+
* Bedrock's request-metadata pattern. A user id that fits is kept, otherwise
|
|
10
|
+
* its embedded session id, otherwise the metadata is dropped. Mutates and
|
|
11
|
+
* returns `payload`.
|
|
12
|
+
*/
|
|
13
|
+
export function fitBedrockAnthropicPayload<T>(payload: T): T {
|
|
14
|
+
if (!isRecord(payload)) return payload;
|
|
15
|
+
const body: Record<string, unknown> = payload;
|
|
16
|
+
if (Array.isArray(body.tools)) {
|
|
17
|
+
for (const tool of body.tools) {
|
|
18
|
+
if (isRecord(tool)) delete tool.strict;
|
|
19
|
+
}
|
|
20
|
+
}
|
|
21
|
+
if (body.metadata === undefined) return payload;
|
|
22
|
+
const userId = isRecord(body.metadata) ? body.metadata.user_id : undefined;
|
|
23
|
+
const fitted =
|
|
24
|
+
typeof userId === "string" && isBedrockRequestMetadataValue(userId)
|
|
25
|
+
? userId
|
|
26
|
+
: extractClaudeMetadataSessionId(userId);
|
|
27
|
+
if (fitted && isBedrockRequestMetadataValue(fitted)) body.metadata = { user_id: fitted };
|
|
28
|
+
else delete body.metadata;
|
|
29
|
+
return payload;
|
|
30
|
+
}
|
|
@@ -0,0 +1,6 @@
|
|
|
1
|
+
const BEDROCK_REQUEST_METADATA_PATTERN = /^[a-zA-Z0-9\s:_@$#=/+,\-.]*$/;
|
|
2
|
+
|
|
3
|
+
/** Check Bedrock's request-metadata character and length limits. Keys must also be nonempty. */
|
|
4
|
+
export function isBedrockRequestMetadataValue(value: string): boolean {
|
|
5
|
+
return value.length <= 256 && BEDROCK_REQUEST_METADATA_PATTERN.test(value);
|
|
6
|
+
}
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { truncate } from "@oh-my-pi/pi-utils";
|
|
1
|
+
import { isRecord, truncate } from "@oh-my-pi/pi-utils";
|
|
2
2
|
|
|
3
3
|
/**
|
|
4
4
|
* Connect-protocol end-stream error formatting.
|
|
@@ -21,10 +21,6 @@ const GENERIC_CONNECT_ERROR_MESSAGES = new Set(["", "error", "unknown", "unknown
|
|
|
21
21
|
/** Upper bound for appended trailer context so errors stay log-line sized. */
|
|
22
22
|
const MAX_EXTRA_DETAIL_CHARS = 400;
|
|
23
23
|
|
|
24
|
-
function isRecord(value: unknown): value is Record<string, unknown> {
|
|
25
|
-
return typeof value === "object" && value !== null && !Array.isArray(value);
|
|
26
|
-
}
|
|
27
|
-
|
|
28
24
|
function safeJson(value: unknown): string | undefined {
|
|
29
25
|
try {
|
|
30
26
|
const text = typeof value === "string" ? value : JSON.stringify(value);
|
|
@@ -32,7 +32,8 @@ type ProtoUnknownBag = { $unknown?: ProtoUnknownField[] };
|
|
|
32
32
|
type InteractionQueryCase = NonNullable<InteractionQuery["query"]["case"]>;
|
|
33
33
|
type InteractionResult = Exclude<InteractionResponse["result"], { case: undefined; value?: undefined }>;
|
|
34
34
|
|
|
35
|
-
|
|
35
|
+
/** Wrap one Connect-protocol message: 1 flag byte + 4-byte big-endian length + payload. */
|
|
36
|
+
export function frameConnectMessage(data: Uint8Array, flags = 0): Buffer {
|
|
36
37
|
const frame = Buffer.alloc(5 + data.length);
|
|
37
38
|
frame[0] = flags;
|
|
38
39
|
frame.writeUInt32BE(data.length, 1);
|
|
@@ -46,7 +47,8 @@ function isProtoUnknownField(value: unknown): value is ProtoUnknownField {
|
|
|
46
47
|
return typeof value.no === "number" && typeof value.wireType === "number" && value.data instanceof Uint8Array;
|
|
47
48
|
}
|
|
48
49
|
|
|
49
|
-
|
|
50
|
+
/** Well-formed protobuf-es `$unknown` entries on `message`; anything else on the bag is ignored. */
|
|
51
|
+
export function protoUnknownFields(message: object): ProtoUnknownField[] {
|
|
50
52
|
if (!("$unknown" in message) || !Array.isArray(message.$unknown)) return [];
|
|
51
53
|
return message.$unknown.filter(isProtoUnknownField);
|
|
52
54
|
}
|
package/src/providers/cursor.ts
CHANGED
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
import * as fs from "node:fs/promises";
|
|
2
2
|
import http2 from "node:http2";
|
|
3
|
+
import { cursorModelParameters } from "@oh-my-pi/pi-catalog/compat/behavior";
|
|
3
4
|
import { isCursorMaxModeWireId } from "@oh-my-pi/pi-catalog/compat/collapse";
|
|
4
5
|
import { classifyModel, collapseVariantId } from "@oh-my-pi/pi-catalog/compat/taxonomy";
|
|
5
6
|
import type {
|
|
@@ -246,7 +247,7 @@ import {
|
|
|
246
247
|
piTimeout,
|
|
247
248
|
shellTimeoutSeconds,
|
|
248
249
|
} from "./cursor/exec-modern";
|
|
249
|
-
import { handleInteractionQuery } from "./cursor/interaction-query";
|
|
250
|
+
import { frameConnectMessage, handleInteractionQuery, protoUnknownFields } from "./cursor/interaction-query";
|
|
250
251
|
|
|
251
252
|
export const CURSOR_API_URL = "https://api2.cursor.sh";
|
|
252
253
|
export const CURSOR_CLIENT_VERSION = "cli-2026.07.23-e383d2b";
|
|
@@ -407,14 +408,14 @@ function log(type: string, subtype?: string, data?: unknown): void {
|
|
|
407
408
|
void appendCursorDebugLog(entry);
|
|
408
409
|
}
|
|
409
410
|
|
|
410
|
-
|
|
411
|
-
|
|
412
|
-
|
|
413
|
-
|
|
414
|
-
|
|
415
|
-
|
|
411
|
+
/**
|
|
412
|
+
* Write one client message. Once the server's end frame has half-closed our
|
|
413
|
+
* side, late writes (heartbeats, exec replies from a handler still running)
|
|
414
|
+
* are dropped: writing after `end()` would error the stream.
|
|
415
|
+
*/
|
|
416
|
+
function writeClientMessage(h2Request: http2.ClientHttp2Stream, data: Uint8Array): void {
|
|
417
|
+
if (!h2Request.writableEnded) h2Request.write(frameConnectMessage(data));
|
|
416
418
|
}
|
|
417
|
-
|
|
418
419
|
class ConnectEndStreamError extends AIError.ProviderResponseError {
|
|
419
420
|
readonly diagnosticMessage: string;
|
|
420
421
|
|
|
@@ -845,6 +846,11 @@ function streamCursorWithWireMode(
|
|
|
845
846
|
if (endError) {
|
|
846
847
|
endStreamError = endError;
|
|
847
848
|
h2Request?.close();
|
|
849
|
+
} else {
|
|
850
|
+
// The end frame is the server's last message. Half-close our
|
|
851
|
+
// side so the stream can finish: a CONNECT proxy holds the
|
|
852
|
+
// HTTP/2 stream open until the client ends its request.
|
|
853
|
+
h2Request?.end();
|
|
848
854
|
}
|
|
849
855
|
continue;
|
|
850
856
|
}
|
|
@@ -903,7 +909,7 @@ function streamCursorWithWireMode(
|
|
|
903
909
|
message: { case: "clientHeartbeat", value: create(ClientHeartbeatSchema, {}) },
|
|
904
910
|
});
|
|
905
911
|
const heartbeatBytes = toBinary(AgentClientMessageSchema, heartbeatMessage);
|
|
906
|
-
h2Request
|
|
912
|
+
writeClientMessage(h2Request, heartbeatBytes);
|
|
907
913
|
};
|
|
908
914
|
|
|
909
915
|
const closeDebugLog = async (): Promise<void> => {
|
|
@@ -1207,8 +1213,6 @@ export async function handleServerMessage(
|
|
|
1207
1213
|
}
|
|
1208
1214
|
}
|
|
1209
1215
|
|
|
1210
|
-
type ProtoUnknownField = { no: number; wireType: number; data: Uint8Array };
|
|
1211
|
-
|
|
1212
1216
|
type HostedFetchCall = {
|
|
1213
1217
|
args?: { url?: string; toolCallId?: string };
|
|
1214
1218
|
result?: { result?: { case?: string; value?: { content?: string; error?: string; url?: string } } };
|
|
@@ -1250,11 +1254,6 @@ function describeHostedFetchResult(call: HostedFetchCall | undefined): { text: s
|
|
|
1250
1254
|
return { text: "Fetch completed", isError: false };
|
|
1251
1255
|
}
|
|
1252
1256
|
|
|
1253
|
-
function protoUnknownFields(message: object): ProtoUnknownField[] {
|
|
1254
|
-
const raw = (message as { $unknown?: ProtoUnknownField[] }).$unknown;
|
|
1255
|
-
return Array.isArray(raw) ? raw : [];
|
|
1256
|
-
}
|
|
1257
|
-
|
|
1258
1257
|
function handleKvServerMessage(
|
|
1259
1258
|
kvMsg: KvServerMessage,
|
|
1260
1259
|
blobStore: Map<string, Uint8Array>,
|
|
@@ -1281,7 +1280,7 @@ function handleKvServerMessage(
|
|
|
1281
1280
|
});
|
|
1282
1281
|
|
|
1283
1282
|
const responseBytes = toBinary(AgentClientMessageSchema, kvClientMessage);
|
|
1284
|
-
h2Request
|
|
1283
|
+
writeClientMessage(h2Request, responseBytes);
|
|
1285
1284
|
|
|
1286
1285
|
log("kvClient", "getBlobResult", { blobId: blobIdKey.slice(0, 40) });
|
|
1287
1286
|
} else if (kvCase === "setBlobArgs") {
|
|
@@ -1302,7 +1301,7 @@ function handleKvServerMessage(
|
|
|
1302
1301
|
});
|
|
1303
1302
|
|
|
1304
1303
|
const responseBytes = toBinary(AgentClientMessageSchema, kvClientMessage);
|
|
1305
|
-
h2Request
|
|
1304
|
+
writeClientMessage(h2Request, responseBytes);
|
|
1306
1305
|
|
|
1307
1306
|
log("kvClient", "setBlobResult", { blobId: blobIdKey.slice(0, 40) });
|
|
1308
1307
|
}
|
|
@@ -2604,7 +2603,7 @@ function sendExecClientMessage<TCase extends NonNullable<ExecClientMessage["mess
|
|
|
2604
2603
|
});
|
|
2605
2604
|
|
|
2606
2605
|
const responseBytes = toBinary(AgentClientMessageSchema, clientMessage);
|
|
2607
|
-
h2Request
|
|
2606
|
+
writeClientMessage(h2Request, responseBytes);
|
|
2608
2607
|
|
|
2609
2608
|
log("execClientMessage", messageCase, value);
|
|
2610
2609
|
}
|
|
@@ -2640,7 +2639,7 @@ function sendExecClientThrow(
|
|
|
2640
2639
|
const clientMessage = create(AgentClientMessageSchema, {
|
|
2641
2640
|
message: { case: "execClientControlMessage", value: controlMessage },
|
|
2642
2641
|
});
|
|
2643
|
-
h2Request
|
|
2642
|
+
writeClientMessage(h2Request, toBinary(AgentClientMessageSchema, clientMessage));
|
|
2644
2643
|
log("execClientControl", "throw", { id: execMsg.id, execId: execMsg.execId, error, errorCode });
|
|
2645
2644
|
sendExecClientStreamClose(h2Request, execMsg);
|
|
2646
2645
|
}
|
|
@@ -2658,7 +2657,7 @@ function sendExecClientStreamClose(h2Request: http2.ClientHttp2Stream, execMsg:
|
|
|
2658
2657
|
message: { case: "execClientControlMessage", value: closeMessage },
|
|
2659
2658
|
});
|
|
2660
2659
|
const responseBytes = toBinary(AgentClientMessageSchema, clientMessage);
|
|
2661
|
-
h2Request
|
|
2660
|
+
writeClientMessage(h2Request, responseBytes);
|
|
2662
2661
|
log("execClientControl", "streamClose", { id: execMsg.id, execId: execMsg.execId });
|
|
2663
2662
|
}
|
|
2664
2663
|
|
|
@@ -5487,13 +5486,18 @@ function resolveCursorWireModel(
|
|
|
5487
5486
|
};
|
|
5488
5487
|
}
|
|
5489
5488
|
}
|
|
5490
|
-
//
|
|
5491
|
-
//
|
|
5492
|
-
//
|
|
5493
|
-
|
|
5489
|
+
// Fixed per-model parameters come from catalog KDL (`cursor-model-parameter`
|
|
5490
|
+
// in `runtime/behavior.kdl`). A bare `composer-2.5` id resolves to the Fast
|
|
5491
|
+
// variant server-side (can1357/oh-my-pi#9012), so the catalog pins the
|
|
5492
|
+
// Standard tier with `fast=false`; `-fast` selections keep the Fast lane by
|
|
5493
|
+
// declaring no parameter.
|
|
5494
|
+
const fixedParameters = cursorModelParameters(wireModelId);
|
|
5495
|
+
if (fixedParameters.length > 0) {
|
|
5494
5496
|
return {
|
|
5495
5497
|
modelId: wireModelId,
|
|
5496
|
-
parameters:
|
|
5498
|
+
parameters: fixedParameters.map(({ id, value }) =>
|
|
5499
|
+
create(RequestedModel_ModelParameterbytesSchema, { id, value }),
|
|
5500
|
+
),
|
|
5497
5501
|
maxMode,
|
|
5498
5502
|
};
|
|
5499
5503
|
}
|
|
@@ -751,9 +751,16 @@ export const streamGoogleGeminiCli: StreamFunction<"google-gemini-cli"> = (
|
|
|
751
751
|
const responseSignal = options?.signal
|
|
752
752
|
? AbortSignal.any([options.signal, responseAbortController.signal])
|
|
753
753
|
: responseAbortController.signal;
|
|
754
|
+
const onSseEvent = options?.onSseEvent;
|
|
754
755
|
const chunks = iterateWithIdleTimeout(
|
|
755
|
-
|
|
756
|
-
|
|
756
|
+
// Attach the observer only when a diagnostic listener exists: any
|
|
757
|
+
// observer turns on per-line raw capture in `readSseJson`.
|
|
758
|
+
readSseJson<CloudCodeAssistResponseChunk>(
|
|
759
|
+
activeResponse.body,
|
|
760
|
+
responseSignal,
|
|
761
|
+
onSseEvent
|
|
762
|
+
? event => onSseEvent({ event: event.event, data: event.data, raw: [...event.raw] }, model)
|
|
763
|
+
: undefined,
|
|
757
764
|
),
|
|
758
765
|
{
|
|
759
766
|
firstItemTimeoutMs: firstEventTimeoutMs,
|
|
@@ -4,7 +4,7 @@
|
|
|
4
4
|
|
|
5
5
|
import { scheduler } from "node:timers/promises";
|
|
6
6
|
import { calculateCost } from "@oh-my-pi/pi-catalog/models";
|
|
7
|
-
import { readSseJson } from "@oh-my-pi/pi-utils";
|
|
7
|
+
import { readSseJson, type SseEventObserver } from "@oh-my-pi/pi-utils";
|
|
8
8
|
import { renderDemotedThinking } from "../dialect/demotion";
|
|
9
9
|
import { ThinkingFenceStripper } from "../dialect/thinking-fence-strip";
|
|
10
10
|
import * as AIError from "../error";
|
|
@@ -828,8 +828,14 @@ export function buildGoogleGenerateContentParams<T extends "google-generative-ai
|
|
|
828
828
|
// Vertex AI ignores a body field and requires the
|
|
829
829
|
// `X-Vertex-AI-LLM-Shared-Request-Type` header instead (added in
|
|
830
830
|
// streamGoogleVertex), so only emit the body field for the direct API.
|
|
831
|
-
|
|
832
|
-
|
|
831
|
+
const serviceTier = options.serviceTier;
|
|
832
|
+
// `!== "ultrafast"` narrows to the Gemini wire type; `shouldSendServiceTier` already rejects it for Google.
|
|
833
|
+
if (
|
|
834
|
+
model.provider === "google" &&
|
|
835
|
+
serviceTier !== "ultrafast" &&
|
|
836
|
+
shouldSendServiceTier(serviceTier, model.provider)
|
|
837
|
+
) {
|
|
838
|
+
config.serviceTier = serviceTier;
|
|
833
839
|
}
|
|
834
840
|
|
|
835
841
|
if (context.tools && context.tools.length > 0 && options.toolChoice) {
|
|
@@ -1014,13 +1020,17 @@ export function streamGoogleGenAI<T extends "google-generative-ai" | "google-ver
|
|
|
1014
1020
|
let body = await openStream();
|
|
1015
1021
|
stream.push({ type: "start", partial: output });
|
|
1016
1022
|
|
|
1023
|
+
// Attach the observer only when a diagnostic listener exists: any
|
|
1024
|
+
// observer turns on per-line raw capture in `readSseJson`.
|
|
1025
|
+
const onSseEvent = options?.onSseEvent;
|
|
1026
|
+
const sseObserver: SseEventObserver | undefined = onSseEvent
|
|
1027
|
+
? event => onSseEvent({ event: event.event, data: event.data, raw: [...event.raw] }, model)
|
|
1028
|
+
: undefined;
|
|
1017
1029
|
// Gemini occasionally finishes with `finishReason: STOP` while emitting only an empty
|
|
1018
1030
|
// text part and no tool call. Delivered as-is the agent receives a blank message and
|
|
1019
1031
|
// silently halts mid-task, so retry a bounded number of times before giving up.
|
|
1020
1032
|
for (let emptyAttempt = 0; ; emptyAttempt++) {
|
|
1021
|
-
const googleStream = readSseJson<GenerateContentResponse>(body, options?.signal,
|
|
1022
|
-
options?.onSseEvent?.({ event: event.event, data: event.data, raw: [...event.raw] }, model),
|
|
1023
|
-
);
|
|
1033
|
+
const googleStream = readSseJson<GenerateContentResponse>(body, options?.signal, sseObserver);
|
|
1024
1034
|
await consumeGoogleStream({
|
|
1025
1035
|
googleStream,
|
|
1026
1036
|
output,
|
|
@@ -551,7 +551,7 @@ export interface ChatCompletionChunk {
|
|
|
551
551
|
/** Moderation results, present on the moderation chunk when requested. */
|
|
552
552
|
moderation?: ChatCompletionChunkModeration | null;
|
|
553
553
|
/** Processing type actually used for serving the request. */
|
|
554
|
-
service_tier?: "auto" | "default" | "flex" | "scale" | "priority" | null;
|
|
554
|
+
service_tier?: "auto" | "default" | "flex" | "scale" | "priority" | "ultrafast" | null;
|
|
555
555
|
/** Deprecated by OpenAI: backend configuration fingerprint, pairs with `seed`. */
|
|
556
556
|
system_fingerprint?: string;
|
|
557
557
|
/** Only with `stream_options: {"include_usage": true}`; null except on the last chunk. */
|
|
@@ -819,7 +819,7 @@ export interface ChatCompletionCreateParamsBase {
|
|
|
819
819
|
/** Deprecated by OpenAI (Beta): best-effort deterministic sampling seed. */
|
|
820
820
|
seed?: number | null;
|
|
821
821
|
/** Processing type used for serving the request. */
|
|
822
|
-
service_tier?: "auto" | "default" | "flex" | "scale" | "priority" | null;
|
|
822
|
+
service_tier?: "auto" | "default" | "flex" | "scale" | "priority" | "ultrafast" | null;
|
|
823
823
|
/** Up to 4 sequences where the API will stop generating further tokens. */
|
|
824
824
|
stop?: string | null | Array<string>;
|
|
825
825
|
/** Whether to store the output for model distillation or evals. */
|
|
@@ -86,7 +86,7 @@ export interface RequestBody {
|
|
|
86
86
|
client_metadata?: Record<string, string>;
|
|
87
87
|
max_output_tokens?: number;
|
|
88
88
|
max_completion_tokens?: number;
|
|
89
|
-
service_tier?: "auto" | "default" | "flex" | "scale" | "priority" | null;
|
|
89
|
+
service_tier?: "auto" | "default" | "flex" | "scale" | "priority" | "ultrafast" | null;
|
|
90
90
|
/** Explicit cyber access program for this request; see `openai-codex/access-programs.ts`. */
|
|
91
91
|
access_programs?: { cyber: string };
|
|
92
92
|
[key: string]: unknown;
|
|
@@ -1325,28 +1325,26 @@ function getCodexServiceTierCostMultiplier(
|
|
|
1325
1325
|
model: Pick<Model<"openai-codex-responses">, "serviceTierCost">,
|
|
1326
1326
|
serviceTier: ServiceTier | "default" | undefined,
|
|
1327
1327
|
): number {
|
|
1328
|
+
// `ultrafast` has no published price (API preview, Codex credits), so it is
|
|
1329
|
+
// shown at 1x rather than an invented multiplier.
|
|
1328
1330
|
if (serviceTier !== "flex" && serviceTier !== "priority") return 1;
|
|
1329
1331
|
return model.serviceTierCost?.[serviceTier] ?? 1;
|
|
1330
1332
|
}
|
|
1331
1333
|
|
|
1332
|
-
|
|
1333
|
-
|
|
1334
|
-
|
|
1335
|
-
|
|
1336
|
-
|
|
1337
|
-
|
|
1338
|
-
|
|
1339
|
-
|
|
1340
|
-
return req;
|
|
1341
|
-
}
|
|
1342
|
-
return "default";
|
|
1343
|
-
}
|
|
1334
|
+
/**
|
|
1335
|
+
* The tier a Codex response was billed at. The response echo is authoritative
|
|
1336
|
+
* whenever it reports a tier (the backend may serve a requested priority/flex
|
|
1337
|
+
* turn as `default`); the requested tier is used only when the echo is absent.
|
|
1338
|
+
*/
|
|
1339
|
+
function resolveCodexCostServiceTier(res: ServiceTier | undefined, req?: unknown): ServiceTier | "default" | undefined {
|
|
1340
|
+
const served = res ?? req;
|
|
1341
|
+
return served === "flex" || served === "priority" ? served : "default";
|
|
1344
1342
|
}
|
|
1345
1343
|
|
|
1346
1344
|
function applyCodexServiceTierPricing(
|
|
1347
1345
|
model: Pick<Model<"openai-codex-responses">, "serviceTierCost">,
|
|
1348
1346
|
usage: AssistantMessage["usage"],
|
|
1349
|
-
resTier:
|
|
1347
|
+
resTier: ServiceTier | undefined,
|
|
1350
1348
|
reqTier: unknown,
|
|
1351
1349
|
): void {
|
|
1352
1350
|
const resolvedTier = resolveCodexCostServiceTier(resTier, reqTier);
|
|
@@ -3530,6 +3528,7 @@ function parseCodexServiceTier(value: unknown): ServiceTier | undefined {
|
|
|
3530
3528
|
case "flex":
|
|
3531
3529
|
case "scale":
|
|
3532
3530
|
case "priority":
|
|
3531
|
+
case "ultrafast":
|
|
3533
3532
|
return value;
|
|
3534
3533
|
default:
|
|
3535
3534
|
return undefined;
|
|
@@ -3749,12 +3748,18 @@ const CODEX_CHAIN_TOP_LEVEL_EXCLUDE_MAP = {
|
|
|
3749
3748
|
* request schema has no `previous_response_id` (codex-rs carries it only on
|
|
3750
3749
|
* websocket `response.create` frames) and strict gateway validators 400 it
|
|
3751
3750
|
* with `{"detail":"Unsupported parameter: previous_response_id"}`.
|
|
3751
|
+
*
|
|
3752
|
+
* Entering or leaving `ultrafast` still breaks the chain: that tier is a
|
|
3753
|
+
* separate serving path, and codex-rs sends a full `response.create` across
|
|
3754
|
+
* such a switch rather than a `previous_response_id` delta.
|
|
3752
3755
|
*/
|
|
3753
3756
|
function buildCodexChainedRequestBody(
|
|
3754
3757
|
requestBody: RequestBody,
|
|
3755
3758
|
state: CodexWebSocketSessionState | undefined,
|
|
3756
3759
|
): RequestBody {
|
|
3757
|
-
const chainable =
|
|
3760
|
+
const chainable =
|
|
3761
|
+
state?.canAppend === true &&
|
|
3762
|
+
(state.lastRequest?.service_tier === "ultrafast") === (requestBody.service_tier === "ultrafast");
|
|
3758
3763
|
const appendInput = chainable
|
|
3759
3764
|
? buildResponsesDeltaInput(
|
|
3760
3765
|
state.lastRequest,
|
|
@@ -4730,8 +4735,14 @@ async function openCodexSseEventStream(
|
|
|
4730
4735
|
if (!response.body) {
|
|
4731
4736
|
throw new CodexProviderStreamError("No response body", false);
|
|
4732
4737
|
}
|
|
4733
|
-
|
|
4734
|
-
|
|
4738
|
+
// Attach the observer only when a diagnostic listener exists: any observer
|
|
4739
|
+
// turns on per-line raw capture in `readSseJson`.
|
|
4740
|
+
return readSseJson<Record<string, unknown>>(
|
|
4741
|
+
response.body,
|
|
4742
|
+
signal,
|
|
4743
|
+
onSseEvent
|
|
4744
|
+
? event => onSseEvent({ event: event.event, data: event.data, raw: [...event.raw] }, undefined)
|
|
4745
|
+
: undefined,
|
|
4735
4746
|
);
|
|
4736
4747
|
}
|
|
4737
4748
|
|