@oh-my-pi/pi-ai 18.4.3 → 18.4.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +20 -17
- package/dist/types/auth-broker/protocol.d.ts +12 -0
- package/dist/types/dialect/rendering.d.ts +4 -0
- package/dist/types/images/shared.d.ts +5 -2
- package/dist/types/providers/anthropic-wire.d.ts +9 -1
- package/dist/types/providers/anthropic.d.ts +17 -0
- package/dist/types/providers/aws-sigv4.d.ts +5 -0
- package/dist/types/providers/bedrock-anthropic.d.ts +9 -0
- package/dist/types/providers/bedrock-request-metadata.d.ts +2 -0
- package/dist/types/providers/cursor/interaction-query.d.ts +10 -0
- package/dist/types/providers/openai-chat-wire.d.ts +2 -2
- package/dist/types/providers/openai-codex/request-transformer.d.ts +1 -1
- package/dist/types/providers/openai-responses-wire.d.ts +2 -2
- package/dist/types/providers/xai-base-url.d.ts +17 -0
- package/dist/types/types.d.ts +14 -3
- package/dist/types/usage/shared.d.ts +13 -1
- package/package.json +6 -6
- package/src/auth-broker/client.ts +1 -12
- package/src/auth-broker/protocol.ts +32 -0
- package/src/auth-broker/remote-store.ts +4 -40
- package/src/auth-broker/server.ts +1 -20
- package/src/auth-broker/snapshot-cache.ts +1 -9
- package/src/dialect/anthropic.ts +3 -25
- package/src/dialect/minimax.ts +3 -24
- package/src/dialect/rendering.ts +18 -0
- package/src/dialect/xml.ts +3 -19
- package/src/images/openai-images.ts +10 -4
- package/src/images/shared.ts +9 -4
- package/src/providers/amazon-bedrock.ts +3 -6
- package/src/providers/anthropic-compaction.ts +10 -1
- package/src/providers/anthropic-wire.ts +12 -1
- package/src/providers/anthropic.ts +58 -15
- package/src/providers/aws-sigv4.ts +1 -1
- package/src/providers/bedrock-anthropic.ts +30 -0
- package/src/providers/bedrock-request-metadata.ts +6 -0
- package/src/providers/connect-error-detail.ts +1 -5
- package/src/providers/cursor/interaction-query.ts +4 -2
- package/src/providers/cursor.ts +30 -26
- package/src/providers/google-shared.ts +8 -2
- package/src/providers/openai-chat-wire.ts +2 -2
- package/src/providers/openai-codex/request-transformer.ts +1 -1
- package/src/providers/openai-codex-responses.ts +19 -14
- package/src/providers/openai-responses-wire.ts +2 -2
- package/src/providers/openai-shared.ts +4 -0
- package/src/providers/xai-base-url.ts +32 -0
- package/src/types.ts +35 -4
- package/src/usage/claude.ts +4 -11
- package/src/usage/cline-pass.ts +2 -14
- package/src/usage/cursor.ts +9 -1
- package/src/usage/openai-codex.ts +3 -5
- package/src/usage/shared.ts +28 -1
- package/src/usage/synthetic.ts +4 -40
- package/src/usage/umans.ts +8 -36
- package/src/usage/zai.ts +10 -38
- package/src/utils/http-inspector.ts +4 -8
- package/src/utils/schema/json-schema-validator.ts +23 -26
- package/src/utils/schema/meta-validator.ts +4 -7
- package/src/utils/schema/wire.ts +17 -21
package/CHANGELOG.md
CHANGED
|
@@ -2,6 +2,25 @@
|
|
|
2
2
|
|
|
3
3
|
## [Unreleased]
|
|
4
4
|
|
|
5
|
+
## [18.4.4] - 2026-09-29
|
|
6
|
+
|
|
7
|
+
### Added
|
|
8
|
+
|
|
9
|
+
- Added the `ultrafast` service tier. It is sent to the OpenAI API as-is, and to Codex only for models that list it in their discovered service tiers; other providers never receive it. On Codex websockets, switching into or out of `ultrafast` starts a new response chain instead of reusing `previous_response_id`, matching the Codex CLI. Ultrafast turns are costed at standard rates because no Ultrafast price is published yet ([#13782](https://github.com/can1357/oh-my-pi/pull/13782) by [@H4vC](https://github.com/H4vC)).
|
|
10
|
+
|
|
11
|
+
### Changed
|
|
12
|
+
|
|
13
|
+
- Changed to fall back to adaptive thinking when between_tools is used with xhigh effort
|
|
14
|
+
- xAI requests (`xai`, `xai-oauth` chat and image generation) honor `XAI_BASE_URL` again when the model uses the bundled `https://api.x.ai/v1` endpoint; a custom `baseUrl` from models.yml still wins, and `xai-oauth` OAuth access tokens always stay on the bundled endpoint.
|
|
15
|
+
|
|
16
|
+
### Fixed
|
|
17
|
+
|
|
18
|
+
- Fixed Claude on Amazon Bedrock's Anthropic Messages routes (`/anthropic` on bedrock-runtime and bedrock-mantle): runtime requests no longer fail with a request-metadata 400, and both routes use Anthropic's on-demand compaction ([#13311](https://github.com/can1357/oh-my-pi/pull/13311) by [@mustafaabidali](https://github.com/mustafaabidali)).
|
|
19
|
+
- `/usage` no longer shows an always-empty `gpt-4 requests` row for Cursor accounts on usage-based plans; the Cursor Models and Other Models meters remain ([#13726](https://github.com/can1357/oh-my-pi/pull/13726) by [@will-bogusz](https://github.com/will-bogusz)).
|
|
20
|
+
- Cursor turns routed through an HTTP proxy now finish instead of hanging after the response completes ([#13724](https://github.com/can1357/oh-my-pi/pull/13724) by [@will-bogusz](https://github.com/will-bogusz)).
|
|
21
|
+
- Fixed Codex requests sending `priority` (and `scale`) to models whose discovered service tiers list other tiers but not that one, matching the Codex CLI; an empty or missing list is treated as not reported, so `priority` is still sent and `/fast` keeps working on accounts whose `/models` lists no tiers (`flex` is always allowed) ([#13782](https://github.com/can1357/oh-my-pi/pull/13782) by [@H4vC](https://github.com/H4vC)).
|
|
22
|
+
- Fixed Codex priority cost: a turn the backend reports as served at `default` is no longer billed at the priority multiplier ([#13782](https://github.com/can1357/oh-my-pi/pull/13782) by [@H4vC](https://github.com/H4vC)).
|
|
23
|
+
|
|
5
24
|
## [18.4.3] - 2026-09-28
|
|
6
25
|
|
|
7
26
|
### Added
|
|
@@ -2300,20 +2319,4 @@
|
|
|
2300
2319
|
- Fixed the platform OpenAI Responses and Codex websocket stale-chain classifiers missing the "Unsupported parameter: previous_response_id" rejection phrasing (FastAPI-style `detail` body with no `error.code`), so a chained turn now falls back to a full-transcript replay instead of surfacing the 400
|
|
2301
2320
|
- Fixed the HTTP-400 raw-request dump for Codex SSE to record the body actually sent on the wire instead of the pre-transport request body, which made chained-request failures look like the rejected parameter was never sent
|
|
2302
2321
|
|
|
2303
|
-
|
|
2304
|
-
|
|
2305
|
-
### Added
|
|
2306
|
-
|
|
2307
|
-
- Added `requestModelId` and `thinking.suppress` options to `google-gemini-cli` so collapsed effort-tier variants serialize their per-effort upstream wire id, and thinking-off requests on models with `thinking.suppressWhenOff` send an explicit `thinkingConfig` (`includeThoughts: false` with `thinkingLevel: "MINIMAL"` or `thinkingBudget: 0`) — Cloud Code Assist re-applies the per-id baked server default when the config is omitted, silently thinking and billing the tokens
|
|
2308
|
-
- Added mandatory-reasoning clamping: models baked with `thinking.requiresEffort` floor omitted or disabled reasoning to the lowest supported effort in every api mapping, and `disableReasoning` no longer emits OpenRouter `reasoning: { enabled: false }` for them — fixes `omp bench` and utility requests 400ing with "Reasoning is mandatory for this endpoint and cannot be disabled" on OpenRouter Gemini 3.x
|
|
2309
|
-
|
|
2310
|
-
### Changed
|
|
2311
|
-
|
|
2312
|
-
- Changed `google-gemini-cli` request mapping to route per-request wire ids via `resolveWireModelId`: the session effort picks the backing variant id (collapsed `gemini-3.5-flash` at high → `gemini-3.5-flash-low`; claude pairs route off → bare id, efforts → `-thinking`) while `AssistantMessage.model` and usage attribution stay on the logical id. A thinking budget clamped to zero now falls through to the thinking-off path (off routing plus suppression) instead of only disabling thinking
|
|
2313
|
-
- Changed `openai-completions` and `anthropic-messages` to serialize per-request wire ids via `resolveWireModelId`, so collapsed `X`/`X-thinking` pairs on aggregators and custom providers switch to the thinking SKU when reasoning is enabled (previously only `google-gemini-cli` routed effort-tier variants)
|
|
2314
|
-
|
|
2315
|
-
### Fixed
|
|
2316
|
-
|
|
2317
|
-
- Fixed `google-gemini-cli` ignoring `Model.requestModelId` when serializing the request model id
|
|
2318
|
-
|
|
2319
|
-
Older entries are archived in [packages/ai/CHANGELOG.md@689a3418cb45](https://github.com/can1357/oh-my-pi/blob/689a3418cb45d54a459cde2e1abf3f66f50e47a4/packages/ai/CHANGELOG.md).
|
|
2322
|
+
Older entries are archived in [packages\ai\CHANGELOG.md@07e9197a3012](https://github.com/can1357/oh-my-pi/blob/07e9197a3012f58c459f1faabeb324decc21f41d/packages\ai\CHANGELOG.md).
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Auth-broker wire-protocol helpers shared by the server and the client store.
|
|
3
|
+
*/
|
|
4
|
+
import type { CredentialBlockSnapshot } from "./types.js";
|
|
5
|
+
/** Parse a snapshot `ETag` / `If-None-Match` value (`"N"`, `W/"N"`, or bare `N`) into a generation. */
|
|
6
|
+
export declare function parseGenerationTag(header: string | null): number | undefined;
|
|
7
|
+
/**
|
|
8
|
+
* Canonical order for a credential's block snapshots. The server and the client
|
|
9
|
+
* store both sort with it, because block lists are compared positionally; the
|
|
10
|
+
* `updatedAtMs` tiebreak keeps otherwise-identical blocks in one stable order.
|
|
11
|
+
*/
|
|
12
|
+
export declare function compareCredentialBlockSnapshots(a: CredentialBlockSnapshot, b: CredentialBlockSnapshot): number;
|
|
@@ -1,4 +1,5 @@
|
|
|
1
1
|
import type { AssistantMessage, Message, ToolCall } from "../types.js";
|
|
2
|
+
import { type ToolArgShape } from "./coercion.js";
|
|
2
3
|
import type { DialectRenderOptions, DialectToolResult } from "./types.js";
|
|
3
4
|
export declare function renderToolResponseResults(results: readonly DialectToolResult[]): string;
|
|
4
5
|
export declare function kimiCallId(name: string, id: string, index: number): string;
|
|
@@ -15,6 +16,9 @@ export declare function pyCall(name: string, args: Record<string, unknown>): str
|
|
|
15
16
|
export declare function pyValue(value: unknown): string;
|
|
16
17
|
export declare function escapeXmlAttr(value: string): string;
|
|
17
18
|
export declare function escapeXmlText(value: string): string;
|
|
19
|
+
/** Render one Anthropic-style `<invoke>`; declared string args stay raw, everything else is JSON. */
|
|
20
|
+
export declare function renderInvoke(call: ToolCall, shape: ToolArgShape | undefined): string;
|
|
21
|
+
export declare function renderInvokes(calls: readonly ToolCall[], tools: NonNullable<DialectRenderOptions["tools"]>): string;
|
|
18
22
|
export type AssistantTranscriptParts = {
|
|
19
23
|
readonly text: string;
|
|
20
24
|
readonly thinking: string;
|
|
@@ -10,9 +10,11 @@ export declare function usageFromWire(value: unknown): Usage;
|
|
|
10
10
|
export declare function imageBaseUrl(model: Model): string;
|
|
11
11
|
export declare function modelHeaders(model: Model, signal?: AbortSignal): Promise<Record<string, string>>;
|
|
12
12
|
export declare function errorMessage(rawText: string): string;
|
|
13
|
+
/** Request URL, or a builder for routes that depend on the bearer (xAI's `XAI_BASE_URL` rule). */
|
|
14
|
+
type ImageRequestUrl = string | ((bearer: string) => string);
|
|
13
15
|
export declare function postJson(options: {
|
|
14
16
|
model: Model;
|
|
15
|
-
url:
|
|
17
|
+
url: ImageRequestUrl;
|
|
16
18
|
body: unknown;
|
|
17
19
|
apiKey: ApiKey;
|
|
18
20
|
fetch: FetchImpl;
|
|
@@ -20,7 +22,7 @@ export declare function postJson(options: {
|
|
|
20
22
|
}): Promise<unknown>;
|
|
21
23
|
export declare function postMultipart(options: {
|
|
22
24
|
model: Model;
|
|
23
|
-
url:
|
|
25
|
+
url: ImageRequestUrl;
|
|
24
26
|
body: FormData;
|
|
25
27
|
apiKey: ApiKey;
|
|
26
28
|
fetch: FetchImpl;
|
|
@@ -32,3 +34,4 @@ export declare function decodeImageResponse(value: unknown, fetch: FetchImpl, si
|
|
|
32
34
|
}>;
|
|
33
35
|
export declare function toDataUrl(image: GeneratedImage): string;
|
|
34
36
|
export declare function resolveOpenAIImageSize(aspectRatio?: string, imageSize?: string): string | undefined;
|
|
37
|
+
export {};
|
|
@@ -225,7 +225,15 @@ export type ThinkingConfigAdaptive = {
|
|
|
225
225
|
/** Preserved-thinking prefix mismatch policy. */
|
|
226
226
|
block_binding?: ThinkingBlockBinding;
|
|
227
227
|
};
|
|
228
|
-
|
|
228
|
+
/**
|
|
229
|
+
* Sonnet 5.5's replacement for `disabled`: no up-front thinking, progress
|
|
230
|
+
* updates between tool calls only. Takes no other field, and effort above
|
|
231
|
+
* `high` is rejected alongside it.
|
|
232
|
+
*/
|
|
233
|
+
export type ThinkingConfigBetweenTools = {
|
|
234
|
+
type: "between_tools";
|
|
235
|
+
};
|
|
236
|
+
export type ThinkingConfigParam = ThinkingConfigEnabled | ThinkingConfigDisabled | ThinkingConfigAdaptive | ThinkingConfigBetweenTools;
|
|
229
237
|
export type OutputConfig = {
|
|
230
238
|
/** Adaptive-thinking effort level (effort beta). */
|
|
231
239
|
effort?: "low" | "medium" | "high" | "xhigh" | "max" | null;
|
|
@@ -217,6 +217,23 @@ type SystemBlockOptions = {
|
|
|
217
217
|
export declare function buildAnthropicSystemBlocks(systemPrompt: readonly string[] | undefined, options?: SystemBlockOptions): AnthropicSystemBlock[] | undefined;
|
|
218
218
|
export declare function normalizeExtraBetas(betas?: string[] | string): string[];
|
|
219
219
|
export declare function buildAnthropicClientOptions(args: AnthropicClientOptionsArgs): AnthropicClientOptionsResult;
|
|
220
|
+
/**
|
|
221
|
+
* True when enabled thinking on `model` is budget thinking
|
|
222
|
+
* (`thinking.type: "enabled"` with `budget_tokens`) rather than adaptive.
|
|
223
|
+
*/
|
|
224
|
+
export declare function usesBudgetThinking(model: Model<"anthropic-messages">): boolean;
|
|
225
|
+
/** The most output tokens a request to `model` may ask for (`max_tokens` ceiling). */
|
|
226
|
+
export declare function anthropicOutputLimit(model: Model<"anthropic-messages">): number;
|
|
227
|
+
/**
|
|
228
|
+
* The `max_tokens` and thinking budget of budget thinking: `max_tokens`
|
|
229
|
+
* rises to leave {@link OUTPUT_FALLBACK_BUFFER} visible output tokens after
|
|
230
|
+
* the budget, within `maxAllowedTokens`, and the budget shrinks when that
|
|
231
|
+
* ceiling leaves less (a non-positive budget means the ceiling is too low).
|
|
232
|
+
*/
|
|
233
|
+
export declare function budgetThinkingOutput(maxTokens: number | undefined, budgetTokens: number, maxAllowedTokens: number): {
|
|
234
|
+
maxTokens: number;
|
|
235
|
+
budgetTokens: number;
|
|
236
|
+
};
|
|
220
237
|
/**
|
|
221
238
|
* A single Anthropic conversation turn, including the mid-conversation
|
|
222
239
|
* `system` role (Opus 4.8+ and Fable/Mythos 5).
|
|
@@ -33,6 +33,11 @@ export interface SignParams {
|
|
|
33
33
|
/** Override clock for deterministic tests. */
|
|
34
34
|
date?: Date;
|
|
35
35
|
}
|
|
36
|
+
/** Coerce a possibly-ArrayBufferLike-backed `Uint8Array` into one over a fresh
|
|
37
|
+
* `ArrayBuffer`, which is what `crypto.subtle.{digest,sign,importKey}` requires
|
|
38
|
+
* under the strict TS DOM typings. No-op when already strict.
|
|
39
|
+
*/
|
|
40
|
+
export declare function asStrict(bytes: Uint8Array): Uint8Array<ArrayBuffer>;
|
|
36
41
|
export declare function toHex(bytes: Uint8Array): string;
|
|
37
42
|
export declare function sha256(data: Uint8Array | string): Promise<Uint8Array>;
|
|
38
43
|
export declare function sha256Hex(data: Uint8Array | string): Promise<string>;
|
|
@@ -0,0 +1,9 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Fit an Anthropic request body to Bedrock's Anthropic Messages API
|
|
3
|
+
* (`compat.bedrockMessagesApi`): both `/anthropic` routes reject the tool
|
|
4
|
+
* `strict` field, and bedrock-runtime rejects a `metadata.user_id` outside
|
|
5
|
+
* Bedrock's request-metadata pattern. A user id that fits is kept, otherwise
|
|
6
|
+
* its embedded session id, otherwise the metadata is dropped. Mutates and
|
|
7
|
+
* returns `payload`.
|
|
8
|
+
*/
|
|
9
|
+
export declare function fitBedrockAnthropicPayload<T>(payload: T): T;
|
|
@@ -1,5 +1,14 @@
|
|
|
1
1
|
import type http2 from "node:http2";
|
|
2
2
|
import { type InteractionQuery } from "@oh-my-pi/pi-catalog/discovery/cursor-proto";
|
|
3
|
+
type ProtoUnknownField = {
|
|
4
|
+
no: number;
|
|
5
|
+
wireType: number;
|
|
6
|
+
data: Uint8Array;
|
|
7
|
+
};
|
|
8
|
+
/** Wrap one Connect-protocol message: 1 flag byte + 4-byte big-endian length + payload. */
|
|
9
|
+
export declare function frameConnectMessage(data: Uint8Array, flags?: number): Buffer;
|
|
10
|
+
/** Well-formed protobuf-es `$unknown` entries on `message`; anything else on the bag is ignored. */
|
|
11
|
+
export declare function protoUnknownFields(message: object): ProtoUnknownField[];
|
|
3
12
|
/**
|
|
4
13
|
* Answer a Cursor `interaction_query` so the Run RPC can continue.
|
|
5
14
|
*
|
|
@@ -13,3 +22,4 @@ import { type InteractionQuery } from "@oh-my-pi/pi-catalog/discovery/cursor-pro
|
|
|
13
22
|
* VM setup is left unanswered rather than reporting a fake success.
|
|
14
23
|
*/
|
|
15
24
|
export declare function handleInteractionQuery(query: InteractionQuery, h2Request: http2.ClientHttp2Stream): void;
|
|
25
|
+
export {};
|
|
@@ -397,7 +397,7 @@ export interface ChatCompletionChunk {
|
|
|
397
397
|
/** Moderation results, present on the moderation chunk when requested. */
|
|
398
398
|
moderation?: ChatCompletionChunkModeration | null;
|
|
399
399
|
/** Processing type actually used for serving the request. */
|
|
400
|
-
service_tier?: "auto" | "default" | "flex" | "scale" | "priority" | null;
|
|
400
|
+
service_tier?: "auto" | "default" | "flex" | "scale" | "priority" | "ultrafast" | null;
|
|
401
401
|
/** Deprecated by OpenAI: backend configuration fingerprint, pairs with `seed`. */
|
|
402
402
|
system_fingerprint?: string;
|
|
403
403
|
/** Only with `stream_options: {"include_usage": true}`; null except on the last chunk. */
|
|
@@ -632,7 +632,7 @@ export interface ChatCompletionCreateParamsBase {
|
|
|
632
632
|
/** Deprecated by OpenAI (Beta): best-effort deterministic sampling seed. */
|
|
633
633
|
seed?: number | null;
|
|
634
634
|
/** Processing type used for serving the request. */
|
|
635
|
-
service_tier?: "auto" | "default" | "flex" | "scale" | "priority" | null;
|
|
635
|
+
service_tier?: "auto" | "default" | "flex" | "scale" | "priority" | "ultrafast" | null;
|
|
636
636
|
/** Up to 4 sequences where the API will stop generating further tokens. */
|
|
637
637
|
stop?: string | null | Array<string>;
|
|
638
638
|
/** Whether to store the output for model distillation or evals. */
|
|
@@ -64,7 +64,7 @@ export interface RequestBody {
|
|
|
64
64
|
client_metadata?: Record<string, string>;
|
|
65
65
|
max_output_tokens?: number;
|
|
66
66
|
max_completion_tokens?: number;
|
|
67
|
-
service_tier?: "auto" | "default" | "flex" | "scale" | "priority" | null;
|
|
67
|
+
service_tier?: "auto" | "default" | "flex" | "scale" | "priority" | "ultrafast" | null;
|
|
68
68
|
/** Explicit cyber access program for this request; see `openai-codex/access-programs.ts`. */
|
|
69
69
|
access_programs?: {
|
|
70
70
|
cyber: string;
|
|
@@ -772,7 +772,7 @@ export interface Response {
|
|
|
772
772
|
* When this parameter is set, the response body will include the `service_tier`
|
|
773
773
|
* utilized.
|
|
774
774
|
*/
|
|
775
|
-
service_tier?: "auto" | "default" | "flex" | "scale" | "priority" | null;
|
|
775
|
+
service_tier?: "auto" | "default" | "flex" | "scale" | "priority" | "ultrafast" | null;
|
|
776
776
|
/**
|
|
777
777
|
* The status of the response generation. One of `completed`, `failed`,
|
|
778
778
|
* `in_progress`, `cancelled`, `queued`, or `incomplete`.
|
|
@@ -5772,7 +5772,7 @@ export interface ResponseCreateParamsBase {
|
|
|
5772
5772
|
* When this parameter is set, the response body will include the `service_tier`
|
|
5773
5773
|
* utilized.
|
|
5774
5774
|
*/
|
|
5775
|
-
service_tier?: "auto" | "default" | "flex" | "scale" | "priority" | null;
|
|
5775
|
+
service_tier?: "auto" | "default" | "flex" | "scale" | "priority" | "ultrafast" | null;
|
|
5776
5776
|
/**
|
|
5777
5777
|
* Whether to store the generated model response for later retrieval via API.
|
|
5778
5778
|
*/
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
/** Bundled xAI API endpoint for the `xai` and `xai-oauth` providers. */
|
|
2
|
+
export declare const XAI_DEFAULT_BASE_URL = "https://api.x.ai/v1";
|
|
3
|
+
/**
|
|
4
|
+
* Resolve the base URL for an xAI request (`xai` / `xai-oauth` chat, image
|
|
5
|
+
* generation, web search, and HTTP tools).
|
|
6
|
+
*
|
|
7
|
+
* `XAI_BASE_URL` redirects traffic that targets the bundled default endpoint
|
|
8
|
+
* (or has no base URL); trailing slashes on the override are stripped. A
|
|
9
|
+
* custom `baseUrl` (models.yml, provider config) always wins.
|
|
10
|
+
*
|
|
11
|
+
* The override never receives official xAI OAuth credentials: an `xai-oauth`
|
|
12
|
+
* request whose bearer is an xAI OAuth access token (a JWT, whether stored,
|
|
13
|
+
* from `XAI_OAUTH_TOKEN`, or unknown because no bearer was supplied) stays on
|
|
14
|
+
* the bundled endpoint. API keys, including command-backed ones, follow the
|
|
15
|
+
* override.
|
|
16
|
+
*/
|
|
17
|
+
export declare function resolveXaiBaseUrl(provider: string, baseUrl: string | undefined, bearer: string | undefined): string | undefined;
|
package/dist/types/types.d.ts
CHANGED
|
@@ -83,7 +83,9 @@ export type CacheRetention = "none" | "short" | "long";
|
|
|
83
83
|
* values providers consume on the wire:
|
|
84
84
|
*
|
|
85
85
|
* - OpenAI / OpenAI-Codex: sent verbatim as the `service_tier` field
|
|
86
|
-
* (`flex`/`scale`/`priority`).
|
|
86
|
+
* (`flex`/`scale`/`priority`/`ultrafast`). `ultrafast` is a separate
|
|
87
|
+
* low-latency serving path: sent to the OpenAI API as-is (preview access is
|
|
88
|
+
* per project), and to Codex only for models whose discovery advertises it.
|
|
87
89
|
* - Google (Gemini API + Vertex AI): sent as the top-level `serviceTier`
|
|
88
90
|
* field (`flex`/`priority`).
|
|
89
91
|
* - OpenRouter: passed through as `service_tier`; OpenRouter realizes it for
|
|
@@ -95,7 +97,7 @@ export type CacheRetention = "none" | "short" | "long";
|
|
|
95
97
|
* Per-family scoping is expressed by {@link ServiceTierByFamily}, not by
|
|
96
98
|
* scoped sentinel values — see {@link serviceTierFamily}.
|
|
97
99
|
*/
|
|
98
|
-
export type ServiceTier = "auto" | "default" | "flex" | "scale" | "priority";
|
|
100
|
+
export type ServiceTier = "auto" | "default" | "flex" | "scale" | "priority" | "ultrafast";
|
|
99
101
|
/** Provider families that expose an independent service-tier knob. */
|
|
100
102
|
export type ServiceTierFamily = "openai" | "anthropic" | "google";
|
|
101
103
|
/**
|
|
@@ -105,7 +107,7 @@ export type ServiceTierFamily = "openai" | "anthropic" | "google";
|
|
|
105
107
|
* models mid-session.
|
|
106
108
|
*/
|
|
107
109
|
export type ServiceTierByFamily = Partial<Record<ServiceTierFamily, ServiceTier>>;
|
|
108
|
-
type ServiceTierModel = Pick<Model, "provider" | "api" | "identity"
|
|
110
|
+
type ServiceTierModel = Pick<Model, "provider" | "api" | "identity"> & Partial<Pick<Model, "serviceTiers">>;
|
|
109
111
|
/**
|
|
110
112
|
* Classify a model into the service-tier family whose knob governs it, or
|
|
111
113
|
* `undefined` when the model exposes no serving-priority control.
|
|
@@ -132,6 +134,15 @@ export declare function resolveModelServiceTier(tiers: ServiceTierByFamily | nul
|
|
|
132
134
|
* Vertex) and OpenRouter accept `flex`/`priority`; Fireworks Serverless
|
|
133
135
|
* realizes only its Priority serving path. Anthropic is absent because it
|
|
134
136
|
* realizes `priority` via `speed: "fast"`.
|
|
137
|
+
*
|
|
138
|
+
* Codex-backend models (`openai-codex-responses`): `ultrafast` is sent only
|
|
139
|
+
* when the model's discovered `service_tiers` lists it. `priority`/`scale`
|
|
140
|
+
* are dropped only when that list is non-empty and omits them (codex-rs
|
|
141
|
+
* `service_tier_for_request`); an empty or missing list counts as "not
|
|
142
|
+
* reported" — accounts whose `/models` lists no tiers keep `/fast` — so the
|
|
143
|
+
* provider-level answer stands. `flex` and `default` are never gated.
|
|
144
|
+
* First-party OpenAI takes `ultrafast` as-is. A bare provider string cannot
|
|
145
|
+
* carry the list, so it answers for the provider alone.
|
|
135
146
|
*/
|
|
136
147
|
export declare function shouldSendServiceTier(serviceTier: ServiceTier | null | undefined, target: Provider | ServiceTierModel | undefined): boolean;
|
|
137
148
|
/**
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import type { UsageStatus } from "../usage.js";
|
|
1
|
+
import type { UsageAmount, UsageStatus } from "../usage.js";
|
|
2
2
|
/** Milliseconds in one hour. */
|
|
3
3
|
export declare const HOUR_MS: number;
|
|
4
4
|
/** Milliseconds in one day. */
|
|
@@ -11,3 +11,15 @@ export declare function parsePositiveTimestamp(value: unknown): number | undefin
|
|
|
11
11
|
export declare function parseIsoTimestamp(value: unknown): number | undefined;
|
|
12
12
|
/** Maps a used fraction to the standard quota status thresholds. */
|
|
13
13
|
export declare function usageStatus(usedFraction: number | undefined): UsageStatus;
|
|
14
|
+
/**
|
|
15
|
+
* Builds an amount from absolute counters. Without an authoritative
|
|
16
|
+
* `usedFraction`, it derives one from `used / limit` (capped at 1).
|
|
17
|
+
* Undefined fields are omitted rather than emitted as `undefined` keys.
|
|
18
|
+
*/
|
|
19
|
+
export declare function buildUsageAmount(args: {
|
|
20
|
+
used: number | undefined;
|
|
21
|
+
limit: number | undefined;
|
|
22
|
+
remaining: number | undefined;
|
|
23
|
+
usedFraction?: number;
|
|
24
|
+
unit: UsageAmount["unit"];
|
|
25
|
+
}): UsageAmount;
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@oh-my-pi/pi-ai",
|
|
3
|
-
"version": "18.4.
|
|
3
|
+
"version": "18.4.4",
|
|
4
4
|
"description": "Unified LLM API with automatic model discovery and provider configuration",
|
|
5
5
|
"keywords": [
|
|
6
6
|
"ai",
|
|
@@ -155,11 +155,11 @@
|
|
|
155
155
|
"fmt": "oxfmt --no-error-on-unmatched-pattern 'src/**/*.{ts,tsx}' '{test,bench,examples,scripts}/**/*.ts' '*.ts'"
|
|
156
156
|
},
|
|
157
157
|
"dependencies": {
|
|
158
|
-
"@oh-my-pi/omptype": "18.4.
|
|
159
|
-
"@oh-my-pi/pi-catalog": "18.4.
|
|
160
|
-
"@oh-my-pi/pi-natives": "18.4.
|
|
161
|
-
"@oh-my-pi/pi-utils": "18.4.
|
|
162
|
-
"@oh-my-pi/pi-wire": "18.4.
|
|
158
|
+
"@oh-my-pi/omptype": "18.4.4",
|
|
159
|
+
"@oh-my-pi/pi-catalog": "18.4.4",
|
|
160
|
+
"@oh-my-pi/pi-natives": "18.4.4",
|
|
161
|
+
"@oh-my-pi/pi-utils": "18.4.4",
|
|
162
|
+
"@oh-my-pi/pi-wire": "18.4.4"
|
|
163
163
|
},
|
|
164
164
|
"devDependencies": {
|
|
165
165
|
"@types/bun": "^1.3.14"
|
|
@@ -31,6 +31,7 @@ import type {
|
|
|
31
31
|
UsageStaleResponse,
|
|
32
32
|
} from "./types";
|
|
33
33
|
import { AUTH_BROKER_CAPABILITIES_HEADER, AUTH_BROKER_CAPABILITY_CODEX_METER_BLOCK_SCOPES } from "./types";
|
|
34
|
+
import { parseGenerationTag } from "./protocol";
|
|
34
35
|
import {
|
|
35
36
|
clientUsageReportResponseSchema,
|
|
36
37
|
clientUsageSummaryResponseSchema,
|
|
@@ -112,18 +113,6 @@ export type FetchSnapshotResult =
|
|
|
112
113
|
| { status: 200; snapshot: SnapshotResponse; generation: number }
|
|
113
114
|
| { status: 304; generation: number };
|
|
114
115
|
|
|
115
|
-
function parseGenerationTag(header: string | null): number | undefined {
|
|
116
|
-
if (!header) return undefined;
|
|
117
|
-
let value = header.trim();
|
|
118
|
-
if (value.startsWith("W/")) value = value.slice(2).trim();
|
|
119
|
-
if (value.startsWith('"') && value.endsWith('"') && value.length >= 2) {
|
|
120
|
-
value = value.slice(1, -1);
|
|
121
|
-
}
|
|
122
|
-
const generation = Number(value);
|
|
123
|
-
if (!Number.isInteger(generation) || generation < 0) return undefined;
|
|
124
|
-
return generation;
|
|
125
|
-
}
|
|
126
|
-
|
|
127
116
|
const DEFAULT_TIMEOUT_MS = 10_000;
|
|
128
117
|
const DEFAULT_MAX_RETRIES = 1;
|
|
129
118
|
|
|
@@ -0,0 +1,32 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Auth-broker wire-protocol helpers shared by the server and the client store.
|
|
3
|
+
*/
|
|
4
|
+
import type { CredentialBlockSnapshot } from "./types";
|
|
5
|
+
|
|
6
|
+
/** Parse a snapshot `ETag` / `If-None-Match` value (`"N"`, `W/"N"`, or bare `N`) into a generation. */
|
|
7
|
+
export function parseGenerationTag(header: string | null): number | undefined {
|
|
8
|
+
if (!header) return undefined;
|
|
9
|
+
let value = header.trim();
|
|
10
|
+
if (value.startsWith("W/")) value = value.slice(2).trim();
|
|
11
|
+
if (value.startsWith('"') && value.endsWith('"') && value.length >= 2) {
|
|
12
|
+
value = value.slice(1, -1);
|
|
13
|
+
}
|
|
14
|
+
const generation = Number(value);
|
|
15
|
+
if (!Number.isInteger(generation) || generation < 0) return undefined;
|
|
16
|
+
return generation;
|
|
17
|
+
}
|
|
18
|
+
|
|
19
|
+
/**
|
|
20
|
+
* Canonical order for a credential's block snapshots. The server and the client
|
|
21
|
+
* store both sort with it, because block lists are compared positionally; the
|
|
22
|
+
* `updatedAtMs` tiebreak keeps otherwise-identical blocks in one stable order.
|
|
23
|
+
*/
|
|
24
|
+
export function compareCredentialBlockSnapshots(a: CredentialBlockSnapshot, b: CredentialBlockSnapshot): number {
|
|
25
|
+
const provider = a.providerKey.localeCompare(b.providerKey);
|
|
26
|
+
if (provider !== 0) return provider;
|
|
27
|
+
const scope = a.blockScope.localeCompare(b.blockScope);
|
|
28
|
+
if (scope !== 0) return scope;
|
|
29
|
+
const blockedUntil = a.blockedUntilMs - b.blockedUntilMs;
|
|
30
|
+
if (blockedUntil !== 0) return blockedUntil;
|
|
31
|
+
return (a.updatedAtMs ?? 0) - (b.updatedAtMs ?? 0);
|
|
32
|
+
}
|
|
@@ -24,7 +24,9 @@ import * as AIError from "../error";
|
|
|
24
24
|
import type { OAuthCredentials } from "../registry/oauth/types";
|
|
25
25
|
import type { Provider } from "../types";
|
|
26
26
|
import type { ClientUsageIdentity, ObservedUsageEntry, UsageReport } from "../usage";
|
|
27
|
+
import { raceSignal } from "../auth/abort";
|
|
27
28
|
import { type AuthBrokerClient, AuthBrokerError, AuthBrokerStreamUnsupportedError } from "./client";
|
|
29
|
+
import { compareCredentialBlockSnapshots } from "./protocol";
|
|
28
30
|
import type {
|
|
29
31
|
CredentialBlockSnapshot,
|
|
30
32
|
RefresherSchedule,
|
|
@@ -67,16 +69,6 @@ const BACKGROUND_BACKOFF_MAX_MS = 30_000;
|
|
|
67
69
|
/** Idle window after the last foreground store use before background sync parks. */
|
|
68
70
|
const BACKGROUND_IDLE_MS = 20_000;
|
|
69
71
|
|
|
70
|
-
function compareCredentialBlockSnapshots(a: CredentialBlockSnapshot, b: CredentialBlockSnapshot): number {
|
|
71
|
-
const provider = a.providerKey.localeCompare(b.providerKey);
|
|
72
|
-
if (provider !== 0) return provider;
|
|
73
|
-
const scope = a.blockScope.localeCompare(b.blockScope);
|
|
74
|
-
if (scope !== 0) return scope;
|
|
75
|
-
const blockedUntil = a.blockedUntilMs - b.blockedUntilMs;
|
|
76
|
-
if (blockedUntil !== 0) return blockedUntil;
|
|
77
|
-
return (a.updatedAtMs ?? 0) - (b.updatedAtMs ?? 0);
|
|
78
|
-
}
|
|
79
|
-
|
|
80
72
|
function toCredentialBlockSnapshot(block: StoredCredentialBlock): CredentialBlockSnapshot {
|
|
81
73
|
return {
|
|
82
74
|
providerKey: block.providerKey,
|
|
@@ -1209,7 +1201,7 @@ export class RemoteAuthCredentialStore implements AuthCredentialStore {
|
|
|
1209
1201
|
*/
|
|
1210
1202
|
async fetchUsageReports(signal?: AbortSignal): Promise<UsageReport[] | null> {
|
|
1211
1203
|
this.#noteActivity();
|
|
1212
|
-
const reports = await
|
|
1204
|
+
const reports = await raceSignal(this.#loadUsageReports(), signal, "auth-broker request aborted");
|
|
1213
1205
|
if (!reports) return null;
|
|
1214
1206
|
return this.#filterUsageReports(this.#applyUsageOverlays(reports));
|
|
1215
1207
|
}
|
|
@@ -1229,7 +1221,7 @@ export class RemoteAuthCredentialStore implements AuthCredentialStore {
|
|
|
1229
1221
|
signal?: AbortSignal,
|
|
1230
1222
|
): Promise<UsageReport | null> {
|
|
1231
1223
|
this.#noteActivity();
|
|
1232
|
-
const reports = await
|
|
1224
|
+
const reports = await raceSignal(this.#loadUsageReports(), signal, "auth-broker request aborted");
|
|
1233
1225
|
const visibleReports = reports ? this.#filterUsageReports(reports) : null;
|
|
1234
1226
|
const matched = visibleReports ? matchUsageReport(visibleReports, provider, credential) : null;
|
|
1235
1227
|
const overlay = this.#getActiveUsageOverlay(provider, credential);
|
|
@@ -1310,34 +1302,6 @@ export class RemoteAuthCredentialStore implements AuthCredentialStore {
|
|
|
1310
1302
|
return merged;
|
|
1311
1303
|
}
|
|
1312
1304
|
|
|
1313
|
-
/**
|
|
1314
|
-
* Reject the awaited promise when the caller's signal aborts, without
|
|
1315
|
-
* affecting the shared upstream fetch. Used to give each caller their
|
|
1316
|
-
* own cancel without one caller's abort cascading into a peer's in-flight
|
|
1317
|
-
* request through the single-flight `#usageInflight`.
|
|
1318
|
-
*/
|
|
1319
|
-
#raceWithSignal<T>(promise: Promise<T>, signal?: AbortSignal): Promise<T> {
|
|
1320
|
-
if (!signal) return promise;
|
|
1321
|
-
if (signal.aborted) return Promise.reject(new AIError.AbortError("auth-broker request aborted"));
|
|
1322
|
-
return new Promise<T>((resolve, reject) => {
|
|
1323
|
-
const onAbort = (): void => {
|
|
1324
|
-
signal.removeEventListener("abort", onAbort);
|
|
1325
|
-
reject(new AIError.AbortError("auth-broker request aborted"));
|
|
1326
|
-
};
|
|
1327
|
-
signal.addEventListener("abort", onAbort, { once: true });
|
|
1328
|
-
promise.then(
|
|
1329
|
-
value => {
|
|
1330
|
-
signal.removeEventListener("abort", onAbort);
|
|
1331
|
-
resolve(value);
|
|
1332
|
-
},
|
|
1333
|
-
err => {
|
|
1334
|
-
signal.removeEventListener("abort", onAbort);
|
|
1335
|
-
reject(err);
|
|
1336
|
-
},
|
|
1337
|
-
);
|
|
1338
|
-
});
|
|
1339
|
-
}
|
|
1340
|
-
|
|
1341
1305
|
#replaceBrokerUsageAccounts(entries: readonly SnapshotEntry[]): void {
|
|
1342
1306
|
this.#brokerUsageProviderByCredentialId.clear();
|
|
1343
1307
|
this.#brokerUsageAccountCounts.clear();
|
|
@@ -41,6 +41,7 @@ import {
|
|
|
41
41
|
DEFAULT_SERVER_IDLE_TIMEOUT_S,
|
|
42
42
|
DEFAULT_STREAM_KEEPALIVE_MS,
|
|
43
43
|
} from "./types";
|
|
44
|
+
import { compareCredentialBlockSnapshots, parseGenerationTag } from "./protocol";
|
|
44
45
|
import {
|
|
45
46
|
clientUsageReportRequestSchema,
|
|
46
47
|
credentialBlockDeleteRequestSchema,
|
|
@@ -165,18 +166,6 @@ function snapshotHeaders(generation: number): Record<string, string> {
|
|
|
165
166
|
};
|
|
166
167
|
}
|
|
167
168
|
|
|
168
|
-
function parseGenerationTag(header: string | null): number | undefined {
|
|
169
|
-
if (!header) return undefined;
|
|
170
|
-
let value = header.trim();
|
|
171
|
-
if (value.startsWith("W/")) value = value.slice(2).trim();
|
|
172
|
-
if (value.startsWith('"') && value.endsWith('"') && value.length >= 2) {
|
|
173
|
-
value = value.slice(1, -1);
|
|
174
|
-
}
|
|
175
|
-
const generation = Number(value);
|
|
176
|
-
if (!Number.isInteger(generation) || generation < 0) return undefined;
|
|
177
|
-
return generation;
|
|
178
|
-
}
|
|
179
|
-
|
|
180
169
|
function parseWaitMs(url: URL): number {
|
|
181
170
|
const raw = url.searchParams.get("wait");
|
|
182
171
|
if (raw === null) return 0;
|
|
@@ -316,14 +305,6 @@ function computeRotatesInMs(
|
|
|
316
305
|
return Math.max(0, rotatesAt - serverNowMs);
|
|
317
306
|
}
|
|
318
307
|
|
|
319
|
-
function compareCredentialBlockSnapshots(a: CredentialBlockSnapshot, b: CredentialBlockSnapshot): number {
|
|
320
|
-
const provider = a.providerKey.localeCompare(b.providerKey);
|
|
321
|
-
if (provider !== 0) return provider;
|
|
322
|
-
const scope = a.blockScope.localeCompare(b.blockScope);
|
|
323
|
-
if (scope !== 0) return scope;
|
|
324
|
-
return a.blockedUntilMs - b.blockedUntilMs;
|
|
325
|
-
}
|
|
326
|
-
|
|
327
308
|
const CODEX_BLOCK_PROVIDER_KEY = "openai-codex:oauth";
|
|
328
309
|
const CODEX_LEGACY_PROJECTED_BLOCK_SCOPES = new Set(["chat", "spark", "shared"]);
|
|
329
310
|
|
|
@@ -9,6 +9,7 @@
|
|
|
9
9
|
import * as fs from "node:fs/promises";
|
|
10
10
|
import * as path from "node:path";
|
|
11
11
|
import { isEnoent, logger } from "@oh-my-pi/pi-utils";
|
|
12
|
+
import { asStrict } from "../providers/aws-sigv4";
|
|
12
13
|
import type { SnapshotResponse } from "./types";
|
|
13
14
|
|
|
14
15
|
const MAGIC = new Uint8Array([0x4f, 0x4d, 0x50, 0x53]); // "OMPS"
|
|
@@ -210,15 +211,6 @@ async function deriveAesKey(token: string, usages: Array<"encrypt" | "decrypt">)
|
|
|
210
211
|
return globalThis.crypto.subtle.importKey("raw", digest, AES_ALGORITHM, false, usages);
|
|
211
212
|
}
|
|
212
213
|
|
|
213
|
-
function asStrict(bytes: Uint8Array): Uint8Array<ArrayBuffer> {
|
|
214
|
-
if (bytes.buffer instanceof ArrayBuffer && bytes.byteOffset === 0 && bytes.byteLength === bytes.buffer.byteLength) {
|
|
215
|
-
return bytes as Uint8Array<ArrayBuffer>;
|
|
216
|
-
}
|
|
217
|
-
const copy = new Uint8Array(bytes.byteLength);
|
|
218
|
-
copy.set(bytes);
|
|
219
|
-
return copy;
|
|
220
|
-
}
|
|
221
|
-
|
|
222
214
|
function randomHex(byteLength: number): string {
|
|
223
215
|
const bytes = new Uint8Array(byteLength);
|
|
224
216
|
globalThis.crypto.getRandomValues(bytes);
|
package/src/dialect/anthropic.ts
CHANGED
|
@@ -1,14 +1,8 @@
|
|
|
1
|
-
import { parseJsonWithRepair } from "@oh-my-pi/pi-utils";
|
|
1
|
+
import { escapeXmlText, parseJsonWithRepair } from "@oh-my-pi/pi-utils";
|
|
2
2
|
import type { Message, ToolCall } from "../types";
|
|
3
3
|
import dialectPrompt from "./anthropic.md" with { type: "text" };
|
|
4
|
-
import { buildArgShapes, buildStringArgsResolver, mintToolCallId
|
|
5
|
-
import {
|
|
6
|
-
escapeXmlAttr,
|
|
7
|
-
escapeXmlText,
|
|
8
|
-
renderDelimitedThinking,
|
|
9
|
-
renderLegacyTextTranscript,
|
|
10
|
-
stringifyJson,
|
|
11
|
-
} from "./rendering";
|
|
4
|
+
import { buildArgShapes, buildStringArgsResolver, mintToolCallId } from "./coercion";
|
|
5
|
+
import { renderDelimitedThinking, renderInvoke, renderInvokes, renderLegacyTextTranscript } from "./rendering";
|
|
12
6
|
import type {
|
|
13
7
|
DialectDefinition,
|
|
14
8
|
DialectRenderOptions,
|
|
@@ -578,22 +572,6 @@ function renderTranscript(messages: readonly Message[], options: DialectRenderOp
|
|
|
578
572
|
});
|
|
579
573
|
}
|
|
580
574
|
|
|
581
|
-
function renderInvoke(call: ToolCall, shape: ToolArgShape | undefined): string {
|
|
582
|
-
let body = `<invoke name="${escapeXmlAttr(call.name)}">`;
|
|
583
|
-
for (const key in call.arguments) {
|
|
584
|
-
const value = call.arguments[key];
|
|
585
|
-
const isString = shape?.stringArgs.has(key) === true;
|
|
586
|
-
const rendered = isString && typeof value === "string" ? value : stringifyJson(value);
|
|
587
|
-
body += `<parameter name="${escapeXmlAttr(key)}">${rendered}</parameter>`;
|
|
588
|
-
}
|
|
589
|
-
return `${body}</invoke>`;
|
|
590
|
-
}
|
|
591
|
-
|
|
592
|
-
function renderInvokes(calls: readonly ToolCall[], tools: NonNullable<DialectRenderOptions["tools"]>): string {
|
|
593
|
-
const shapes = buildArgShapes(tools);
|
|
594
|
-
return calls.map(call => renderInvoke(call, shapes.get(call.name))).join("\n");
|
|
595
|
-
}
|
|
596
|
-
|
|
597
575
|
const definition: DialectDefinition = {
|
|
598
576
|
dialect: "anthropic",
|
|
599
577
|
prompt: dialectPrompt,
|