@oh-my-pi/pi-ai 18.4.3 → 18.4.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (58) hide show
  1. package/CHANGELOG.md +20 -17
  2. package/dist/types/auth-broker/protocol.d.ts +12 -0
  3. package/dist/types/dialect/rendering.d.ts +4 -0
  4. package/dist/types/images/shared.d.ts +5 -2
  5. package/dist/types/providers/anthropic-wire.d.ts +9 -1
  6. package/dist/types/providers/anthropic.d.ts +17 -0
  7. package/dist/types/providers/aws-sigv4.d.ts +5 -0
  8. package/dist/types/providers/bedrock-anthropic.d.ts +9 -0
  9. package/dist/types/providers/bedrock-request-metadata.d.ts +2 -0
  10. package/dist/types/providers/cursor/interaction-query.d.ts +10 -0
  11. package/dist/types/providers/openai-chat-wire.d.ts +2 -2
  12. package/dist/types/providers/openai-codex/request-transformer.d.ts +1 -1
  13. package/dist/types/providers/openai-responses-wire.d.ts +2 -2
  14. package/dist/types/providers/xai-base-url.d.ts +17 -0
  15. package/dist/types/types.d.ts +14 -3
  16. package/dist/types/usage/shared.d.ts +13 -1
  17. package/package.json +6 -6
  18. package/src/auth-broker/client.ts +1 -12
  19. package/src/auth-broker/protocol.ts +32 -0
  20. package/src/auth-broker/remote-store.ts +4 -40
  21. package/src/auth-broker/server.ts +1 -20
  22. package/src/auth-broker/snapshot-cache.ts +1 -9
  23. package/src/dialect/anthropic.ts +3 -25
  24. package/src/dialect/minimax.ts +3 -24
  25. package/src/dialect/rendering.ts +18 -0
  26. package/src/dialect/xml.ts +3 -19
  27. package/src/images/openai-images.ts +10 -4
  28. package/src/images/shared.ts +9 -4
  29. package/src/providers/amazon-bedrock.ts +3 -6
  30. package/src/providers/anthropic-compaction.ts +10 -1
  31. package/src/providers/anthropic-wire.ts +12 -1
  32. package/src/providers/anthropic.ts +58 -15
  33. package/src/providers/aws-sigv4.ts +1 -1
  34. package/src/providers/bedrock-anthropic.ts +30 -0
  35. package/src/providers/bedrock-request-metadata.ts +6 -0
  36. package/src/providers/connect-error-detail.ts +1 -5
  37. package/src/providers/cursor/interaction-query.ts +4 -2
  38. package/src/providers/cursor.ts +30 -26
  39. package/src/providers/google-shared.ts +8 -2
  40. package/src/providers/openai-chat-wire.ts +2 -2
  41. package/src/providers/openai-codex/request-transformer.ts +1 -1
  42. package/src/providers/openai-codex-responses.ts +19 -14
  43. package/src/providers/openai-responses-wire.ts +2 -2
  44. package/src/providers/openai-shared.ts +4 -0
  45. package/src/providers/xai-base-url.ts +32 -0
  46. package/src/types.ts +35 -4
  47. package/src/usage/claude.ts +4 -11
  48. package/src/usage/cline-pass.ts +2 -14
  49. package/src/usage/cursor.ts +9 -1
  50. package/src/usage/openai-codex.ts +3 -5
  51. package/src/usage/shared.ts +28 -1
  52. package/src/usage/synthetic.ts +4 -40
  53. package/src/usage/umans.ts +8 -36
  54. package/src/usage/zai.ts +10 -38
  55. package/src/utils/http-inspector.ts +4 -8
  56. package/src/utils/schema/json-schema-validator.ts +23 -26
  57. package/src/utils/schema/meta-validator.ts +4 -7
  58. package/src/utils/schema/wire.ts +17 -21
package/CHANGELOG.md CHANGED
@@ -2,6 +2,25 @@
2
2
 
3
3
  ## [Unreleased]
4
4
 
5
+ ## [18.4.4] - 2026-09-29
6
+
7
+ ### Added
8
+
9
+ - Added the `ultrafast` service tier. It is sent to the OpenAI API as-is, and to Codex only for models that list it in their discovered service tiers; other providers never receive it. On Codex websockets, switching into or out of `ultrafast` starts a new response chain instead of reusing `previous_response_id`, matching the Codex CLI. Ultrafast turns are costed at standard rates because no Ultrafast price is published yet ([#13782](https://github.com/can1357/oh-my-pi/pull/13782) by [@H4vC](https://github.com/H4vC)).
10
+
11
+ ### Changed
12
+
13
+ - Changed to fall back to adaptive thinking when between_tools is used with xhigh effort
14
+ - xAI requests (`xai`, `xai-oauth` chat and image generation) honor `XAI_BASE_URL` again when the model uses the bundled `https://api.x.ai/v1` endpoint; a custom `baseUrl` from models.yml still wins, and `xai-oauth` OAuth access tokens always stay on the bundled endpoint.
15
+
16
+ ### Fixed
17
+
18
+ - Fixed Claude on Amazon Bedrock's Anthropic Messages routes (`/anthropic` on bedrock-runtime and bedrock-mantle): runtime requests no longer fail with a request-metadata 400, and both routes use Anthropic's on-demand compaction ([#13311](https://github.com/can1357/oh-my-pi/pull/13311) by [@mustafaabidali](https://github.com/mustafaabidali)).
19
+ - `/usage` no longer shows an always-empty `gpt-4 requests` row for Cursor accounts on usage-based plans; the Cursor Models and Other Models meters remain ([#13726](https://github.com/can1357/oh-my-pi/pull/13726) by [@will-bogusz](https://github.com/will-bogusz)).
20
+ - Cursor turns routed through an HTTP proxy now finish instead of hanging after the response completes ([#13724](https://github.com/can1357/oh-my-pi/pull/13724) by [@will-bogusz](https://github.com/will-bogusz)).
21
+ - Fixed Codex requests sending `priority` (and `scale`) to models whose discovered service tiers list other tiers but not that one, matching the Codex CLI; an empty or missing list is treated as not reported, so `priority` is still sent and `/fast` keeps working on accounts whose `/models` lists no tiers (`flex` is always allowed) ([#13782](https://github.com/can1357/oh-my-pi/pull/13782) by [@H4vC](https://github.com/H4vC)).
22
+ - Fixed Codex priority cost: a turn the backend reports as served at `default` is no longer billed at the priority multiplier ([#13782](https://github.com/can1357/oh-my-pi/pull/13782) by [@H4vC](https://github.com/H4vC)).
23
+
5
24
  ## [18.4.3] - 2026-09-28
6
25
 
7
26
  ### Added
@@ -2300,20 +2319,4 @@
2300
2319
  - Fixed the platform OpenAI Responses and Codex websocket stale-chain classifiers missing the "Unsupported parameter: previous_response_id" rejection phrasing (FastAPI-style `detail` body with no `error.code`), so a chained turn now falls back to a full-transcript replay instead of surfacing the 400
2301
2320
  - Fixed the HTTP-400 raw-request dump for Codex SSE to record the body actually sent on the wire instead of the pre-transport request body, which made chained-request failures look like the rejected parameter was never sent
2302
2321
 
2303
- ## [15.11.7] - 2026-06-12
2304
-
2305
- ### Added
2306
-
2307
- - Added `requestModelId` and `thinking.suppress` options to `google-gemini-cli` so collapsed effort-tier variants serialize their per-effort upstream wire id, and thinking-off requests on models with `thinking.suppressWhenOff` send an explicit `thinkingConfig` (`includeThoughts: false` with `thinkingLevel: "MINIMAL"` or `thinkingBudget: 0`) — Cloud Code Assist re-applies the per-id baked server default when the config is omitted, silently thinking and billing the tokens
2308
- - Added mandatory-reasoning clamping: models baked with `thinking.requiresEffort` floor omitted or disabled reasoning to the lowest supported effort in every api mapping, and `disableReasoning` no longer emits OpenRouter `reasoning: { enabled: false }` for them — fixes `omp bench` and utility requests 400ing with "Reasoning is mandatory for this endpoint and cannot be disabled" on OpenRouter Gemini 3.x
2309
-
2310
- ### Changed
2311
-
2312
- - Changed `google-gemini-cli` request mapping to route per-request wire ids via `resolveWireModelId`: the session effort picks the backing variant id (collapsed `gemini-3.5-flash` at high → `gemini-3.5-flash-low`; claude pairs route off → bare id, efforts → `-thinking`) while `AssistantMessage.model` and usage attribution stay on the logical id. A thinking budget clamped to zero now falls through to the thinking-off path (off routing plus suppression) instead of only disabling thinking
2313
- - Changed `openai-completions` and `anthropic-messages` to serialize per-request wire ids via `resolveWireModelId`, so collapsed `X`/`X-thinking` pairs on aggregators and custom providers switch to the thinking SKU when reasoning is enabled (previously only `google-gemini-cli` routed effort-tier variants)
2314
-
2315
- ### Fixed
2316
-
2317
- - Fixed `google-gemini-cli` ignoring `Model.requestModelId` when serializing the request model id
2318
-
2319
- Older entries are archived in [packages/ai/CHANGELOG.md@689a3418cb45](https://github.com/can1357/oh-my-pi/blob/689a3418cb45d54a459cde2e1abf3f66f50e47a4/packages/ai/CHANGELOG.md).
2322
+ Older entries are archived in [packages\ai\CHANGELOG.md@07e9197a3012](https://github.com/can1357/oh-my-pi/blob/07e9197a3012f58c459f1faabeb324decc21f41d/packages\ai\CHANGELOG.md).
@@ -0,0 +1,12 @@
1
+ /**
2
+ * Auth-broker wire-protocol helpers shared by the server and the client store.
3
+ */
4
+ import type { CredentialBlockSnapshot } from "./types.js";
5
+ /** Parse a snapshot `ETag` / `If-None-Match` value (`"N"`, `W/"N"`, or bare `N`) into a generation. */
6
+ export declare function parseGenerationTag(header: string | null): number | undefined;
7
+ /**
8
+ * Canonical order for a credential's block snapshots. The server and the client
9
+ * store both sort with it, because block lists are compared positionally; the
10
+ * `updatedAtMs` tiebreak keeps otherwise-identical blocks in one stable order.
11
+ */
12
+ export declare function compareCredentialBlockSnapshots(a: CredentialBlockSnapshot, b: CredentialBlockSnapshot): number;
@@ -1,4 +1,5 @@
1
1
  import type { AssistantMessage, Message, ToolCall } from "../types.js";
2
+ import { type ToolArgShape } from "./coercion.js";
2
3
  import type { DialectRenderOptions, DialectToolResult } from "./types.js";
3
4
  export declare function renderToolResponseResults(results: readonly DialectToolResult[]): string;
4
5
  export declare function kimiCallId(name: string, id: string, index: number): string;
@@ -15,6 +16,9 @@ export declare function pyCall(name: string, args: Record<string, unknown>): str
15
16
  export declare function pyValue(value: unknown): string;
16
17
  export declare function escapeXmlAttr(value: string): string;
17
18
  export declare function escapeXmlText(value: string): string;
19
+ /** Render one Anthropic-style `<invoke>`; declared string args stay raw, everything else is JSON. */
20
+ export declare function renderInvoke(call: ToolCall, shape: ToolArgShape | undefined): string;
21
+ export declare function renderInvokes(calls: readonly ToolCall[], tools: NonNullable<DialectRenderOptions["tools"]>): string;
18
22
  export type AssistantTranscriptParts = {
19
23
  readonly text: string;
20
24
  readonly thinking: string;
@@ -10,9 +10,11 @@ export declare function usageFromWire(value: unknown): Usage;
10
10
  export declare function imageBaseUrl(model: Model): string;
11
11
  export declare function modelHeaders(model: Model, signal?: AbortSignal): Promise<Record<string, string>>;
12
12
  export declare function errorMessage(rawText: string): string;
13
+ /** Request URL, or a builder for routes that depend on the bearer (xAI's `XAI_BASE_URL` rule). */
14
+ type ImageRequestUrl = string | ((bearer: string) => string);
13
15
  export declare function postJson(options: {
14
16
  model: Model;
15
- url: string;
17
+ url: ImageRequestUrl;
16
18
  body: unknown;
17
19
  apiKey: ApiKey;
18
20
  fetch: FetchImpl;
@@ -20,7 +22,7 @@ export declare function postJson(options: {
20
22
  }): Promise<unknown>;
21
23
  export declare function postMultipart(options: {
22
24
  model: Model;
23
- url: string;
25
+ url: ImageRequestUrl;
24
26
  body: FormData;
25
27
  apiKey: ApiKey;
26
28
  fetch: FetchImpl;
@@ -32,3 +34,4 @@ export declare function decodeImageResponse(value: unknown, fetch: FetchImpl, si
32
34
  }>;
33
35
  export declare function toDataUrl(image: GeneratedImage): string;
34
36
  export declare function resolveOpenAIImageSize(aspectRatio?: string, imageSize?: string): string | undefined;
37
+ export {};
@@ -225,7 +225,15 @@ export type ThinkingConfigAdaptive = {
225
225
  /** Preserved-thinking prefix mismatch policy. */
226
226
  block_binding?: ThinkingBlockBinding;
227
227
  };
228
- export type ThinkingConfigParam = ThinkingConfigEnabled | ThinkingConfigDisabled | ThinkingConfigAdaptive;
228
+ /**
229
+ * Sonnet 5.5's replacement for `disabled`: no up-front thinking, progress
230
+ * updates between tool calls only. Takes no other field, and effort above
231
+ * `high` is rejected alongside it.
232
+ */
233
+ export type ThinkingConfigBetweenTools = {
234
+ type: "between_tools";
235
+ };
236
+ export type ThinkingConfigParam = ThinkingConfigEnabled | ThinkingConfigDisabled | ThinkingConfigAdaptive | ThinkingConfigBetweenTools;
229
237
  export type OutputConfig = {
230
238
  /** Adaptive-thinking effort level (effort beta). */
231
239
  effort?: "low" | "medium" | "high" | "xhigh" | "max" | null;
@@ -217,6 +217,23 @@ type SystemBlockOptions = {
217
217
  export declare function buildAnthropicSystemBlocks(systemPrompt: readonly string[] | undefined, options?: SystemBlockOptions): AnthropicSystemBlock[] | undefined;
218
218
  export declare function normalizeExtraBetas(betas?: string[] | string): string[];
219
219
  export declare function buildAnthropicClientOptions(args: AnthropicClientOptionsArgs): AnthropicClientOptionsResult;
220
+ /**
221
+ * True when enabled thinking on `model` is budget thinking
222
+ * (`thinking.type: "enabled"` with `budget_tokens`) rather than adaptive.
223
+ */
224
+ export declare function usesBudgetThinking(model: Model<"anthropic-messages">): boolean;
225
+ /** The most output tokens a request to `model` may ask for (`max_tokens` ceiling). */
226
+ export declare function anthropicOutputLimit(model: Model<"anthropic-messages">): number;
227
+ /**
228
+ * The `max_tokens` and thinking budget of budget thinking: `max_tokens`
229
+ * rises to leave {@link OUTPUT_FALLBACK_BUFFER} visible output tokens after
230
+ * the budget, within `maxAllowedTokens`, and the budget shrinks when that
231
+ * ceiling leaves less (a non-positive budget means the ceiling is too low).
232
+ */
233
+ export declare function budgetThinkingOutput(maxTokens: number | undefined, budgetTokens: number, maxAllowedTokens: number): {
234
+ maxTokens: number;
235
+ budgetTokens: number;
236
+ };
220
237
  /**
221
238
  * A single Anthropic conversation turn, including the mid-conversation
222
239
  * `system` role (Opus 4.8+ and Fable/Mythos 5).
@@ -33,6 +33,11 @@ export interface SignParams {
33
33
  /** Override clock for deterministic tests. */
34
34
  date?: Date;
35
35
  }
36
+ /** Coerce a possibly-ArrayBufferLike-backed `Uint8Array` into one over a fresh
37
+ * `ArrayBuffer`, which is what `crypto.subtle.{digest,sign,importKey}` requires
38
+ * under the strict TS DOM typings. No-op when already strict.
39
+ */
40
+ export declare function asStrict(bytes: Uint8Array): Uint8Array<ArrayBuffer>;
36
41
  export declare function toHex(bytes: Uint8Array): string;
37
42
  export declare function sha256(data: Uint8Array | string): Promise<Uint8Array>;
38
43
  export declare function sha256Hex(data: Uint8Array | string): Promise<string>;
@@ -0,0 +1,9 @@
1
+ /**
2
+ * Fit an Anthropic request body to Bedrock's Anthropic Messages API
3
+ * (`compat.bedrockMessagesApi`): both `/anthropic` routes reject the tool
4
+ * `strict` field, and bedrock-runtime rejects a `metadata.user_id` outside
5
+ * Bedrock's request-metadata pattern. A user id that fits is kept, otherwise
6
+ * its embedded session id, otherwise the metadata is dropped. Mutates and
7
+ * returns `payload`.
8
+ */
9
+ export declare function fitBedrockAnthropicPayload<T>(payload: T): T;
@@ -0,0 +1,2 @@
1
+ /** Check Bedrock's request-metadata character and length limits. Keys must also be nonempty. */
2
+ export declare function isBedrockRequestMetadataValue(value: string): boolean;
@@ -1,5 +1,14 @@
1
1
  import type http2 from "node:http2";
2
2
  import { type InteractionQuery } from "@oh-my-pi/pi-catalog/discovery/cursor-proto";
3
+ type ProtoUnknownField = {
4
+ no: number;
5
+ wireType: number;
6
+ data: Uint8Array;
7
+ };
8
+ /** Wrap one Connect-protocol message: 1 flag byte + 4-byte big-endian length + payload. */
9
+ export declare function frameConnectMessage(data: Uint8Array, flags?: number): Buffer;
10
+ /** Well-formed protobuf-es `$unknown` entries on `message`; anything else on the bag is ignored. */
11
+ export declare function protoUnknownFields(message: object): ProtoUnknownField[];
3
12
  /**
4
13
  * Answer a Cursor `interaction_query` so the Run RPC can continue.
5
14
  *
@@ -13,3 +22,4 @@ import { type InteractionQuery } from "@oh-my-pi/pi-catalog/discovery/cursor-pro
13
22
  * VM setup is left unanswered rather than reporting a fake success.
14
23
  */
15
24
  export declare function handleInteractionQuery(query: InteractionQuery, h2Request: http2.ClientHttp2Stream): void;
25
+ export {};
@@ -397,7 +397,7 @@ export interface ChatCompletionChunk {
397
397
  /** Moderation results, present on the moderation chunk when requested. */
398
398
  moderation?: ChatCompletionChunkModeration | null;
399
399
  /** Processing type actually used for serving the request. */
400
- service_tier?: "auto" | "default" | "flex" | "scale" | "priority" | null;
400
+ service_tier?: "auto" | "default" | "flex" | "scale" | "priority" | "ultrafast" | null;
401
401
  /** Deprecated by OpenAI: backend configuration fingerprint, pairs with `seed`. */
402
402
  system_fingerprint?: string;
403
403
  /** Only with `stream_options: {"include_usage": true}`; null except on the last chunk. */
@@ -632,7 +632,7 @@ export interface ChatCompletionCreateParamsBase {
632
632
  /** Deprecated by OpenAI (Beta): best-effort deterministic sampling seed. */
633
633
  seed?: number | null;
634
634
  /** Processing type used for serving the request. */
635
- service_tier?: "auto" | "default" | "flex" | "scale" | "priority" | null;
635
+ service_tier?: "auto" | "default" | "flex" | "scale" | "priority" | "ultrafast" | null;
636
636
  /** Up to 4 sequences where the API will stop generating further tokens. */
637
637
  stop?: string | null | Array<string>;
638
638
  /** Whether to store the output for model distillation or evals. */
@@ -64,7 +64,7 @@ export interface RequestBody {
64
64
  client_metadata?: Record<string, string>;
65
65
  max_output_tokens?: number;
66
66
  max_completion_tokens?: number;
67
- service_tier?: "auto" | "default" | "flex" | "scale" | "priority" | null;
67
+ service_tier?: "auto" | "default" | "flex" | "scale" | "priority" | "ultrafast" | null;
68
68
  /** Explicit cyber access program for this request; see `openai-codex/access-programs.ts`. */
69
69
  access_programs?: {
70
70
  cyber: string;
@@ -772,7 +772,7 @@ export interface Response {
772
772
  * When this parameter is set, the response body will include the `service_tier`
773
773
  * utilized.
774
774
  */
775
- service_tier?: "auto" | "default" | "flex" | "scale" | "priority" | null;
775
+ service_tier?: "auto" | "default" | "flex" | "scale" | "priority" | "ultrafast" | null;
776
776
  /**
777
777
  * The status of the response generation. One of `completed`, `failed`,
778
778
  * `in_progress`, `cancelled`, `queued`, or `incomplete`.
@@ -5772,7 +5772,7 @@ export interface ResponseCreateParamsBase {
5772
5772
  * When this parameter is set, the response body will include the `service_tier`
5773
5773
  * utilized.
5774
5774
  */
5775
- service_tier?: "auto" | "default" | "flex" | "scale" | "priority" | null;
5775
+ service_tier?: "auto" | "default" | "flex" | "scale" | "priority" | "ultrafast" | null;
5776
5776
  /**
5777
5777
  * Whether to store the generated model response for later retrieval via API.
5778
5778
  */
@@ -0,0 +1,17 @@
1
+ /** Bundled xAI API endpoint for the `xai` and `xai-oauth` providers. */
2
+ export declare const XAI_DEFAULT_BASE_URL = "https://api.x.ai/v1";
3
+ /**
4
+ * Resolve the base URL for an xAI request (`xai` / `xai-oauth` chat, image
5
+ * generation, web search, and HTTP tools).
6
+ *
7
+ * `XAI_BASE_URL` redirects traffic that targets the bundled default endpoint
8
+ * (or has no base URL); trailing slashes on the override are stripped. A
9
+ * custom `baseUrl` (models.yml, provider config) always wins.
10
+ *
11
+ * The override never receives official xAI OAuth credentials: an `xai-oauth`
12
+ * request whose bearer is an xAI OAuth access token (a JWT, whether stored,
13
+ * from `XAI_OAUTH_TOKEN`, or unknown because no bearer was supplied) stays on
14
+ * the bundled endpoint. API keys, including command-backed ones, follow the
15
+ * override.
16
+ */
17
+ export declare function resolveXaiBaseUrl(provider: string, baseUrl: string | undefined, bearer: string | undefined): string | undefined;
@@ -83,7 +83,9 @@ export type CacheRetention = "none" | "short" | "long";
83
83
  * values providers consume on the wire:
84
84
  *
85
85
  * - OpenAI / OpenAI-Codex: sent verbatim as the `service_tier` field
86
- * (`flex`/`scale`/`priority`).
86
+ * (`flex`/`scale`/`priority`/`ultrafast`). `ultrafast` is a separate
87
+ * low-latency serving path: sent to the OpenAI API as-is (preview access is
88
+ * per project), and to Codex only for models whose discovery advertises it.
87
89
  * - Google (Gemini API + Vertex AI): sent as the top-level `serviceTier`
88
90
  * field (`flex`/`priority`).
89
91
  * - OpenRouter: passed through as `service_tier`; OpenRouter realizes it for
@@ -95,7 +97,7 @@ export type CacheRetention = "none" | "short" | "long";
95
97
  * Per-family scoping is expressed by {@link ServiceTierByFamily}, not by
96
98
  * scoped sentinel values — see {@link serviceTierFamily}.
97
99
  */
98
- export type ServiceTier = "auto" | "default" | "flex" | "scale" | "priority";
100
+ export type ServiceTier = "auto" | "default" | "flex" | "scale" | "priority" | "ultrafast";
99
101
  /** Provider families that expose an independent service-tier knob. */
100
102
  export type ServiceTierFamily = "openai" | "anthropic" | "google";
101
103
  /**
@@ -105,7 +107,7 @@ export type ServiceTierFamily = "openai" | "anthropic" | "google";
105
107
  * models mid-session.
106
108
  */
107
109
  export type ServiceTierByFamily = Partial<Record<ServiceTierFamily, ServiceTier>>;
108
- type ServiceTierModel = Pick<Model, "provider" | "api" | "identity">;
110
+ type ServiceTierModel = Pick<Model, "provider" | "api" | "identity"> & Partial<Pick<Model, "serviceTiers">>;
109
111
  /**
110
112
  * Classify a model into the service-tier family whose knob governs it, or
111
113
  * `undefined` when the model exposes no serving-priority control.
@@ -132,6 +134,15 @@ export declare function resolveModelServiceTier(tiers: ServiceTierByFamily | nul
132
134
  * Vertex) and OpenRouter accept `flex`/`priority`; Fireworks Serverless
133
135
  * realizes only its Priority serving path. Anthropic is absent because it
134
136
  * realizes `priority` via `speed: "fast"`.
137
+ *
138
+ * Codex-backend models (`openai-codex-responses`): `ultrafast` is sent only
139
+ * when the model's discovered `service_tiers` lists it. `priority`/`scale`
140
+ * are dropped only when that list is non-empty and omits them (codex-rs
141
+ * `service_tier_for_request`); an empty or missing list counts as "not
142
+ * reported" — accounts whose `/models` lists no tiers keep `/fast` — so the
143
+ * provider-level answer stands. `flex` and `default` are never gated.
144
+ * First-party OpenAI takes `ultrafast` as-is. A bare provider string cannot
145
+ * carry the list, so it answers for the provider alone.
135
146
  */
136
147
  export declare function shouldSendServiceTier(serviceTier: ServiceTier | null | undefined, target: Provider | ServiceTierModel | undefined): boolean;
137
148
  /**
@@ -1,4 +1,4 @@
1
- import type { UsageStatus } from "../usage.js";
1
+ import type { UsageAmount, UsageStatus } from "../usage.js";
2
2
  /** Milliseconds in one hour. */
3
3
  export declare const HOUR_MS: number;
4
4
  /** Milliseconds in one day. */
@@ -11,3 +11,15 @@ export declare function parsePositiveTimestamp(value: unknown): number | undefin
11
11
  export declare function parseIsoTimestamp(value: unknown): number | undefined;
12
12
  /** Maps a used fraction to the standard quota status thresholds. */
13
13
  export declare function usageStatus(usedFraction: number | undefined): UsageStatus;
14
+ /**
15
+ * Builds an amount from absolute counters. Without an authoritative
16
+ * `usedFraction`, it derives one from `used / limit` (capped at 1).
17
+ * Undefined fields are omitted rather than emitted as `undefined` keys.
18
+ */
19
+ export declare function buildUsageAmount(args: {
20
+ used: number | undefined;
21
+ limit: number | undefined;
22
+ remaining: number | undefined;
23
+ usedFraction?: number;
24
+ unit: UsageAmount["unit"];
25
+ }): UsageAmount;
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@oh-my-pi/pi-ai",
3
- "version": "18.4.3",
3
+ "version": "18.4.4",
4
4
  "description": "Unified LLM API with automatic model discovery and provider configuration",
5
5
  "keywords": [
6
6
  "ai",
@@ -155,11 +155,11 @@
155
155
  "fmt": "oxfmt --no-error-on-unmatched-pattern 'src/**/*.{ts,tsx}' '{test,bench,examples,scripts}/**/*.ts' '*.ts'"
156
156
  },
157
157
  "dependencies": {
158
- "@oh-my-pi/omptype": "18.4.3",
159
- "@oh-my-pi/pi-catalog": "18.4.3",
160
- "@oh-my-pi/pi-natives": "18.4.3",
161
- "@oh-my-pi/pi-utils": "18.4.3",
162
- "@oh-my-pi/pi-wire": "18.4.3"
158
+ "@oh-my-pi/omptype": "18.4.4",
159
+ "@oh-my-pi/pi-catalog": "18.4.4",
160
+ "@oh-my-pi/pi-natives": "18.4.4",
161
+ "@oh-my-pi/pi-utils": "18.4.4",
162
+ "@oh-my-pi/pi-wire": "18.4.4"
163
163
  },
164
164
  "devDependencies": {
165
165
  "@types/bun": "^1.3.14"
@@ -31,6 +31,7 @@ import type {
31
31
  UsageStaleResponse,
32
32
  } from "./types";
33
33
  import { AUTH_BROKER_CAPABILITIES_HEADER, AUTH_BROKER_CAPABILITY_CODEX_METER_BLOCK_SCOPES } from "./types";
34
+ import { parseGenerationTag } from "./protocol";
34
35
  import {
35
36
  clientUsageReportResponseSchema,
36
37
  clientUsageSummaryResponseSchema,
@@ -112,18 +113,6 @@ export type FetchSnapshotResult =
112
113
  | { status: 200; snapshot: SnapshotResponse; generation: number }
113
114
  | { status: 304; generation: number };
114
115
 
115
- function parseGenerationTag(header: string | null): number | undefined {
116
- if (!header) return undefined;
117
- let value = header.trim();
118
- if (value.startsWith("W/")) value = value.slice(2).trim();
119
- if (value.startsWith('"') && value.endsWith('"') && value.length >= 2) {
120
- value = value.slice(1, -1);
121
- }
122
- const generation = Number(value);
123
- if (!Number.isInteger(generation) || generation < 0) return undefined;
124
- return generation;
125
- }
126
-
127
116
  const DEFAULT_TIMEOUT_MS = 10_000;
128
117
  const DEFAULT_MAX_RETRIES = 1;
129
118
 
@@ -0,0 +1,32 @@
1
+ /**
2
+ * Auth-broker wire-protocol helpers shared by the server and the client store.
3
+ */
4
+ import type { CredentialBlockSnapshot } from "./types";
5
+
6
+ /** Parse a snapshot `ETag` / `If-None-Match` value (`"N"`, `W/"N"`, or bare `N`) into a generation. */
7
+ export function parseGenerationTag(header: string | null): number | undefined {
8
+ if (!header) return undefined;
9
+ let value = header.trim();
10
+ if (value.startsWith("W/")) value = value.slice(2).trim();
11
+ if (value.startsWith('"') && value.endsWith('"') && value.length >= 2) {
12
+ value = value.slice(1, -1);
13
+ }
14
+ const generation = Number(value);
15
+ if (!Number.isInteger(generation) || generation < 0) return undefined;
16
+ return generation;
17
+ }
18
+
19
+ /**
20
+ * Canonical order for a credential's block snapshots. The server and the client
21
+ * store both sort with it, because block lists are compared positionally; the
22
+ * `updatedAtMs` tiebreak keeps otherwise-identical blocks in one stable order.
23
+ */
24
+ export function compareCredentialBlockSnapshots(a: CredentialBlockSnapshot, b: CredentialBlockSnapshot): number {
25
+ const provider = a.providerKey.localeCompare(b.providerKey);
26
+ if (provider !== 0) return provider;
27
+ const scope = a.blockScope.localeCompare(b.blockScope);
28
+ if (scope !== 0) return scope;
29
+ const blockedUntil = a.blockedUntilMs - b.blockedUntilMs;
30
+ if (blockedUntil !== 0) return blockedUntil;
31
+ return (a.updatedAtMs ?? 0) - (b.updatedAtMs ?? 0);
32
+ }
@@ -24,7 +24,9 @@ import * as AIError from "../error";
24
24
  import type { OAuthCredentials } from "../registry/oauth/types";
25
25
  import type { Provider } from "../types";
26
26
  import type { ClientUsageIdentity, ObservedUsageEntry, UsageReport } from "../usage";
27
+ import { raceSignal } from "../auth/abort";
27
28
  import { type AuthBrokerClient, AuthBrokerError, AuthBrokerStreamUnsupportedError } from "./client";
29
+ import { compareCredentialBlockSnapshots } from "./protocol";
28
30
  import type {
29
31
  CredentialBlockSnapshot,
30
32
  RefresherSchedule,
@@ -67,16 +69,6 @@ const BACKGROUND_BACKOFF_MAX_MS = 30_000;
67
69
  /** Idle window after the last foreground store use before background sync parks. */
68
70
  const BACKGROUND_IDLE_MS = 20_000;
69
71
 
70
- function compareCredentialBlockSnapshots(a: CredentialBlockSnapshot, b: CredentialBlockSnapshot): number {
71
- const provider = a.providerKey.localeCompare(b.providerKey);
72
- if (provider !== 0) return provider;
73
- const scope = a.blockScope.localeCompare(b.blockScope);
74
- if (scope !== 0) return scope;
75
- const blockedUntil = a.blockedUntilMs - b.blockedUntilMs;
76
- if (blockedUntil !== 0) return blockedUntil;
77
- return (a.updatedAtMs ?? 0) - (b.updatedAtMs ?? 0);
78
- }
79
-
80
72
  function toCredentialBlockSnapshot(block: StoredCredentialBlock): CredentialBlockSnapshot {
81
73
  return {
82
74
  providerKey: block.providerKey,
@@ -1209,7 +1201,7 @@ export class RemoteAuthCredentialStore implements AuthCredentialStore {
1209
1201
  */
1210
1202
  async fetchUsageReports(signal?: AbortSignal): Promise<UsageReport[] | null> {
1211
1203
  this.#noteActivity();
1212
- const reports = await this.#raceWithSignal(this.#loadUsageReports(), signal);
1204
+ const reports = await raceSignal(this.#loadUsageReports(), signal, "auth-broker request aborted");
1213
1205
  if (!reports) return null;
1214
1206
  return this.#filterUsageReports(this.#applyUsageOverlays(reports));
1215
1207
  }
@@ -1229,7 +1221,7 @@ export class RemoteAuthCredentialStore implements AuthCredentialStore {
1229
1221
  signal?: AbortSignal,
1230
1222
  ): Promise<UsageReport | null> {
1231
1223
  this.#noteActivity();
1232
- const reports = await this.#raceWithSignal(this.#loadUsageReports(), signal);
1224
+ const reports = await raceSignal(this.#loadUsageReports(), signal, "auth-broker request aborted");
1233
1225
  const visibleReports = reports ? this.#filterUsageReports(reports) : null;
1234
1226
  const matched = visibleReports ? matchUsageReport(visibleReports, provider, credential) : null;
1235
1227
  const overlay = this.#getActiveUsageOverlay(provider, credential);
@@ -1310,34 +1302,6 @@ export class RemoteAuthCredentialStore implements AuthCredentialStore {
1310
1302
  return merged;
1311
1303
  }
1312
1304
 
1313
- /**
1314
- * Reject the awaited promise when the caller's signal aborts, without
1315
- * affecting the shared upstream fetch. Used to give each caller their
1316
- * own cancel without one caller's abort cascading into a peer's in-flight
1317
- * request through the single-flight `#usageInflight`.
1318
- */
1319
- #raceWithSignal<T>(promise: Promise<T>, signal?: AbortSignal): Promise<T> {
1320
- if (!signal) return promise;
1321
- if (signal.aborted) return Promise.reject(new AIError.AbortError("auth-broker request aborted"));
1322
- return new Promise<T>((resolve, reject) => {
1323
- const onAbort = (): void => {
1324
- signal.removeEventListener("abort", onAbort);
1325
- reject(new AIError.AbortError("auth-broker request aborted"));
1326
- };
1327
- signal.addEventListener("abort", onAbort, { once: true });
1328
- promise.then(
1329
- value => {
1330
- signal.removeEventListener("abort", onAbort);
1331
- resolve(value);
1332
- },
1333
- err => {
1334
- signal.removeEventListener("abort", onAbort);
1335
- reject(err);
1336
- },
1337
- );
1338
- });
1339
- }
1340
-
1341
1305
  #replaceBrokerUsageAccounts(entries: readonly SnapshotEntry[]): void {
1342
1306
  this.#brokerUsageProviderByCredentialId.clear();
1343
1307
  this.#brokerUsageAccountCounts.clear();
@@ -41,6 +41,7 @@ import {
41
41
  DEFAULT_SERVER_IDLE_TIMEOUT_S,
42
42
  DEFAULT_STREAM_KEEPALIVE_MS,
43
43
  } from "./types";
44
+ import { compareCredentialBlockSnapshots, parseGenerationTag } from "./protocol";
44
45
  import {
45
46
  clientUsageReportRequestSchema,
46
47
  credentialBlockDeleteRequestSchema,
@@ -165,18 +166,6 @@ function snapshotHeaders(generation: number): Record<string, string> {
165
166
  };
166
167
  }
167
168
 
168
- function parseGenerationTag(header: string | null): number | undefined {
169
- if (!header) return undefined;
170
- let value = header.trim();
171
- if (value.startsWith("W/")) value = value.slice(2).trim();
172
- if (value.startsWith('"') && value.endsWith('"') && value.length >= 2) {
173
- value = value.slice(1, -1);
174
- }
175
- const generation = Number(value);
176
- if (!Number.isInteger(generation) || generation < 0) return undefined;
177
- return generation;
178
- }
179
-
180
169
  function parseWaitMs(url: URL): number {
181
170
  const raw = url.searchParams.get("wait");
182
171
  if (raw === null) return 0;
@@ -316,14 +305,6 @@ function computeRotatesInMs(
316
305
  return Math.max(0, rotatesAt - serverNowMs);
317
306
  }
318
307
 
319
- function compareCredentialBlockSnapshots(a: CredentialBlockSnapshot, b: CredentialBlockSnapshot): number {
320
- const provider = a.providerKey.localeCompare(b.providerKey);
321
- if (provider !== 0) return provider;
322
- const scope = a.blockScope.localeCompare(b.blockScope);
323
- if (scope !== 0) return scope;
324
- return a.blockedUntilMs - b.blockedUntilMs;
325
- }
326
-
327
308
  const CODEX_BLOCK_PROVIDER_KEY = "openai-codex:oauth";
328
309
  const CODEX_LEGACY_PROJECTED_BLOCK_SCOPES = new Set(["chat", "spark", "shared"]);
329
310
 
@@ -9,6 +9,7 @@
9
9
  import * as fs from "node:fs/promises";
10
10
  import * as path from "node:path";
11
11
  import { isEnoent, logger } from "@oh-my-pi/pi-utils";
12
+ import { asStrict } from "../providers/aws-sigv4";
12
13
  import type { SnapshotResponse } from "./types";
13
14
 
14
15
  const MAGIC = new Uint8Array([0x4f, 0x4d, 0x50, 0x53]); // "OMPS"
@@ -210,15 +211,6 @@ async function deriveAesKey(token: string, usages: Array<"encrypt" | "decrypt">)
210
211
  return globalThis.crypto.subtle.importKey("raw", digest, AES_ALGORITHM, false, usages);
211
212
  }
212
213
 
213
- function asStrict(bytes: Uint8Array): Uint8Array<ArrayBuffer> {
214
- if (bytes.buffer instanceof ArrayBuffer && bytes.byteOffset === 0 && bytes.byteLength === bytes.buffer.byteLength) {
215
- return bytes as Uint8Array<ArrayBuffer>;
216
- }
217
- const copy = new Uint8Array(bytes.byteLength);
218
- copy.set(bytes);
219
- return copy;
220
- }
221
-
222
214
  function randomHex(byteLength: number): string {
223
215
  const bytes = new Uint8Array(byteLength);
224
216
  globalThis.crypto.getRandomValues(bytes);
@@ -1,14 +1,8 @@
1
- import { parseJsonWithRepair } from "@oh-my-pi/pi-utils";
1
+ import { escapeXmlText, parseJsonWithRepair } from "@oh-my-pi/pi-utils";
2
2
  import type { Message, ToolCall } from "../types";
3
3
  import dialectPrompt from "./anthropic.md" with { type: "text" };
4
- import { buildArgShapes, buildStringArgsResolver, mintToolCallId, type ToolArgShape } from "./coercion";
5
- import {
6
- escapeXmlAttr,
7
- escapeXmlText,
8
- renderDelimitedThinking,
9
- renderLegacyTextTranscript,
10
- stringifyJson,
11
- } from "./rendering";
4
+ import { buildArgShapes, buildStringArgsResolver, mintToolCallId } from "./coercion";
5
+ import { renderDelimitedThinking, renderInvoke, renderInvokes, renderLegacyTextTranscript } from "./rendering";
12
6
  import type {
13
7
  DialectDefinition,
14
8
  DialectRenderOptions,
@@ -578,22 +572,6 @@ function renderTranscript(messages: readonly Message[], options: DialectRenderOp
578
572
  });
579
573
  }
580
574
 
581
- function renderInvoke(call: ToolCall, shape: ToolArgShape | undefined): string {
582
- let body = `<invoke name="${escapeXmlAttr(call.name)}">`;
583
- for (const key in call.arguments) {
584
- const value = call.arguments[key];
585
- const isString = shape?.stringArgs.has(key) === true;
586
- const rendered = isString && typeof value === "string" ? value : stringifyJson(value);
587
- body += `<parameter name="${escapeXmlAttr(key)}">${rendered}</parameter>`;
588
- }
589
- return `${body}</invoke>`;
590
- }
591
-
592
- function renderInvokes(calls: readonly ToolCall[], tools: NonNullable<DialectRenderOptions["tools"]>): string {
593
- const shapes = buildArgShapes(tools);
594
- return calls.map(call => renderInvoke(call, shapes.get(call.name))).join("\n");
595
- }
596
-
597
575
  const definition: DialectDefinition = {
598
576
  dialect: "anthropic",
599
577
  prompt: dialectPrompt,