@oh-my-pi/pi-ai 18.2.7 → 18.2.9

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (133) hide show
  1. package/CHANGELOG.md +49 -18
  2. package/dist/types/auth-gateway/dispatch.d.ts +80 -0
  3. package/dist/types/auth-gateway/http.d.ts +5 -6
  4. package/dist/types/auth-gateway/index.d.ts +1 -0
  5. package/dist/types/auth-gateway/routes/embeddings.d.ts +3 -0
  6. package/dist/types/auth-gateway/routes/images.d.ts +3 -0
  7. package/dist/types/auth-gateway/routes/rerank.d.ts +3 -0
  8. package/dist/types/auth-gateway/routes/speech.d.ts +2 -0
  9. package/dist/types/auth-gateway/routes/systemone.d.ts +2 -0
  10. package/dist/types/auth-gateway/routes/transcriptions.d.ts +3 -0
  11. package/dist/types/auth-gateway/routes/video.d.ts +7 -0
  12. package/dist/types/auth-gateway/server.d.ts +10 -16
  13. package/dist/types/auth-gateway/types.d.ts +5 -0
  14. package/dist/types/auth-storage.d.ts +35 -32
  15. package/dist/types/embeddings/index.d.ts +7 -0
  16. package/dist/types/embeddings/openai-embeddings.d.ts +15 -0
  17. package/dist/types/embeddings/types.d.ts +15 -0
  18. package/dist/types/images/google-antigravity.d.ts +9 -0
  19. package/dist/types/images/google-generative-ai.d.ts +3 -0
  20. package/dist/types/images/index.d.ts +14 -0
  21. package/dist/types/images/openai-hosted.d.ts +3 -0
  22. package/dist/types/images/openai-images.d.ts +5 -0
  23. package/dist/types/images/openrouter-images.d.ts +3 -0
  24. package/dist/types/images/shared.d.ts +34 -0
  25. package/dist/types/images/types.d.ts +31 -0
  26. package/dist/types/index.d.ts +8 -1
  27. package/dist/types/judgment/typesafe.d.ts +2 -0
  28. package/dist/types/providers/amazon-bedrock.d.ts +7 -0
  29. package/dist/types/providers/claude-code-fingerprint.d.ts +22 -3
  30. package/dist/types/providers/embeddings-server.d.ts +32 -0
  31. package/dist/types/providers/google-gemini-cli.d.ts +0 -2
  32. package/dist/types/providers/images-server.d.ts +22 -0
  33. package/dist/types/providers/openai-chat-server-schema.d.ts +2 -2
  34. package/dist/types/providers/rerank-server.d.ts +35 -0
  35. package/dist/types/providers/speech-server.d.ts +8 -0
  36. package/dist/types/providers/systemone-server.d.ts +26 -0
  37. package/dist/types/providers/transcriptions-server.d.ts +32 -0
  38. package/dist/types/providers/video-server.d.ts +39 -0
  39. package/dist/types/rerank/index.d.ts +7 -0
  40. package/dist/types/rerank/openrouter-rerank.d.ts +15 -0
  41. package/dist/types/rerank/types.d.ts +17 -0
  42. package/dist/types/speech/index.d.ts +13 -0
  43. package/dist/types/speech/openai-speech.d.ts +3 -0
  44. package/dist/types/speech/transport.d.ts +7 -0
  45. package/dist/types/speech/types.d.ts +24 -0
  46. package/dist/types/speech/xai-tts.d.ts +7 -0
  47. package/dist/types/transcription/index.d.ts +7 -0
  48. package/dist/types/transcription/openai-transcriptions.d.ts +15 -0
  49. package/dist/types/transcription/types.d.ts +41 -0
  50. package/dist/types/usage/claude-api.d.ts +22 -0
  51. package/dist/types/usage/claude-reset.d.ts +44 -0
  52. package/dist/types/usage.d.ts +111 -5
  53. package/dist/types/utils/schema/json-schema-validator.d.ts +5 -2
  54. package/dist/types/utils/tool-call-loop-guard.d.ts +1 -1
  55. package/dist/types/video/index.d.ts +11 -0
  56. package/dist/types/video/openrouter-video.d.ts +19 -0
  57. package/dist/types/video/types.d.ts +62 -0
  58. package/package.json +30 -6
  59. package/src/auth/sqlite-credential-store.ts +44 -1
  60. package/src/auth-broker/remote-store.ts +6 -6
  61. package/src/auth-broker/wire-schemas.ts +14 -0
  62. package/src/auth-gateway/dispatch.ts +273 -0
  63. package/src/auth-gateway/http.ts +6 -7
  64. package/src/auth-gateway/index.ts +1 -0
  65. package/src/auth-gateway/routes/embeddings.ts +98 -0
  66. package/src/auth-gateway/routes/images.ts +131 -0
  67. package/src/auth-gateway/routes/rerank.ts +87 -0
  68. package/src/auth-gateway/routes/speech.ts +101 -0
  69. package/src/auth-gateway/routes/systemone.ts +116 -0
  70. package/src/auth-gateway/routes/transcriptions.ts +98 -0
  71. package/src/auth-gateway/routes/video.ts +243 -0
  72. package/src/auth-gateway/server.ts +127 -260
  73. package/src/auth-gateway/types.ts +5 -0
  74. package/src/auth-storage.ts +263 -140
  75. package/src/embeddings/index.ts +17 -0
  76. package/src/embeddings/openai-embeddings.ts +141 -0
  77. package/src/embeddings/types.ts +14 -0
  78. package/src/error/flags.ts +10 -0
  79. package/src/error/rate-limit.ts +1 -1
  80. package/src/images/google-antigravity.ts +180 -0
  81. package/src/images/google-generative-ai.ts +92 -0
  82. package/src/images/index.ts +59 -0
  83. package/src/images/openai-hosted.ts +185 -0
  84. package/src/images/openai-images.ts +110 -0
  85. package/src/images/openrouter-images.ts +33 -0
  86. package/src/images/shared.ts +193 -0
  87. package/src/images/types.ts +36 -0
  88. package/src/index.ts +8 -1
  89. package/src/judgment/typesafe.ts +5 -0
  90. package/src/providers/amazon-bedrock.ts +55 -5
  91. package/src/providers/anthropic.ts +45 -11
  92. package/src/providers/aws-credentials.ts +124 -11
  93. package/src/providers/claude-code-fingerprint.ts +55 -3
  94. package/src/providers/embeddings-server.ts +151 -0
  95. package/src/providers/gitlab-duo.ts +20 -4
  96. package/src/providers/google-gemini-cli.ts +0 -8
  97. package/src/providers/google-shared.ts +1 -18
  98. package/src/providers/images-server.ts +159 -0
  99. package/src/providers/openai-chat-server-schema.ts +1 -1
  100. package/src/providers/openai-chat-server.ts +3 -1
  101. package/src/providers/openai-codex-responses.ts +27 -5
  102. package/src/providers/openai-completions.ts +122 -19
  103. package/src/providers/pi-native-server.ts +1 -0
  104. package/src/providers/rerank-server.ts +166 -0
  105. package/src/providers/speech-server.ts +53 -0
  106. package/src/providers/systemone-server.ts +73 -0
  107. package/src/providers/transcriptions-server.ts +243 -0
  108. package/src/providers/video-server.ts +286 -0
  109. package/src/registry/oauth/anthropic.ts +2 -3
  110. package/src/rerank/index.ts +13 -0
  111. package/src/rerank/openrouter-rerank.ts +136 -0
  112. package/src/rerank/types.ts +20 -0
  113. package/src/speech/index.ts +35 -0
  114. package/src/speech/openai-speech.ts +26 -0
  115. package/src/speech/transport.ts +66 -0
  116. package/src/speech/types.ts +37 -0
  117. package/src/speech/xai-tts.ts +41 -0
  118. package/src/stream.ts +13 -4
  119. package/src/transcription/index.ts +17 -0
  120. package/src/transcription/openai-transcriptions.ts +133 -0
  121. package/src/transcription/types.ts +46 -0
  122. package/src/usage/alibaba-token-plan.ts +7 -1
  123. package/src/usage/claude-api.ts +66 -0
  124. package/src/usage/claude-reset.ts +638 -0
  125. package/src/usage/claude.ts +37 -59
  126. package/src/usage/kimi.ts +32 -1
  127. package/src/usage.ts +52 -5
  128. package/src/utils/schema/json-schema-validator.ts +23 -10
  129. package/src/utils/tool-call-loop-guard.ts +2 -2
  130. package/src/utils/validation.ts +145 -50
  131. package/src/video/index.ts +34 -0
  132. package/src/video/openrouter-video.ts +210 -0
  133. package/src/video/types.ts +72 -0
package/CHANGELOG.md CHANGED
@@ -2,6 +2,54 @@
2
2
 
3
3
  ## [Unreleased]
4
4
 
5
+ ## [18.2.9] - 2026-09-22
6
+
7
+ ### Added
8
+
9
+ - Added Claude saved-reset discovery and redemption, including session-only resets, grant eligibility, expiry, and safe retry handling.
10
+
11
+ ### Fixed
12
+
13
+ - Fixed custom OpenAI-compatible extension streamers failing when no compatibility configuration was provided.
14
+ - Fixed local provider sign-in for LM Studio, llama.cpp, and vLLM so an empty API key is not treated as successful authentication.
15
+ - Fixed Vercel AI Gateway models backed by non-Anthropic providers failing tool calls because of strict schema validation; requests now retry with compatible non-strict tool handling.
16
+ - Improved AWS Bedrock authentication by refreshing expired AWS SSO sessions automatically, reducing the need to run `aws sso login` again.
17
+ - Fixed Bedrock tool-enabled requests when tool descriptions are included in the system prompt.
18
+ - Improved Alibaba Token Plan (Beijing) quota reporting across workspaces and made gateway rejection codes visible in error logs.
19
+ - Fixed the tool-call loop guard so repeated identical calls continue to be redirected after the detection threshold is reached.
20
+ - Fixed valid required null values inside tool argument unions being removed before dispatch ([#12523](https://github.com/can1357/oh-my-pi/pull/12523) by [@cswenor](https://github.com/cswenor)).
21
+ - Signing in to a local provider (lm-studio, llama.cpp, vllm) with an empty key paste no longer reports the provider as logged in while its requests go out unauthenticated. ([#12436](https://github.com/can1357/oh-my-pi/pull/12436) by [@xiechimon](https://github.com/xiechimon))
22
+ - Fixed every turn failing with `400 Invalid schema for function '<tool>' … Missing '<param>'` on Vercel AI Gateway models served from a non-Anthropic upstream (e.g. `openai/gpt-5.6-sol`): the translated strict-tool rejection now triggers the existing non-strict retry instead of failing the turn ([#12760](https://github.com/can1357/oh-my-pi/pull/12760) by [@primitive-type](https://github.com/primitive-type)).
23
+ - Expired AWS SSO access tokens are now refreshed via the SSO OIDC `refresh_token` grant instead of failing with `sso-token-expired`, so Bedrock profiles keep working between `aws sso login` runs the same way the AWS CLI does ([#12736](https://github.com/can1357/oh-my-pi/pull/12736) by [@nwbb](https://github.com/nwbb)).
24
+ - Fixed Bedrock rejecting tool-enabled requests when tool descriptions are inlined into the system prompt ([#12732](https://github.com/can1357/oh-my-pi/pull/12732) by [@mustafaabidali](https://github.com/mustafaabidali)).
25
+ - Alibaba Token Plan (Beijing) quota reporting no longer pins requests to a single workspace, and HTTP-200 gateway rejections now log their error code ([#12395](https://github.com/can1357/oh-my-pi/pull/12395) by [@Dante-dan](https://github.com/Dante-dan)).
26
+ - The tool-call loop guard keeps redirecting when a model continues the same identical call past the detection threshold instead of firing only once ([#12709](https://github.com/can1357/oh-my-pi/pull/12709) by [@F0Rextasy](https://github.com/F0Rextasy)).
27
+ - Fixed OpenAI-compatible Gemini gateways losing message-level thought signatures when replaying tool-call history.
28
+ - Added support for explicitly disabling reasoning with `reasoning_effort: "none"` through Chat Completions authentication gateways.
29
+ - Improved Bedrock resilience by retrying transient in-stream internal server, service unavailable, and throttling errors.
30
+ - Fixed Anthropic prompt-cache keep-alive refreshes for turns that used forced tool selection.
31
+ - Fixed auth-broker usage reports incorrectly sharing usage limits between Team members with shared workspace and organization identifiers.
32
+ - Fixed OpenAI Codex requests hanging when an error response body is delayed.
33
+ - Fixed Kimi usage reporting so monthly totals and code quotas are shown alongside the five-hour usage window.
34
+ - Bedrock no longer sends provider-invalid payloads when an errored tool result contains an image; the image is hoisted into a sibling block ([#12865](https://github.com/can1357/oh-my-pi/pull/12865) by [@roboomp](https://github.com/roboomp)).
35
+ - Gemini, Vertex, and Cloud Code Assist requests no longer include the unsupported `minP`/`repetitionPenalty` sampling fields, which caused 400s when set globally ([#12850](https://github.com/can1357/oh-my-pi/pull/12850) by [@roboomp](https://github.com/roboomp)).
36
+
37
+ ## [18.2.8] - 2026-09-21
38
+
39
+ ### Added
40
+
41
+ - Added support for text embeddings, document reranking, video generation, image generation across multiple providers, audio speech synthesis, and audio transcription services.
42
+ - Added support for the System One judgment API, including configurable request headers for proxy routing and custom authentication.
43
+
44
+ ### Changed
45
+
46
+ - Updated API response cost reporting to use aggregate usage totals.
47
+ - Model list responses now optionally include a model kind.
48
+
49
+ ### Fixed
50
+
51
+ - Fixed detection of Claude usage-limit errors.
52
+
5
53
  ## [18.2.7] - 2026-09-21
6
54
 
7
55
  ### Breaking Changes
@@ -2239,21 +2287,4 @@
2239
2287
  - Made the openai-completions non-strict retry reachable for `"mixed"` strict mode (previously gated to `all_strict`, i.e. Cerebras only) and taught it to recognize upstream tool-schema validation 400s (`Invalid tool parameters schema …`, `Invalid schema for function …`). A matching rejection now retries the request with base (non-strict) schemas and persists `strictToolsDisabled` on the provider session, so later requests skip the doomed strict attempt instead of paying a 400 + retry round-trip each turn. ([#2270](https://github.com/can1357/oh-my-pi/issues/2270))
2240
2288
  - Cross-model `anthropic-messages → anthropic-messages` continuations now preserve prior assistant turns' reasoning chains end-to-end: every prior `thinking`/`redactedThinking` block survives (not just the latest surviving assistant), and third-party ↔ third-party replays keep their signatures intact so the reasoning chain stays signed for the next turn. Signatures are stripped (and any `redacted_thinking` sibling without a native landing spot is dropped) only when an official Anthropic endpoint is on either end of the replay — official Anthropic cryptographically binds reasoning signatures to its key+session+model, while compatible reasoning endpoints (Z.AI, DeepSeek, custom anthropic-messages providers configured via `models.yaml`) treat them as opaque continuation hints. Source-side official detection uses the canonical catalog provider id `"anthropic"` (assistant messages carry no `baseUrl`); target-side detection reuses the baked `compat.officialEndpoint` flag. Latest-turn byte-for-byte behavior (Anthropic's "thinking blocks in the latest assistant message cannot be modified" rule) and existing aborted/errored last-block sanitization are unchanged. ([#2257](https://github.com/can1357/oh-my-pi/issues/2257), [#2265](https://github.com/can1357/oh-my-pi/issues/2265))
2241
2289
 
2242
- ## [15.10.12] - 2026-06-10
2243
-
2244
- ### Added
2245
-
2246
- - Added `antigravityRankingStrategy` and registered it for `google-antigravity` in `DEFAULT_RANKING_STRATEGIES`, so new sessions are routed to OAuth credentials with quota headroom for the requested model backend (lowest relevant `remainingFraction` counter as the sole ranked window, 24h `windowDefaults` matching `daily-cloudcode-pa.googleapis.com` resets). Without it, the existing `antigravityUsageProvider` data never reached credential selection. ([#2198](https://github.com/can1357/oh-my-pi/issues/2198))
2247
-
2248
- ### Changed
2249
-
2250
- - Updated MiniMax and MiniMax Token Plan defaults to `MiniMax-M3` and refreshed Token Plan login copy/links ([#1725](https://github.com/can1357/oh-my-pi/issues/1725)).
2251
-
2252
- ### Fixed
2253
-
2254
- - Fixed OpenAI Responses and Azure OpenAI Responses streams silently surfacing incomplete output as successful when a custom/proxy provider drops the connection without sending a terminal `response.completed`/`response.incomplete` event. Both providers now detect premature stream closure and throw with `stopReason: "error"` ([#2184](https://github.com/can1357/oh-my-pi/pull/2184))
2255
- - Fixed `isUsageLimitError` missing Antigravity / Cloud Code Assist's `Individual quota reached` 429 phrasing. The `USAGE_LIMIT_PATTERN` only knew `quota.?exceeded` / `limit_reached`, so `auth-retry` and `AuthStorage.markUsageLimitReached` treated the response as a terminal provider error and pinned sessions to the exhausted OAuth account instead of rotating to a sibling credential. The pattern now also matches `quota.?reached`. ([#2198](https://github.com/can1357/oh-my-pi/issues/2198))
2256
- - Scoped Antigravity usage blocking and ranking by model family (`gemini-*`/`gemma-*` → Google, `claude-*` → Anthropic, `gpt-*`/`openai/*` → OpenAI), so an exhausted Gemini counter no longer makes a healthy Claude/OpenAI Antigravity credential unavailable until reset. ([#2198](https://github.com/can1357/oh-my-pi/issues/2198))
2257
- - Fixed no-model Antigravity credential lookups (e.g. image-provider discovery) inheriting provider-wide exhaustion: `scopeLimits` now returns no limits without a concrete backend counter, and `blockScope` always returns a counter scope so missing model context can never fall through to AuthStorage's provider-wide block bucket. ([#2198](https://github.com/can1357/oh-my-pi/issues/2198))
2258
-
2259
- Older entries are archived in [packages/ai/CHANGELOG.md@8a9097246135](https://github.com/can1357/oh-my-pi/blob/8a9097246135bd572ff96fb552121fe1194d2906/packages/ai/CHANGELOG.md).
2290
+ Older entries are archived in [packages/ai/CHANGELOG.md@1f7329fc2c7c](https://github.com/can1357/oh-my-pi/blob/1f7329fc2c7c366b38731738e0db9c170f9bb348/packages/ai/CHANGELOG.md).
@@ -0,0 +1,80 @@
1
+ import type { ApiKeyResolver } from "../auth-retry.js";
2
+ import type { AuthStorage } from "../auth-storage.js";
3
+ import { type GatewayErrorClassification } from "../error/gateway.js";
4
+ import type { Api, FetchImpl, Model, Usage } from "../types.js";
5
+ import type { ClientUsageIdentity } from "../usage.js";
6
+ import type { AuthGatewayServerOptions } from "./types.js";
7
+ export type ModelResolver = (modelId: string) => Model<Api> | undefined;
8
+ export interface AuthGatewayBootOptions extends AuthGatewayServerOptions {
9
+ /** Source of credentials. Caller wires this to a broker-backed AuthStorage. */
10
+ storage: AuthStorage;
11
+ /**
12
+ * Resolve a client-requested model id to a pi-ai Model. Caller supplies
13
+ * this from a ModelRegistry (lives in `coding-agent` to avoid an inverse
14
+ * dependency in `pi-ai`).
15
+ */
16
+ resolveModel: ModelResolver;
17
+ /** Optional supplier for `/v1/models` listing. Returns the full model array. */
18
+ listModels?: () => Iterable<Model<Api>>;
19
+ /** Upstream transport for every provider call; defaults to global `fetch`. Test seam. */
20
+ fetch?: FetchImpl;
21
+ }
22
+ /**
23
+ * The client's own session key, or `undefined` when it sent none. A blank key
24
+ * counts as none: honouring it would collapse every caller that sends an empty
25
+ * key into one shared credential-sticky, prefix-cache and provider-session
26
+ * bucket.
27
+ */
28
+ export declare function normalizeClientSessionKey(clientKey: string | undefined): string | undefined;
29
+ /**
30
+ * Stable identity of the account a request's credential belongs to.
31
+ *
32
+ * `markUsageLimitReached` and the auth-retry resolver switch a session to a
33
+ * sibling credential, so the provider state retained for that session can
34
+ * outlive the account that taught it. OAuth rows expose an account id / email
35
+ * that survives token refresh — fingerprinting the bearer instead would look
36
+ * like a rotation every time a token refreshes and discard the retained
37
+ * lessons for nothing. Key-based rows fall back to a hash of the key, never
38
+ * the key itself: this value is held for the lifetime of the entry.
39
+ */
40
+ export declare function resolveGatewayAccount(storage: AuthStorage, provider: string, sessionId: string, apiKey: string): string;
41
+ /**
42
+ * Resolve the credential for one request from broker-backed storage.
43
+ *
44
+ * pi-ai clients never consult `AuthStorage`; the gateway resolves the bearer
45
+ * (an OAuth access token refreshed through the broker when needed) and hands
46
+ * it to the client. Returns the key, or the error classification the route
47
+ * should encode in its own envelope: storage failures map through
48
+ * {@link classifyGatewayError}, a provider without any credential is a 401.
49
+ */
50
+ export declare function resolveGatewayApiKey(storage: AuthStorage, model: Model<Api>, sessionId: string, signal: AbortSignal, peer: string): Promise<string | GatewayErrorClassification>;
51
+ /**
52
+ * Build the {@link ApiKeyResolver} handed to a pi-ai client for a gateway
53
+ * request. Drives the central a/b/c auth-retry policy server-side:
54
+ *
55
+ * - initial resolve → the credential already resolved for this request.
56
+ * - step (b) `!lastChance` → force-refresh the SAME session-sticky credential
57
+ * (a peer/broker may have rotated its token out from under our cached copy).
58
+ * - step (c) `lastChance` → {@link refreshGatewayApiKeyAfterAuthError} switches
59
+ * to a sibling (usage-limit block vs credential invalidation by error class).
60
+ *
61
+ * `lastKey` tracks the most recent bearer so the switch step invalidates the
62
+ * credential that actually failed. `onResolvedKey` observes every rotation;
63
+ * routes that retain provider session state use it to re-key the account
64
+ * lease, one-shot routes pass `undefined`.
65
+ */
66
+ export declare function buildGatewayApiKeyResolver(storage: AuthStorage, model: Model<Api>, sessionId: string, initialKey: string, requestSignal: AbortSignal, format: string, peer: string, onResolvedKey?: (apiKey: string) => void): ApiKeyResolver;
67
+ /**
68
+ * Attribute one settled upstream request to the originating client via the
69
+ * broker's observed-usage channel (`AuthStorage.recordObservedUsage`, batched
70
+ * by the remote store). Error/aborted turns still record — the provider
71
+ * billed whatever tokens the partial turn consumed; zero-usage results
72
+ * (pre-flight failures) are skipped. `at` defaults to now.
73
+ */
74
+ export declare function recordGatewayUsage(storage: AuthStorage, model: Model<Api>, client: ClientUsageIdentity, usage: Usage, at?: number): void;
75
+ /**
76
+ * An `AbortController` that follows the inbound request's abort signal. Routes
77
+ * abort it themselves when the response body is cancelled mid-stream, which
78
+ * `req.signal` alone does not observe.
79
+ */
80
+ export declare function mirrorRequestAbort(req: Request): AbortController;
@@ -1,4 +1,4 @@
1
- import type { Api, AssistantMessage, Model } from "../types.js";
1
+ import type { Api, Model } from "../types.js";
2
2
  import type { ClientUsageIdentity } from "../usage.js";
3
3
  export declare function json(status: number, body: unknown, headers?: Record<string, string>): Response;
4
4
  /**
@@ -7,14 +7,13 @@ export declare function json(status: number, body: unknown, headers?: Record<str
7
7
  * `request-id` (surfaced as `_request_id` by the OpenAI and Anthropic SDKs,
8
8
  * matches the gateway log line), LiteLLM's model-resolution and cost headers,
9
9
  * and OpenAI's `openai-processing-ms`. Model/request-id headers are always
10
- * present; `message` — the final assistant message, available only on
11
- * non-streaming responses — adds the computed cost, and `startedAt` the wall
12
- * time. Streaming responses send headers before usage exists, so they carry
13
- * only the identity headers.
10
+ * present; `costUsd` — known only once a non-streaming response has settled —
11
+ * adds the computed cost, and `startedAt` the wall time. Streaming responses
12
+ * send headers before usage exists, so they carry only the identity headers.
14
13
  */
15
14
  export declare function gatewayResponseHeaders(model: Model<Api>, info: {
16
15
  requestId: string;
17
- message?: AssistantMessage;
16
+ costUsd?: number;
18
17
  startedAt?: number;
19
18
  }): Record<string, string>;
20
19
  export declare function resolvePeer(req: Request): string;
@@ -1,3 +1,4 @@
1
+ export * from "./dispatch.js";
1
2
  export * from "./http.js";
2
3
  export * from "./session-state.js";
3
4
  export * from "./server.js";
@@ -0,0 +1,3 @@
1
+ import { type AuthGatewayBootOptions } from "../dispatch.js";
2
+ /** OpenAI-compatible `POST /v1/embeddings` gateway handler. */
3
+ export declare function handleEmbeddings(bootOpts: AuthGatewayBootOptions, req: Request, peer: string): Promise<Response>;
@@ -0,0 +1,3 @@
1
+ import { type AuthGatewayBootOptions } from "../dispatch.js";
2
+ export declare function handleImageGenerations(bootOpts: AuthGatewayBootOptions, req: Request, peer: string): Promise<Response>;
3
+ export declare function handleImageEdits(bootOpts: AuthGatewayBootOptions, req: Request, peer: string): Promise<Response>;
@@ -0,0 +1,3 @@
1
+ import { type AuthGatewayBootOptions } from "../dispatch.js";
2
+ /** OpenRouter-compatible `POST /v1/rerank` gateway handler. */
3
+ export declare function handleRerank(bootOpts: AuthGatewayBootOptions, req: Request, peer: string): Promise<Response>;
@@ -0,0 +1,2 @@
1
+ import { type AuthGatewayBootOptions } from "../dispatch.js";
2
+ export declare function handleSpeech(bootOpts: AuthGatewayBootOptions, req: Request, peer: string): Promise<Response>;
@@ -0,0 +1,2 @@
1
+ import { type AuthGatewayBootOptions } from "../dispatch.js";
2
+ export declare function handleSystemOne(bootOpts: AuthGatewayBootOptions, req: Request, peer: string): Promise<Response>;
@@ -0,0 +1,3 @@
1
+ import { type AuthGatewayBootOptions } from "../dispatch.js";
2
+ /** OpenAI-compatible `POST /v1/audio/transcriptions` gateway handler. */
3
+ export declare function handleTranscriptions(bootOpts: AuthGatewayBootOptions, req: Request, peer: string): Promise<Response>;
@@ -0,0 +1,7 @@
1
+ import { type AuthGatewayBootOptions } from "../dispatch.js";
2
+ /** OpenRouter-compatible `POST /v1/videos` asynchronous video submit handler. */
3
+ export declare function handleVideoSubmit(bootOpts: AuthGatewayBootOptions, req: Request, peer: string): Promise<Response>;
4
+ /** OpenRouter-compatible `GET /v1/videos/:id` asynchronous video poll handler. */
5
+ export declare function handleVideoPoll(bootOpts: AuthGatewayBootOptions, req: Request, peer: string, gatewayId: string): Promise<Response>;
6
+ /** OpenRouter-compatible `GET /v1/videos/:id/content` streaming video content handler. */
7
+ export declare function handleVideoContent(bootOpts: AuthGatewayBootOptions, req: Request, peer: string, gatewayId: string): Promise<Response>;
@@ -16,21 +16,15 @@
16
16
  * POST /v1/chat/completions → OpenAI chat-completions in/out
17
17
  * POST /v1/messages → Anthropic messages in/out
18
18
  * POST /v1/responses → OpenAI Responses in/out
19
+ * POST /v1/pi/stream → native pi-ai stream in/out
20
+ * POST /v1/systemone | /alpha/decisions → TypeSafe System One judgments (routes/systemone)
21
+ * POST /v1/images[/generations|/edits] → image generation, OpenAI/OpenRouter wire (routes/images)
22
+ * POST /v1/audio/speech → text-to-speech, raw audio out (routes/speech)
23
+ * POST /v1/audio/transcriptions → speech-to-text, multipart or JSON base64 in (routes/transcriptions)
24
+ *
25
+ * Chat routes live in this file; every other modality is a `routes/*` module
26
+ * built on the shared plumbing in `dispatch.ts`.
19
27
  */
20
- import type { AuthStorage } from "../auth-storage.js";
21
- import type { Api, Model } from "../types.js";
22
- import type { AuthGatewayServerHandle, AuthGatewayServerOptions } from "./types.js";
23
- export type ModelResolver = (modelId: string) => Model<Api> | undefined;
24
- export interface AuthGatewayBootOptions extends AuthGatewayServerOptions {
25
- /** Source of credentials. Caller wires this to a broker-backed AuthStorage. */
26
- storage: AuthStorage;
27
- /**
28
- * Resolve a client-requested model id to a pi-ai Model. Caller supplies
29
- * this from a ModelRegistry (lives in `coding-agent` to avoid an inverse
30
- * dependency in `pi-ai`).
31
- */
32
- resolveModel: ModelResolver;
33
- /** Optional supplier for `/v1/models` listing. Returns the full model array. */
34
- listModels?: () => Iterable<Model<Api>>;
35
- }
28
+ import { type AuthGatewayBootOptions } from "./dispatch.js";
29
+ import type { AuthGatewayServerHandle } from "./types.js";
36
30
  export declare function startAuthGateway(opts: AuthGatewayBootOptions): AuthGatewayServerHandle;
@@ -47,6 +47,11 @@ export interface AuthGatewayParsedRequestOptions {
47
47
  reasoning?: Effort;
48
48
  /** Force-disable reasoning (Anthropic `thinking: { type: "disabled" }`). */
49
49
  disableReasoning?: boolean;
50
+ /**
51
+ * Preserve an explicit wire-level reasoning-off request through providers
52
+ * that distinguish it from the generic disable hint.
53
+ */
54
+ forceReasoningOff?: boolean;
50
55
  /**
51
56
  * Explicit Anthropic `thinking.budget_tokens`. Mirrors Rust's
52
57
  * `resolve_thinking_budget`: pins onto whichever effort the client
@@ -1,8 +1,7 @@
1
1
  import type { ApiKeyResolver } from "./auth-retry.js";
2
2
  import type { OAuthAuthInfo, OAuthController, OAuthCredentials, OAuthPrompt, OAuthProviderId } from "./registry/oauth/types.js";
3
3
  import type { Provider } from "./types.js";
4
- import type { ClientUsageIdentity, ClientUsageReport, ClientUsageSummary, CredentialRankingStrategy, ObservedUsageEntry, UsageHistoryEntry, UsageHistoryQuery, UsageLogger, UsageProvider, UsageReport } from "./usage.js";
5
- import { type CodexResetConsumeCode, type CodexResetCredit } from "./usage/openai-codex-reset.js";
4
+ import type { ClientUsageIdentity, ClientUsageReport, ClientUsageSummary, CredentialRankingStrategy, ObservedUsageEntry, UsageHistoryEntry, UsageHistoryQuery, UsageLogger, UsageProvider, UsageReport, UsageResetCredit, UsageResetCredits } from "./usage.js";
6
5
  export { isSqliteBusyError, isSqliteCorruptionError, SqliteAuthCredentialStore } from "./auth/sqlite-credential-store.js";
7
6
  export type ApiKeyCredential = {
8
7
  type: "api_key";
@@ -685,14 +684,15 @@ export interface StoredOAuthRefreshResult<T extends OAuthCredential = OAuthCrede
685
684
  refreshed: boolean;
686
685
  removed: boolean;
687
686
  }
688
- /**
689
- * Identifies which stored account to redeem a saved rate-limit reset for.
690
- * Any one field is enough; `credentialId` is the most precise.
691
- */
687
+ /** A saved-reset option bound to one provider and durable stored credential. */
692
688
  export interface ResetCreditTarget {
693
- credentialId?: number;
689
+ provider: string;
690
+ credentialId: number;
691
+ /** Grant selected by the caller; a changed offer must be confirmed again. */
692
+ creditId?: string;
694
693
  accountId?: string;
695
694
  email?: string;
695
+ orgId?: string;
696
696
  }
697
697
  /** Outcome of {@link AuthStorage.redeemResetCredit}. */
698
698
  export interface ResetCreditRedeemOutcome {
@@ -706,20 +706,27 @@ export interface ResetCreditRedeemOutcome {
706
706
  * retryable, unlike a genuine `no_credit`), `http_<status>` (unexpected
707
707
  * HTTP).
708
708
  */
709
- code: CodexResetConsumeCode;
709
+ code: string;
710
+ provider?: string;
710
711
  accountId?: string;
711
712
  email?: string;
713
+ orgId?: string;
714
+ /** Provider explanation for an unavailable or refused reset. */
715
+ reason?: string;
716
+ /** Normalized usage limit IDs the provider confirmed it cleared. */
717
+ cleared?: string[];
712
718
  /** The credit that was spent (when one was). */
713
719
  creditId?: string;
714
720
  }
715
721
  /** One stored account's live saved-reset status, from {@link AuthStorage.listResetCredits}. */
716
- export interface ResetCreditAccountStatus {
717
- credentialId?: number;
722
+ export interface ResetCreditAccountStatus extends UsageResetCredits {
723
+ provider: string;
724
+ credentialId: number;
718
725
  accountId?: string;
719
726
  email?: string;
720
- /** Resets redeemable for this account right now (live, not cached). */
721
- availableCount: number;
722
- credits: CodexResetCredit[];
727
+ orgId?: string;
728
+ orgName?: string;
729
+ credits: UsageResetCredit[];
723
730
  /** Whether this is the given session's active account. */
724
731
  active: boolean;
725
732
  /** Set when the account's token refresh or list call failed. */
@@ -858,11 +865,23 @@ export declare class AuthStorage {
858
865
  * Check if credentials exist for a provider in storage.
859
866
  */
860
867
  has(provider: string): boolean;
868
+ /**
869
+ * True when the provider has stored credentials but none of them carries
870
+ * auth — i.e. its only credential is the KDL `empty-fallback` keyless-mode
871
+ * marker (an empty paste at an optional-key login prompt). Such a provider
872
+ * is configured-but-keyless: model availability treats it like an
873
+ * `auth: none` endpoint instead of locking it out (issue #12281).
874
+ */
875
+ hasKeylessPlaceholder(provider: string): boolean;
861
876
  /**
862
877
  * Dedicated auth for default-model availability (picker / `getAvailable`).
863
878
  * Unlike {@link getApiKey}, this does not refresh OAuth tokens, and unlike
864
879
  * {@link hasResolvableAuth} it ignores cross-provider env aliases so
865
880
  * `XAI_API_KEY` does not auto-select SuperGrok (`xai-oauth`).
881
+ *
882
+ * A stored keyless-fallback marker (empty paste at an optional-key login)
883
+ * does not count: it never reaches the wire as a bearer, so treating it as
884
+ * auth would present the provider as signed-in while requests go out bare.
866
885
  */
867
886
  hasAuth(provider: string): boolean;
868
887
  /**
@@ -1170,16 +1189,7 @@ export declare class AuthStorage {
1170
1189
  * explicit runtime/config API-key override suppresses OAuth.
1171
1190
  */
1172
1191
  getOAuthAccessByCredentialId(provider: string, credentialId: number, options?: AuthApiKeyOptions): Promise<OAuthAccessResolution | undefined>;
1173
- /**
1174
- * List saved rate-limit resets for every stored OAuth account of `provider`
1175
- * (Codex), fetched LIVE from the dedicated `rate-limit-reset-credits` route.
1176
- *
1177
- * This deliberately bypasses the usage-report cache: `/wham/usage` is
1178
- * IP-rate-limited and may serve stale (or pre-feature) snapshots when many
1179
- * accounts are polled, which would hide redeemable credits. One entry per
1180
- * account, with the session's active account flagged and unreachable
1181
- * accounts carrying an `error`.
1182
- */
1192
+ /** List live saved-reset balances and eligibility for one provider's stored OAuth accounts. */
1183
1193
  listResetCredits(options?: {
1184
1194
  provider?: string;
1185
1195
  sessionId?: string;
@@ -1187,18 +1197,11 @@ export declare class AuthStorage {
1187
1197
  signal?: AbortSignal;
1188
1198
  }): Promise<ResetCreditAccountStatus[]>;
1189
1199
  /**
1190
- * Redeem one saved rate-limit reset (OpenAI Codex "saved resets") for a
1191
- * specific stored account.
1192
- *
1193
- * Resolves a fresh access token for the target account, picks an available
1194
- * credit (the given `creditId`, else the first redeemable one), spends it,
1195
- * and invalidates the cached usage report so the next `/usage` reflects the
1196
- * reset. Never throws for business outcomes — inspect the returned `code`.
1200
+ * Redeem a stored account's saved reset after checking its live offer.
1201
+ * Business refusals return a code; transport errors may throw without losing Claude's request ID.
1197
1202
  */
1198
1203
  redeemResetCredit(options: {
1199
1204
  target: ResetCreditTarget;
1200
- provider?: string;
1201
- creditId?: string;
1202
1205
  baseUrlResolver?: (provider: string) => string | undefined;
1203
1206
  signal?: AbortSignal;
1204
1207
  }): Promise<ResetCreditRedeemOutcome>;
@@ -0,0 +1,7 @@
1
+ import type { Api, Model } from "@oh-my-pi/pi-catalog/types";
2
+ import { type EmbeddingOptions } from "./openai-embeddings.js";
3
+ import type { EmbeddingRequest, EmbeddingResult } from "./types.js";
4
+ export * from "./openai-embeddings.js";
5
+ export * from "./types.js";
6
+ /** Dispatch an embedding request through the transport selected by the catalog model. */
7
+ export declare function embed(model: Model<Api>, request: EmbeddingRequest, options: EmbeddingOptions): Promise<EmbeddingResult>;
@@ -0,0 +1,15 @@
1
+ import type { Api, FetchImpl, Model } from "@oh-my-pi/pi-catalog/types";
2
+ import { type ApiKey } from "../auth-retry.js";
3
+ import * as AIError from "../error/index.js";
4
+ import type { EmbeddingRequest, EmbeddingResult } from "./types.js";
5
+ export interface EmbeddingOptions {
6
+ apiKey: ApiKey;
7
+ fetch?: FetchImpl;
8
+ signal?: AbortSignal;
9
+ }
10
+ /** Non-2xx response from an OpenAI-compatible embeddings endpoint. */
11
+ export declare class EmbeddingApiError extends AIError.ProviderHttpError {
12
+ readonly name = "EmbeddingApiError";
13
+ }
14
+ /** Call an OpenAI/OpenRouter-compatible embeddings endpoint. */
15
+ export declare function embedOpenAI(model: Model<Api>, request: EmbeddingRequest, options: EmbeddingOptions): Promise<EmbeddingResult>;
@@ -0,0 +1,15 @@
1
+ import type { Usage } from "@oh-my-pi/pi-catalog/types";
2
+ export interface EmbeddingRequest {
3
+ input: string | string[] | number[] | number[][];
4
+ dimensions?: number;
5
+ encodingFormat: "float" | "base64";
6
+ user?: string;
7
+ }
8
+ export interface EmbeddingResult {
9
+ embeddings: Array<{
10
+ index: number;
11
+ embedding: number[] | string;
12
+ }>;
13
+ model: string;
14
+ usage: Usage;
15
+ }
@@ -0,0 +1,9 @@
1
+ import type { Model } from "@oh-my-pi/pi-catalog/types";
2
+ import type { ImageGenerationOptions, ImageGenerationRequest, ImageGenerationResult } from "./types.js";
3
+ interface AntigravityCredentials {
4
+ accessToken: string;
5
+ projectId: string;
6
+ }
7
+ export declare function parseAntigravityCredentials(raw: string): AntigravityCredentials | undefined;
8
+ export declare function generateAntigravityImage(model: Model, request: ImageGenerationRequest, options: ImageGenerationOptions): Promise<ImageGenerationResult>;
9
+ export {};
@@ -0,0 +1,3 @@
1
+ import type { Model } from "@oh-my-pi/pi-catalog/types";
2
+ import type { ImageGenerationOptions, ImageGenerationRequest, ImageGenerationResult } from "./types.js";
3
+ export declare function generateGoogleImage(model: Model, request: ImageGenerationRequest, options: ImageGenerationOptions): Promise<ImageGenerationResult>;
@@ -0,0 +1,14 @@
1
+ import type { Api, Model } from "@oh-my-pi/pi-catalog/types";
2
+ import type { ImageGenerationOptions, ImageGenerationRequest, ImageGenerationResult } from "./types.js";
3
+ export * from "./google-antigravity.js";
4
+ export * from "./google-generative-ai.js";
5
+ export * from "./openai-hosted.js";
6
+ export * from "./openai-images.js";
7
+ export * from "./openrouter-images.js";
8
+ export * from "./types.js";
9
+ /** Catalog APIs {@link generateImage} serves; the hosted Responses pair needs an explicit carrier model. */
10
+ export type ImageGenerationApi = "openai-images" | "openrouter-images" | "google-generative-ai" | "google-gemini-cli" | "openai-responses" | "openai-codex-responses";
11
+ /** Whether a catalog API generates images through one of the pi-ai image clients. */
12
+ export declare function isImageGenerationApi(api: Api): api is ImageGenerationApi;
13
+ /** Generate (or edit, when `request.inputImages` is set) images through the transport selected by the model's `api`. */
14
+ export declare function generateImage(model: Model<Api>, request: ImageGenerationRequest, options: ImageGenerationOptions): Promise<ImageGenerationResult>;
@@ -0,0 +1,3 @@
1
+ import type { Model } from "@oh-my-pi/pi-catalog/types";
2
+ import type { ImageGenerationOptions, ImageGenerationRequest, ImageGenerationResult } from "./types.js";
3
+ export declare function generateHostedImage(model: Model, request: ImageGenerationRequest, options: ImageGenerationOptions): Promise<ImageGenerationResult>;
@@ -0,0 +1,5 @@
1
+ import type { Model } from "@oh-my-pi/pi-catalog/types";
2
+ import type { ImageGenerationOptions, ImageGenerationRequest, ImageGenerationResult } from "./types.js";
3
+ export declare const XAI_MAX_EDIT_IMAGES = 3;
4
+ export declare function resolveXAIResolution(imageSize?: string): "1k" | "2k";
5
+ export declare function generateOpenAIImage(model: Model, request: ImageGenerationRequest, options: ImageGenerationOptions): Promise<ImageGenerationResult>;
@@ -0,0 +1,3 @@
1
+ import type { Model } from "@oh-my-pi/pi-catalog/types";
2
+ import type { ImageGenerationOptions, ImageGenerationRequest, ImageGenerationResult } from "./types.js";
3
+ export declare function generateOpenRouterImage(model: Model, request: ImageGenerationRequest, options: ImageGenerationOptions): Promise<ImageGenerationResult>;
@@ -0,0 +1,34 @@
1
+ import type { FetchImpl, Model, Usage } from "@oh-my-pi/pi-catalog/types";
2
+ import type { ApiKey } from "../auth-retry.js";
3
+ import * as AIError from "../error/index.js";
4
+ import type { GeneratedImage } from "./types.js";
5
+ export declare class ImageApiError extends AIError.ProviderHttpError {
6
+ readonly name = "ImageApiError";
7
+ }
8
+ export declare function emptyUsage(input?: number, output?: number, cost?: number): Usage;
9
+ export declare function usageFromWire(value: unknown): Usage;
10
+ export declare function imageBaseUrl(model: Model): string;
11
+ export declare function modelHeaders(model: Model, signal?: AbortSignal): Promise<Record<string, string>>;
12
+ export declare function errorMessage(rawText: string): string;
13
+ export declare function postJson(options: {
14
+ model: Model;
15
+ url: string;
16
+ body: unknown;
17
+ apiKey: ApiKey;
18
+ fetch: FetchImpl;
19
+ signal?: AbortSignal;
20
+ }): Promise<unknown>;
21
+ export declare function postMultipart(options: {
22
+ model: Model;
23
+ url: string;
24
+ body: FormData;
25
+ apiKey: ApiKey;
26
+ fetch: FetchImpl;
27
+ signal?: AbortSignal;
28
+ }): Promise<unknown>;
29
+ export declare function decodeImageResponse(value: unknown, fetch: FetchImpl, signal?: AbortSignal): Promise<{
30
+ images: GeneratedImage[];
31
+ usage: Usage;
32
+ }>;
33
+ export declare function toDataUrl(image: GeneratedImage): string;
34
+ export declare function resolveOpenAIImageSize(aspectRatio?: string, imageSize?: string): string | undefined;
@@ -0,0 +1,31 @@
1
+ import type { Api, FetchImpl, Model, Usage } from "@oh-my-pi/pi-catalog/types";
2
+ import type { ApiKey } from "../auth-retry.js";
3
+ export interface ImageInput {
4
+ data: string;
5
+ mimeType: string;
6
+ }
7
+ export interface ImageGenerationRequest {
8
+ prompt: string;
9
+ inputImages?: ImageInput[];
10
+ aspectRatio?: string;
11
+ imageSize?: string;
12
+ count?: number;
13
+ }
14
+ export interface GeneratedImage {
15
+ data: string;
16
+ mimeType: string;
17
+ }
18
+ export interface ImageGenerationResult {
19
+ images: GeneratedImage[];
20
+ text?: string;
21
+ usage: Usage;
22
+ }
23
+ export interface ImageGenerationOptions {
24
+ apiKey: ApiKey;
25
+ fetch?: FetchImpl;
26
+ signal?: AbortSignal;
27
+ /** Chat model that carries a Responses `image_generation` tool call. */
28
+ carrier?: Model<Api>;
29
+ /** Stable provider session id, used by the Codex Responses carrier. */
30
+ sessionId?: string;
31
+ }
@@ -1,12 +1,18 @@
1
1
  export { type Type, type } from "@oh-my-pi/omptype";
2
2
  export * from "./api-registry.js";
3
3
  export type * from "./auth-broker/index.js";
4
- export type { AuthGatewayBootOptions, ModelResolver } from "./auth-gateway/server.js";
4
+ export type { AuthGatewayBootOptions, ModelResolver } from "./auth-gateway/dispatch.js";
5
5
  export * from "./auth-gateway/types.js";
6
6
  export * from "./auth-retry.js";
7
7
  export * from "./auth-storage.js";
8
8
  export * from "./error/rate-limit.js";
9
+ export * from "./embeddings/index.js";
10
+ export * from "./images/index.js";
9
11
  export * from "./judgment/index.js";
12
+ export * from "./rerank/index.js";
13
+ export * from "./speech/index.js";
14
+ export * from "./transcription/index.js";
15
+ export * from "./video/index.js";
10
16
  export * from "./oneshot-retry.js";
11
17
  export * from "./provider-details.js";
12
18
  export * from "./provider-session-state.js";
@@ -33,6 +39,7 @@ export * from "./stream.js";
33
39
  export * from "./types.js";
34
40
  export * from "./usage.js";
35
41
  export * from "./usage/claude.js";
42
+ export * from "./usage/claude-reset.js";
36
43
  export * from "./usage/cursor.js";
37
44
  export * from "./usage/gemini.js";
38
45
  export * from "./usage/github-copilot.js";
@@ -27,6 +27,8 @@ export interface TypeSafeJudgeOptions {
27
27
  baseUrl?: string;
28
28
  /** Defaults to {@link typesafeModel}. */
29
29
  model?: string;
30
+ /** Static headers attached to judgment requests (e.g. proxy routing, gateway auth). */
31
+ headers?: Record<string, string>;
30
32
  fetch?: FetchImpl;
31
33
  /** Per-attempt timeout; defaults to {@link DEFAULT_TIMEOUT_MS}. */
32
34
  timeoutMs?: number;
@@ -11,6 +11,13 @@
11
11
  */
12
12
  import type { Effort } from "@oh-my-pi/pi-catalog/effort";
13
13
  import type { StreamFunction, StreamOptions, ThinkingBudgets } from "../types.js";
14
+ /**
15
+ * Resolve the service-model status for an in-stream exception/error code.
16
+ * Frame headers carry the bare shape name in either camelCase
17
+ * (`internalServerException`) or PascalCase (`InternalServerException`); both
18
+ * normalize to the same map key. Unknown shapes default to 400.
19
+ */
20
+ export declare function bedrockStreamExceptionStatus(code: string): number;
14
21
  export type BedrockThinkingDisplay = "summarized" | "omitted";
15
22
  /** Bedrock guardrail trace verbosity, mirrors the Converse `guardrailConfig.trace` values. */
16
23
  export type BedrockGuardrailTrace = "enabled" | "disabled" | "enabled_full";