@oh-my-pi/pi-ai 18.2.7 → 18.2.9

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (133) hide show
  1. package/CHANGELOG.md +49 -18
  2. package/dist/types/auth-gateway/dispatch.d.ts +80 -0
  3. package/dist/types/auth-gateway/http.d.ts +5 -6
  4. package/dist/types/auth-gateway/index.d.ts +1 -0
  5. package/dist/types/auth-gateway/routes/embeddings.d.ts +3 -0
  6. package/dist/types/auth-gateway/routes/images.d.ts +3 -0
  7. package/dist/types/auth-gateway/routes/rerank.d.ts +3 -0
  8. package/dist/types/auth-gateway/routes/speech.d.ts +2 -0
  9. package/dist/types/auth-gateway/routes/systemone.d.ts +2 -0
  10. package/dist/types/auth-gateway/routes/transcriptions.d.ts +3 -0
  11. package/dist/types/auth-gateway/routes/video.d.ts +7 -0
  12. package/dist/types/auth-gateway/server.d.ts +10 -16
  13. package/dist/types/auth-gateway/types.d.ts +5 -0
  14. package/dist/types/auth-storage.d.ts +35 -32
  15. package/dist/types/embeddings/index.d.ts +7 -0
  16. package/dist/types/embeddings/openai-embeddings.d.ts +15 -0
  17. package/dist/types/embeddings/types.d.ts +15 -0
  18. package/dist/types/images/google-antigravity.d.ts +9 -0
  19. package/dist/types/images/google-generative-ai.d.ts +3 -0
  20. package/dist/types/images/index.d.ts +14 -0
  21. package/dist/types/images/openai-hosted.d.ts +3 -0
  22. package/dist/types/images/openai-images.d.ts +5 -0
  23. package/dist/types/images/openrouter-images.d.ts +3 -0
  24. package/dist/types/images/shared.d.ts +34 -0
  25. package/dist/types/images/types.d.ts +31 -0
  26. package/dist/types/index.d.ts +8 -1
  27. package/dist/types/judgment/typesafe.d.ts +2 -0
  28. package/dist/types/providers/amazon-bedrock.d.ts +7 -0
  29. package/dist/types/providers/claude-code-fingerprint.d.ts +22 -3
  30. package/dist/types/providers/embeddings-server.d.ts +32 -0
  31. package/dist/types/providers/google-gemini-cli.d.ts +0 -2
  32. package/dist/types/providers/images-server.d.ts +22 -0
  33. package/dist/types/providers/openai-chat-server-schema.d.ts +2 -2
  34. package/dist/types/providers/rerank-server.d.ts +35 -0
  35. package/dist/types/providers/speech-server.d.ts +8 -0
  36. package/dist/types/providers/systemone-server.d.ts +26 -0
  37. package/dist/types/providers/transcriptions-server.d.ts +32 -0
  38. package/dist/types/providers/video-server.d.ts +39 -0
  39. package/dist/types/rerank/index.d.ts +7 -0
  40. package/dist/types/rerank/openrouter-rerank.d.ts +15 -0
  41. package/dist/types/rerank/types.d.ts +17 -0
  42. package/dist/types/speech/index.d.ts +13 -0
  43. package/dist/types/speech/openai-speech.d.ts +3 -0
  44. package/dist/types/speech/transport.d.ts +7 -0
  45. package/dist/types/speech/types.d.ts +24 -0
  46. package/dist/types/speech/xai-tts.d.ts +7 -0
  47. package/dist/types/transcription/index.d.ts +7 -0
  48. package/dist/types/transcription/openai-transcriptions.d.ts +15 -0
  49. package/dist/types/transcription/types.d.ts +41 -0
  50. package/dist/types/usage/claude-api.d.ts +22 -0
  51. package/dist/types/usage/claude-reset.d.ts +44 -0
  52. package/dist/types/usage.d.ts +111 -5
  53. package/dist/types/utils/schema/json-schema-validator.d.ts +5 -2
  54. package/dist/types/utils/tool-call-loop-guard.d.ts +1 -1
  55. package/dist/types/video/index.d.ts +11 -0
  56. package/dist/types/video/openrouter-video.d.ts +19 -0
  57. package/dist/types/video/types.d.ts +62 -0
  58. package/package.json +30 -6
  59. package/src/auth/sqlite-credential-store.ts +44 -1
  60. package/src/auth-broker/remote-store.ts +6 -6
  61. package/src/auth-broker/wire-schemas.ts +14 -0
  62. package/src/auth-gateway/dispatch.ts +273 -0
  63. package/src/auth-gateway/http.ts +6 -7
  64. package/src/auth-gateway/index.ts +1 -0
  65. package/src/auth-gateway/routes/embeddings.ts +98 -0
  66. package/src/auth-gateway/routes/images.ts +131 -0
  67. package/src/auth-gateway/routes/rerank.ts +87 -0
  68. package/src/auth-gateway/routes/speech.ts +101 -0
  69. package/src/auth-gateway/routes/systemone.ts +116 -0
  70. package/src/auth-gateway/routes/transcriptions.ts +98 -0
  71. package/src/auth-gateway/routes/video.ts +243 -0
  72. package/src/auth-gateway/server.ts +127 -260
  73. package/src/auth-gateway/types.ts +5 -0
  74. package/src/auth-storage.ts +263 -140
  75. package/src/embeddings/index.ts +17 -0
  76. package/src/embeddings/openai-embeddings.ts +141 -0
  77. package/src/embeddings/types.ts +14 -0
  78. package/src/error/flags.ts +10 -0
  79. package/src/error/rate-limit.ts +1 -1
  80. package/src/images/google-antigravity.ts +180 -0
  81. package/src/images/google-generative-ai.ts +92 -0
  82. package/src/images/index.ts +59 -0
  83. package/src/images/openai-hosted.ts +185 -0
  84. package/src/images/openai-images.ts +110 -0
  85. package/src/images/openrouter-images.ts +33 -0
  86. package/src/images/shared.ts +193 -0
  87. package/src/images/types.ts +36 -0
  88. package/src/index.ts +8 -1
  89. package/src/judgment/typesafe.ts +5 -0
  90. package/src/providers/amazon-bedrock.ts +55 -5
  91. package/src/providers/anthropic.ts +45 -11
  92. package/src/providers/aws-credentials.ts +124 -11
  93. package/src/providers/claude-code-fingerprint.ts +55 -3
  94. package/src/providers/embeddings-server.ts +151 -0
  95. package/src/providers/gitlab-duo.ts +20 -4
  96. package/src/providers/google-gemini-cli.ts +0 -8
  97. package/src/providers/google-shared.ts +1 -18
  98. package/src/providers/images-server.ts +159 -0
  99. package/src/providers/openai-chat-server-schema.ts +1 -1
  100. package/src/providers/openai-chat-server.ts +3 -1
  101. package/src/providers/openai-codex-responses.ts +27 -5
  102. package/src/providers/openai-completions.ts +122 -19
  103. package/src/providers/pi-native-server.ts +1 -0
  104. package/src/providers/rerank-server.ts +166 -0
  105. package/src/providers/speech-server.ts +53 -0
  106. package/src/providers/systemone-server.ts +73 -0
  107. package/src/providers/transcriptions-server.ts +243 -0
  108. package/src/providers/video-server.ts +286 -0
  109. package/src/registry/oauth/anthropic.ts +2 -3
  110. package/src/rerank/index.ts +13 -0
  111. package/src/rerank/openrouter-rerank.ts +136 -0
  112. package/src/rerank/types.ts +20 -0
  113. package/src/speech/index.ts +35 -0
  114. package/src/speech/openai-speech.ts +26 -0
  115. package/src/speech/transport.ts +66 -0
  116. package/src/speech/types.ts +37 -0
  117. package/src/speech/xai-tts.ts +41 -0
  118. package/src/stream.ts +13 -4
  119. package/src/transcription/index.ts +17 -0
  120. package/src/transcription/openai-transcriptions.ts +133 -0
  121. package/src/transcription/types.ts +46 -0
  122. package/src/usage/alibaba-token-plan.ts +7 -1
  123. package/src/usage/claude-api.ts +66 -0
  124. package/src/usage/claude-reset.ts +638 -0
  125. package/src/usage/claude.ts +37 -59
  126. package/src/usage/kimi.ts +32 -1
  127. package/src/usage.ts +52 -5
  128. package/src/utils/schema/json-schema-validator.ts +23 -10
  129. package/src/utils/tool-call-loop-guard.ts +2 -2
  130. package/src/utils/validation.ts +145 -50
  131. package/src/video/index.ts +34 -0
  132. package/src/video/openrouter-video.ts +210 -0
  133. package/src/video/types.ts +72 -0
@@ -16,25 +16,38 @@
16
16
  * POST /v1/chat/completions → OpenAI chat-completions in/out
17
17
  * POST /v1/messages → Anthropic messages in/out
18
18
  * POST /v1/responses → OpenAI Responses in/out
19
+ * POST /v1/pi/stream → native pi-ai stream in/out
20
+ * POST /v1/systemone | /alpha/decisions → TypeSafe System One judgments (routes/systemone)
21
+ * POST /v1/images[/generations|/edits] → image generation, OpenAI/OpenRouter wire (routes/images)
22
+ * POST /v1/audio/speech → text-to-speech, raw audio out (routes/speech)
23
+ * POST /v1/audio/transcriptions → speech-to-text, multipart or JSON base64 in (routes/transcriptions)
24
+ *
25
+ * Chat routes live in this file; every other modality is a `routes/*` module
26
+ * built on the shared plumbing in `dispatch.ts`.
19
27
  */
20
28
 
21
29
  import { Effort } from "@oh-my-pi/pi-catalog/effort";
22
- import { extractHttpStatusFromError, logger } from "@oh-my-pi/pi-utils";
23
- import type { ApiKeyResolver } from "../auth-retry";
30
+ import { type ModelKind, modelKind } from "@oh-my-pi/pi-catalog/types";
31
+ import { logger } from "@oh-my-pi/pi-utils";
24
32
  import type { AuthStorage } from "../auth-storage";
25
- import * as AIError from "../error";
26
33
  import { classifyGatewayError } from "../error/gateway";
27
- import { isUsageLimitOutcome } from "../error/rate-limit";
28
34
  import * as anthropicMessages from "../providers/anthropic-messages-server";
29
35
  import * as openaiChat from "../providers/openai-chat-server";
30
36
  import * as openaiResponses from "../providers/openai-responses-server";
31
37
  import * as piNative from "../providers/pi-native-server";
32
38
  import { completeSimple, streamSimple } from "../stream";
33
- import type { Api, AssistantMessage, AssistantMessageEventStream, Context, Model, SimpleStreamOptions } from "../types";
34
- import type { ClientUsageIdentity } from "../usage";
39
+ import type { Api, AssistantMessageEventStream, Context, Model, SimpleStreamOptions } from "../types";
35
40
  import { deterministicUuid } from "../utils/deterministic-id";
36
41
  import { parseBind } from "../utils/parse-bind";
37
- import { extractProviderRetryHint } from "../utils/retry-after";
42
+ import {
43
+ type AuthGatewayBootOptions,
44
+ buildGatewayApiKeyResolver,
45
+ mirrorRequestAbort,
46
+ normalizeClientSessionKey,
47
+ recordGatewayUsage,
48
+ resolveGatewayAccount,
49
+ resolveGatewayApiKey,
50
+ } from "./dispatch";
38
51
  import {
39
52
  captureRequestHeaders,
40
53
  corsHeaders,
@@ -45,32 +58,21 @@ import {
45
58
  resolvePeer,
46
59
  withCors,
47
60
  } from "./http";
61
+ import { handleEmbeddings } from "./routes/embeddings";
62
+ import { handleImageEdits, handleImageGenerations } from "./routes/images";
63
+ import { handleRerank } from "./routes/rerank";
64
+ import { handleSpeech } from "./routes/speech";
65
+ import { handleSystemOne } from "./routes/systemone";
66
+ import { handleTranscriptions } from "./routes/transcriptions";
67
+ import { handleVideoContent, handleVideoPoll, handleVideoSubmit } from "./routes/video";
48
68
  import { AuthGatewaySessionStateStore } from "./session-state";
49
69
  import type {
50
70
  AuthGatewayServerHandle,
51
- AuthGatewayServerOptions,
52
71
  AuthGatewayFormatModule as FormatModule,
53
72
  AuthGatewayParsedRequest as ParsedFormatRequest,
54
73
  } from "./types";
55
74
  import { DEFAULT_AUTH_GATEWAY_BIND } from "./types";
56
75
 
57
- // ParsedFormatRequest / ParsedFormatOptions / FormatModule come from ./types.
58
-
59
- export type ModelResolver = (modelId: string) => Model<Api> | undefined;
60
-
61
- export interface AuthGatewayBootOptions extends AuthGatewayServerOptions {
62
- /** Source of credentials. Caller wires this to a broker-backed AuthStorage. */
63
- storage: AuthStorage;
64
- /**
65
- * Resolve a client-requested model id to a pi-ai Model. Caller supplies
66
- * this from a ModelRegistry (lives in `coding-agent` to avoid an inverse
67
- * dependency in `pi-ai`).
68
- */
69
- resolveModel: ModelResolver;
70
- /** Optional supplier for `/v1/models` listing. Returns the full model array. */
71
- listModels?: () => Iterable<Model<Api>>;
72
- }
73
-
74
76
  // `parseBind` lives in ../utils/parse-bind so the gateway and broker can't
75
77
  // drift on accepted inputs (e.g. empty hostname, IPv6 brackets).
76
78
 
@@ -130,40 +132,6 @@ function deriveSessionId(modelId: string, context: Context): string {
130
132
  return deterministicUuid(seed);
131
133
  }
132
134
 
133
- /**
134
- * The client's own session key, or `undefined` when it sent none. A blank key
135
- * counts as none: honouring it would collapse every caller that sends an empty
136
- * key into one shared credential-sticky, prefix-cache and provider-session
137
- * bucket.
138
- */
139
- function normalizeClientSessionKey(clientKey: string | undefined): string | undefined {
140
- return clientKey !== undefined && clientKey.trim().length > 0 ? clientKey : undefined;
141
- }
142
-
143
- /**
144
- * Stable identity of the account a request's credential belongs to.
145
- *
146
- * `markUsageLimitReached` and the auth-retry resolver switch a session to a
147
- * sibling credential, so the provider state retained for that session can
148
- * outlive the account that taught it. OAuth rows expose an account id / email
149
- * that survives token refresh — fingerprinting the bearer instead would look
150
- * like a rotation every time a token refreshes and discard the retained
151
- * lessons for nothing. Key-based rows fall back to a hash of the key, never
152
- * the key itself: this value is held for the lifetime of the entry.
153
- */
154
- function resolveGatewayAccount(storage: AuthStorage, provider: string, sessionId: string, apiKey: string): string {
155
- const identity = storage.getOAuthAccountIdentity(provider, sessionId);
156
- if (identity) {
157
- return `oauth:${JSON.stringify([
158
- identity.accountId ?? "",
159
- identity.email ?? "",
160
- identity.projectId ?? "",
161
- identity.orgId ?? "",
162
- ])}`;
163
- }
164
- return `key:${Bun.hash(apiKey).toString(36)}`;
165
- }
166
-
167
135
  function buildStreamOptions(parsed: ParsedFormatRequest, api: Api, signal: AbortSignal): SimpleStreamOptions {
168
136
  const opts: SimpleStreamOptions = { signal, cursorExternalToolExecutor: true };
169
137
  const { options } = parsed;
@@ -193,6 +161,10 @@ function buildStreamOptions(parsed: ParsedFormatRequest, api: Api, signal: Abort
193
161
  }
194
162
  if (options.reasoning !== undefined) opts.reasoning = options.reasoning;
195
163
  if (options.disableReasoning !== undefined) opts.disableReasoning = options.disableReasoning;
164
+ if (options.forceReasoningOff !== undefined) {
165
+ opts.disableReasoning = options.forceReasoningOff;
166
+ opts.forceReasoningOff = options.forceReasoningOff;
167
+ }
196
168
  if (options.hideThinkingSummary !== undefined) opts.hideThinkingSummary = options.hideThinkingSummary;
197
169
  if (options.taskBudget !== undefined) opts.taskBudget = options.taskBudget;
198
170
  if (options.anthropicPrefixMismatchBehavior !== undefined) {
@@ -249,167 +221,26 @@ function buildStreamOptions(parsed: ParsedFormatRequest, api: Api, signal: Abort
249
221
  return opts;
250
222
  }
251
223
 
252
- /**
253
- * Hook fired by {@link streamSimple} when the upstream request fails in a
254
- * way that's rotatable — today that's HTTP 401 (credential is bad) and
255
- * usage-limit phrasing matched by {@link isUsageLimitError} (Codex's
256
- * `usage_limit_reached`, Anthropic's `usage_limit_reached`, Google's
257
- * `resource_exhausted`, …). The two cases need different storage actions:
258
- *
259
- * - **usage-limit** → {@link AuthStorage.markUsageLimitReached}. Marks just
260
- * the current session's credential as temporarily blocked (honouring
261
- * `retry-after` / `resets_at` hints when present) and returns `true` only
262
- * when a sibling credential is still available. Burning the credential
263
- * with `invalidateCredentialMatching` here would orphan accounts whose
264
- * reset window is several hours away — exactly the bug this helper exists
265
- * to avoid.
266
- * - **auth-failure** → {@link AuthStorage.invalidateCredentialMatching}.
267
- * Suspect/delete the row so it doesn't get re-picked next request.
268
- *
269
- * In both branches we return the next `getApiKey` result (sticky on the
270
- * same `sessionId`) so streamSimple can transparently retry the pre-emit
271
- * failure with a fresh credential. Returning `undefined` aborts the retry
272
- * and surfaces the original error to the caller.
273
- */
274
- async function refreshGatewayApiKeyAfterAuthError(
275
- storage: AuthStorage,
276
- model: Model<Api>,
277
- sessionId: string,
278
- provider: string,
279
- oldKey: string,
280
- error: unknown,
281
- signal: AbortSignal,
282
- format: string,
283
- peer: string,
284
- ): Promise<string | undefined> {
285
- const message = error instanceof Error ? error.message : String(error);
286
- const status = extractHttpStatusFromError(error);
287
- if (AIError.isUsageLimit(error) || isUsageLimitOutcome(status, message)) {
288
- const retryAfterMs = extractProviderRetryHint(provider, message);
289
- const { switched, retryAtMs } = await storage.markUsageLimitReached(provider, sessionId, {
290
- retryAfterMs,
291
- providerTimed: retryAfterMs !== undefined,
292
- baseUrl: model.baseUrl,
293
- modelId: model.id,
294
- apiKey: oldKey,
295
- signal,
296
- });
297
- logger.debug("auth-gateway retrying provider request after usage-limit block", {
298
- format,
299
- provider,
300
- peer,
301
- switched,
302
- retryAfterMs,
303
- retryAtMs,
304
- error: message,
305
- });
306
- if (!switched) return undefined;
307
- return storage.getApiKey(provider, sessionId, { modelId: model.id, signal });
308
- }
309
- await storage.invalidateCredentialMatching(provider, oldKey, { sessionId, signal });
310
- logger.debug("auth-gateway retrying provider request after credential invalidation", {
311
- format,
312
- provider,
313
- peer,
314
- error: message,
315
- });
316
- return storage.getApiKey(provider, sessionId, { modelId: model.id, signal });
317
- }
318
-
319
- /**
320
- * Build the {@link ApiKeyResolver} handed to `streamSimple` for a gateway
321
- * request. Drives the central a/b/c auth-retry policy server-side:
322
- *
323
- * - initial resolve → the credential already resolved for this request.
324
- * - step (b) `!lastChance` → force-refresh the SAME session-sticky credential
325
- * (a peer/broker may have rotated its token out from under our cached copy).
326
- * - step (c) `lastChance` → {@link refreshGatewayApiKeyAfterAuthError} switches
327
- * to a sibling (usage-limit block vs credential invalidation by error class).
328
- *
329
- * `lastKey` tracks the most recent bearer so the switch step invalidates the
330
- * credential that actually failed.
331
- */
332
- function buildGatewayApiKeyResolver(
333
- storage: AuthStorage,
334
- model: Model<Api>,
335
- sessionId: string,
336
- initialKey: string,
337
- requestSignal: AbortSignal,
338
- format: string,
339
- peer: string,
340
- onResolvedKey: (apiKey: string) => void,
341
- ): ApiKeyResolver {
342
- let lastKey = initialKey;
343
- return async ({ lastChance, error, signal }) => {
344
- const sig = signal ?? requestSignal;
345
- if (error === undefined) {
346
- lastKey = initialKey;
347
- return initialKey;
348
- }
349
- if (!lastChance) {
350
- const refreshed = await storage.getApiKey(model.provider, sessionId, {
351
- modelId: model.id,
352
- signal: sig,
353
- forceRefresh: true,
354
- });
355
- lastKey = refreshed ?? lastKey;
356
- if (refreshed) onResolvedKey(refreshed);
357
- return refreshed;
358
- }
359
- const next = await refreshGatewayApiKeyAfterAuthError(
360
- storage,
361
- model,
362
- sessionId,
363
- model.provider,
364
- lastKey,
365
- error,
366
- sig,
367
- format,
368
- peer,
369
- );
370
- lastKey = next ?? lastKey;
371
- if (next) onResolvedKey(next);
372
- return next;
373
- };
374
- }
375
-
376
224
  function clientClosedResponse(route: { module: FormatModule }): Response {
377
225
  return route.module.formatError(499, "request_aborted", "client closed request");
378
226
  }
379
227
 
380
- /**
381
- * Attribute one settled upstream request to the originating client via the
382
- * broker's observed-usage channel (`AuthStorage.recordObservedUsage`, batched
383
- * by the remote store). Error/aborted turns still record — the provider
384
- * billed whatever tokens the partial turn consumed; zero-usage messages
385
- * (pre-flight failures) are skipped.
386
- */
387
- function recordGatewayUsage(
388
- storage: AuthStorage,
389
- model: Model<Api>,
390
- client: ClientUsageIdentity,
391
- message: AssistantMessage,
392
- ): void {
393
- const usage = message.usage;
394
- if (usage.input + usage.output + usage.cacheRead + usage.cacheWrite === 0) return;
395
- storage.recordObservedUsage({
396
- provider: model.provider,
397
- model: model.id,
398
- at: message.timestamp || Date.now(),
399
- usage: { input: usage.input, output: usage.output, cacheRead: usage.cacheRead, cacheWrite: usage.cacheWrite },
400
- costUsd: usage.cost.total,
401
- client,
402
- });
403
- }
228
+ /** Route that serves each non-chat catalog kind the gateway advertises. */
229
+ const KIND_ROUTES: Partial<Record<ModelKind, string>> = {
230
+ judge: "POST /v1/systemone",
231
+ image: "POST /v1/images/generations",
232
+ tts: "POST /v1/audio/speech",
233
+ stt: "POST /v1/audio/transcriptions",
234
+ embedding: "POST /v1/embeddings",
235
+ rerank: "POST /v1/rerank",
236
+ video: "POST /v1/videos",
237
+ };
404
238
 
405
- function mirrorRequestAbort(req: Request): AbortController {
406
- const controller = new AbortController();
407
- if (req.signal.aborted) {
408
- controller.abort(req.signal.reason);
409
- } else {
410
- req.signal.addEventListener("abort", () => controller.abort(req.signal.reason), { once: true });
411
- }
412
- return controller;
239
+ /** Chat routes cannot drive a non-chat model; name the route that does, or `undefined` for chat models. */
240
+ function chatRouteRejection(model: Model<Api>): string | undefined {
241
+ const kind = modelKind(model);
242
+ const route = KIND_ROUTES[kind];
243
+ return route && `Model ${model.provider}/${model.id} is a ${kind} model; use ${route}`;
413
244
  }
414
245
 
415
246
  // (handlePassthrough removed — see note above.)
@@ -450,6 +281,8 @@ async function handleFormatEndpoint(
450
281
  if (!model) {
451
282
  return route.module.formatError(404, "invalid_request_error", `Unknown model: ${modelId}`);
452
283
  }
284
+ const kindRejection = chatRouteRejection(model);
285
+ if (kindRejection) return route.module.formatError(400, "invalid_request_error", kindRejection);
453
286
  const client = resolveClientIdentity(req.headers);
454
287
 
455
288
  // Parse the wire-format request BEFORE resolving the credential so we
@@ -510,28 +343,12 @@ async function handleFormatEndpoint(
510
343
  // expected to resolve the credential and pass it as `options.apiKey`.
511
344
  // For OAuth providers this returns the access token (refreshed via the
512
345
  // broker override on AuthStorage when needed).
513
- let apiKey: string | undefined;
514
- try {
515
- apiKey = await bootOpts.storage.getApiKey(model.provider, sessionId, {
516
- modelId: model.id,
517
- signal: controller.signal,
518
- });
519
- } catch (error) {
520
- if (controller.signal.aborted) return clientClosedResponse(route);
521
- const classified = classifyGatewayError(error);
522
- logger.warn("auth-gateway getApiKey threw", { provider: model.provider, peer, error: classified.message });
523
- return route.module.formatError(classified.status, classified.type, classified.message);
524
- }
346
+ const apiKey = await resolveGatewayApiKey(bootOpts.storage, model, sessionId, controller.signal, peer);
525
347
  if (controller.signal.aborted) return clientClosedResponse(route);
526
- if (!apiKey) {
527
- return route.module.formatError(
528
- 401,
529
- "authentication_error",
530
- `No credential available for provider ${model.provider}`,
531
- );
532
- }
348
+ if (typeof apiKey !== "string") return route.module.formatError(apiKey.status, apiKey.type, apiKey.message);
533
349
 
534
350
  const streamOpts = buildStreamOptions(parsed, model.api, controller.signal);
351
+ if (bootOpts.fetch) streamOpts.fetch = bootOpts.fetch;
535
352
  // Per-session provider learning (sticky strict-tools / fast-mode / thinking
536
353
  // fallbacks, Codex transport sessions). Owned by this gateway instance: the
537
354
  // map is non-serializable, so no client can supply it and every turn would
@@ -571,7 +388,7 @@ async function handleFormatEndpoint(
571
388
  try {
572
389
  if (controller.signal.aborted) return clientClosedResponse(route);
573
390
  const message = await completeSimple(model, parsed.context, streamOpts);
574
- recordGatewayUsage(bootOpts.storage, model, client, message);
391
+ recordGatewayUsage(bootOpts.storage, model, client, message.usage, message.timestamp || undefined);
575
392
  if (message.stopReason === "aborted" || message.stopReason === "error") {
576
393
  const errorMessage =
577
394
  message.errorMessage ??
@@ -591,7 +408,7 @@ async function handleFormatEndpoint(
591
408
  return json(
592
409
  200,
593
410
  route.module.encodeResponse(message, parsed.modelId),
594
- gatewayResponseHeaders(model, { requestId, message, startedAt }),
411
+ gatewayResponseHeaders(model, { requestId, costUsd: message.usage.cost.total, startedAt }),
595
412
  );
596
413
  } catch (error) {
597
414
  if (controller.signal.aborted) return clientClosedResponse(route);
@@ -626,7 +443,9 @@ async function handleFormatEndpoint(
626
443
  if (controller.signal.aborted) return clientClosedResponse(route);
627
444
  void events
628
445
  .result()
629
- .then(message => recordGatewayUsage(bootOpts.storage, model, client, message))
446
+ .then(message =>
447
+ recordGatewayUsage(bootOpts.storage, model, client, message.usage, message.timestamp || undefined),
448
+ )
630
449
  .catch(() => {})
631
450
  .finally(() => lease.release());
632
451
  streamOwnsLease = true;
@@ -705,6 +524,8 @@ async function handlePiNative(
705
524
  if (!model) {
706
525
  return piNative.formatError(404, "invalid_request_error", `Unknown model: ${parsed.modelId}`);
707
526
  }
527
+ const kindRejection = chatRouteRejection(model);
528
+ if (kindRejection) return piNative.formatError(400, "invalid_request_error", kindRejection);
708
529
  const client = resolveClientIdentity(req.headers);
709
530
  // Pi-native already parsed `streamOpts.sessionId` (when set by the
710
531
  // client); fall back to the derived key so credential-stickiness lines
@@ -715,26 +536,9 @@ async function handlePiNative(
715
536
  const sessionId = clientKey ?? deriveSessionId(parsed.modelId, parsed.context);
716
537
  parsed.options.sessionId = sessionId;
717
538
 
718
- let apiKey: string | undefined;
719
- try {
720
- apiKey = await bootOpts.storage.getApiKey(model.provider, sessionId, {
721
- modelId: model.id,
722
- signal: controller.signal,
723
- });
724
- } catch (error) {
725
- if (controller.signal.aborted) return aborted();
726
- const classified = classifyGatewayError(error);
727
- logger.warn("auth-gateway getApiKey threw", { provider: model.provider, peer, error: classified.message });
728
- return piNative.formatError(classified.status, classified.type, classified.message);
729
- }
539
+ const apiKey = await resolveGatewayApiKey(bootOpts.storage, model, sessionId, controller.signal, peer);
730
540
  if (controller.signal.aborted) return aborted();
731
- if (!apiKey) {
732
- return piNative.formatError(
733
- 401,
734
- "authentication_error",
735
- `No credential available for provider ${model.provider}`,
736
- );
737
- }
541
+ if (typeof apiKey !== "string") return piNative.formatError(apiKey.status, apiKey.type, apiKey.message);
738
542
 
739
543
  // Per-session provider learning, owned by this gateway instance. The map is
740
544
  // non-serializable, so `parseRequest` cannot accept one from the wire and
@@ -758,6 +562,7 @@ async function handlePiNative(
758
562
  cursorExternalToolExecutor: true,
759
563
  providerSessionState: lease.states,
760
564
  };
565
+ if (bootOpts.fetch) streamOpts.fetch = bootOpts.fetch;
761
566
  streamOpts.apiKey = buildGatewayApiKeyResolver(
762
567
  bootOpts.storage,
763
568
  model,
@@ -799,7 +604,7 @@ async function handlePiNative(
799
604
  try {
800
605
  if (controller.signal.aborted) return aborted();
801
606
  const message = await completeSimple(model, parsed.context, streamOpts);
802
- recordGatewayUsage(bootOpts.storage, model, client, message);
607
+ recordGatewayUsage(bootOpts.storage, model, client, message.usage, message.timestamp || undefined);
803
608
  if (message.stopReason === "aborted" || message.stopReason === "error") {
804
609
  const errorMessage =
805
610
  message.errorMessage ??
@@ -816,7 +621,11 @@ async function handlePiNative(
816
621
  const classified = classifyGatewayError(message.errorClassificationMessage ?? errorMessage);
817
622
  return piNative.formatError(classified.status, classified.type, errorMessage);
818
623
  }
819
- return json(200, { message }, gatewayResponseHeaders(model, { requestId, message, startedAt }));
624
+ return json(
625
+ 200,
626
+ { message },
627
+ gatewayResponseHeaders(model, { requestId, costUsd: message.usage.cost.total, startedAt }),
628
+ );
820
629
  } catch (error) {
821
630
  if (controller.signal.aborted) return aborted();
822
631
  const classified = classifyGatewayError(error);
@@ -846,7 +655,9 @@ async function handlePiNative(
846
655
  if (controller.signal.aborted) return aborted();
847
656
  void events
848
657
  .result()
849
- .then(message => recordGatewayUsage(bootOpts.storage, model, client, message))
658
+ .then(message =>
659
+ recordGatewayUsage(bootOpts.storage, model, client, message.usage, message.timestamp || undefined),
660
+ )
850
661
  .catch(() => {})
851
662
  .finally(() => lease.release());
852
663
  streamOwnsLease = true;
@@ -911,13 +722,16 @@ async function handleCredentialsCheck(storage: AuthStorage, signal: AbortSignal)
911
722
  * (omp's own proxy discovery, Zed's openai_compatible provider, ...) read to
912
723
  * size and capability-gate discovered models: `context_length`,
913
724
  * `max_output_tokens`, `input_modalities`, and `supports_tools` (only emitted
914
- * when the catalog explicitly reports `false`; absent means usable).
725
+ * when the catalog explicitly reports `false`; absent means usable). `kind` is
726
+ * emitted for non-chat rows (`judge`, `image`, `tts`, `stt`, `embedding`,
727
+ * `rerank`, `video`) so clients can keep them off chat routes; absent means chat.
915
728
  */
916
729
  interface ModelListRow {
917
730
  id: string;
918
731
  object: "model";
919
732
  owned_by: string;
920
733
  api: Api;
734
+ kind?: ModelKind;
921
735
  display_name: string;
922
736
  context_length?: number;
923
737
  max_output_tokens?: number;
@@ -940,6 +754,7 @@ function handleModelsList(opts: AuthGatewayBootOptions): Response {
940
754
  display_name: model.name,
941
755
  input_modalities: model.input,
942
756
  };
757
+ if (modelKind(model) !== "chat") row.kind = modelKind(model);
943
758
  if (model.contextWindow != null) row.context_length = model.contextWindow;
944
759
  if (model.maxTokens != null) row.max_output_tokens = model.maxTokens;
945
760
  if (model.supportsTools === false) row.supports_tools = false;
@@ -948,6 +763,9 @@ function handleModelsList(opts: AuthGatewayBootOptions): Response {
948
763
  return json(200, { object: "list", data });
949
764
  }
950
765
 
766
+ /** `GET /v1/videos/:id` (poll) and `GET /v1/videos/:id/content` (download); group 1 = id, group 2 = `/content`. */
767
+ const VIDEO_JOB_PATH = /^\/v1\/videos\/([^/]+)(\/content)?$/;
768
+
951
769
  export function startAuthGateway(opts: AuthGatewayBootOptions): AuthGatewayServerHandle {
952
770
  const bind = parseBind(opts.bind ?? DEFAULT_AUTH_GATEWAY_BIND);
953
771
  const tokens = new Set<string>(opts.bearerTokens);
@@ -1004,6 +822,55 @@ export function startAuthGateway(opts: AuthGatewayBootOptions): AuthGatewayServe
1004
822
  return withCors(await handlePiNative(opts, req, peer, sessionStates), req);
1005
823
  }
1006
824
 
825
+ // TypeSafe System One judgments (jev). TypeSafe SDKs and omp's own
826
+ // judge point `TYPESAFE_BASE_URL` at the gateway; OpenRouter SDKs
827
+ // reach the same handler through their Decisions path.
828
+ if (req.method === "POST" && (pathname === "/v1/systemone" || pathname === "/alpha/decisions")) {
829
+ return withCors(await handleSystemOne(opts, req, peer), req);
830
+ }
831
+
832
+ // Image generation: OpenAI `/v1/images/generations` + OpenRouter `/v1/images`
833
+ // (JSON), and OpenAI multipart / OpenRouter JSON edits.
834
+ if (req.method === "POST" && (pathname === "/v1/images/generations" || pathname === "/v1/images")) {
835
+ return withCors(await handleImageGenerations(opts, req, peer), req);
836
+ }
837
+ if (req.method === "POST" && pathname === "/v1/images/edits") {
838
+ return withCors(await handleImageEdits(opts, req, peer), req);
839
+ }
840
+
841
+ // Text-to-speech, OpenAI/OpenRouter wire; answers raw audio bytes.
842
+ if (req.method === "POST" && pathname === "/v1/audio/speech") {
843
+ return withCors(await handleSpeech(opts, req, peer), req);
844
+ }
845
+
846
+ // Speech-to-text, OpenAI multipart or OpenRouter JSON base64 wire.
847
+ if (req.method === "POST" && pathname === "/v1/audio/transcriptions") {
848
+ return withCors(await handleTranscriptions(opts, req, peer), req);
849
+ }
850
+
851
+ // Embeddings, OpenAI wire (OpenRouter is compatible).
852
+ if (req.method === "POST" && pathname === "/v1/embeddings") {
853
+ return withCors(await handleEmbeddings(opts, req, peer), req);
854
+ }
855
+
856
+ // Rerank, OpenRouter wire.
857
+ if (req.method === "POST" && pathname === "/v1/rerank") {
858
+ return withCors(await handleRerank(opts, req, peer), req);
859
+ }
860
+
861
+ // Video generation, OpenRouter's asynchronous wire: submit, then poll
862
+ // and download by the gateway-issued job id (stateless — the id
863
+ // encodes provider, model, and upstream job).
864
+ if (req.method === "POST" && pathname === "/v1/videos") {
865
+ return withCors(await handleVideoSubmit(opts, req, peer), req);
866
+ }
867
+ const videoJob = req.method === "GET" ? VIDEO_JOB_PATH.exec(pathname) : null;
868
+ if (videoJob) {
869
+ const gatewayId = decodeURIComponent(videoJob[1]);
870
+ const handler = videoJob[2] ? handleVideoContent : handleVideoPoll;
871
+ return withCors(await handler(opts, req, peer, gatewayId), req);
872
+ }
873
+
1007
874
  // Model catalog.
1008
875
  if (req.method === "GET" && pathname === "/v1/models") {
1009
876
  return withCors(handleModelsList(opts), req);
@@ -59,6 +59,11 @@ export interface AuthGatewayParsedRequestOptions {
59
59
  reasoning?: Effort;
60
60
  /** Force-disable reasoning (Anthropic `thinking: { type: "disabled" }`). */
61
61
  disableReasoning?: boolean;
62
+ /**
63
+ * Preserve an explicit wire-level reasoning-off request through providers
64
+ * that distinguish it from the generic disable hint.
65
+ */
66
+ forceReasoningOff?: boolean;
62
67
  /**
63
68
  * Explicit Anthropic `thinking.budget_tokens`. Mirrors Rust's
64
69
  * `resolve_thinking_budget`: pins onto whichever effort the client