@oh-my-pi/pi-ai 18.2.7 → 18.2.8

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (93) hide show
  1. package/CHANGELOG.md +16 -0
  2. package/dist/types/auth-gateway/dispatch.d.ts +80 -0
  3. package/dist/types/auth-gateway/http.d.ts +5 -6
  4. package/dist/types/auth-gateway/index.d.ts +1 -0
  5. package/dist/types/auth-gateway/routes/embeddings.d.ts +3 -0
  6. package/dist/types/auth-gateway/routes/images.d.ts +3 -0
  7. package/dist/types/auth-gateway/routes/rerank.d.ts +3 -0
  8. package/dist/types/auth-gateway/routes/speech.d.ts +2 -0
  9. package/dist/types/auth-gateway/routes/systemone.d.ts +2 -0
  10. package/dist/types/auth-gateway/routes/transcriptions.d.ts +3 -0
  11. package/dist/types/auth-gateway/routes/video.d.ts +7 -0
  12. package/dist/types/auth-gateway/server.d.ts +10 -16
  13. package/dist/types/embeddings/index.d.ts +7 -0
  14. package/dist/types/embeddings/openai-embeddings.d.ts +15 -0
  15. package/dist/types/embeddings/types.d.ts +15 -0
  16. package/dist/types/images/google-antigravity.d.ts +9 -0
  17. package/dist/types/images/google-generative-ai.d.ts +3 -0
  18. package/dist/types/images/index.d.ts +14 -0
  19. package/dist/types/images/openai-hosted.d.ts +3 -0
  20. package/dist/types/images/openai-images.d.ts +5 -0
  21. package/dist/types/images/openrouter-images.d.ts +3 -0
  22. package/dist/types/images/shared.d.ts +34 -0
  23. package/dist/types/images/types.d.ts +31 -0
  24. package/dist/types/index.d.ts +7 -1
  25. package/dist/types/judgment/typesafe.d.ts +2 -0
  26. package/dist/types/providers/embeddings-server.d.ts +32 -0
  27. package/dist/types/providers/images-server.d.ts +22 -0
  28. package/dist/types/providers/rerank-server.d.ts +35 -0
  29. package/dist/types/providers/speech-server.d.ts +8 -0
  30. package/dist/types/providers/systemone-server.d.ts +26 -0
  31. package/dist/types/providers/transcriptions-server.d.ts +32 -0
  32. package/dist/types/providers/video-server.d.ts +39 -0
  33. package/dist/types/rerank/index.d.ts +7 -0
  34. package/dist/types/rerank/openrouter-rerank.d.ts +15 -0
  35. package/dist/types/rerank/types.d.ts +17 -0
  36. package/dist/types/speech/index.d.ts +13 -0
  37. package/dist/types/speech/openai-speech.d.ts +3 -0
  38. package/dist/types/speech/transport.d.ts +7 -0
  39. package/dist/types/speech/types.d.ts +24 -0
  40. package/dist/types/speech/xai-tts.d.ts +7 -0
  41. package/dist/types/transcription/index.d.ts +7 -0
  42. package/dist/types/transcription/openai-transcriptions.d.ts +15 -0
  43. package/dist/types/transcription/types.d.ts +41 -0
  44. package/dist/types/video/index.d.ts +11 -0
  45. package/dist/types/video/openrouter-video.d.ts +19 -0
  46. package/dist/types/video/types.d.ts +62 -0
  47. package/package.json +30 -6
  48. package/src/auth-gateway/dispatch.ts +273 -0
  49. package/src/auth-gateway/http.ts +6 -7
  50. package/src/auth-gateway/index.ts +1 -0
  51. package/src/auth-gateway/routes/embeddings.ts +98 -0
  52. package/src/auth-gateway/routes/images.ts +131 -0
  53. package/src/auth-gateway/routes/rerank.ts +87 -0
  54. package/src/auth-gateway/routes/speech.ts +101 -0
  55. package/src/auth-gateway/routes/systemone.ts +116 -0
  56. package/src/auth-gateway/routes/transcriptions.ts +98 -0
  57. package/src/auth-gateway/routes/video.ts +243 -0
  58. package/src/auth-gateway/server.ts +123 -260
  59. package/src/embeddings/index.ts +17 -0
  60. package/src/embeddings/openai-embeddings.ts +141 -0
  61. package/src/embeddings/types.ts +14 -0
  62. package/src/error/rate-limit.ts +1 -1
  63. package/src/images/google-antigravity.ts +180 -0
  64. package/src/images/google-generative-ai.ts +92 -0
  65. package/src/images/index.ts +59 -0
  66. package/src/images/openai-hosted.ts +185 -0
  67. package/src/images/openai-images.ts +110 -0
  68. package/src/images/openrouter-images.ts +33 -0
  69. package/src/images/shared.ts +193 -0
  70. package/src/images/types.ts +36 -0
  71. package/src/index.ts +7 -1
  72. package/src/judgment/typesafe.ts +5 -0
  73. package/src/providers/embeddings-server.ts +151 -0
  74. package/src/providers/images-server.ts +159 -0
  75. package/src/providers/rerank-server.ts +166 -0
  76. package/src/providers/speech-server.ts +53 -0
  77. package/src/providers/systemone-server.ts +73 -0
  78. package/src/providers/transcriptions-server.ts +243 -0
  79. package/src/providers/video-server.ts +286 -0
  80. package/src/rerank/index.ts +13 -0
  81. package/src/rerank/openrouter-rerank.ts +136 -0
  82. package/src/rerank/types.ts +20 -0
  83. package/src/speech/index.ts +35 -0
  84. package/src/speech/openai-speech.ts +26 -0
  85. package/src/speech/transport.ts +66 -0
  86. package/src/speech/types.ts +37 -0
  87. package/src/speech/xai-tts.ts +41 -0
  88. package/src/transcription/index.ts +17 -0
  89. package/src/transcription/openai-transcriptions.ts +133 -0
  90. package/src/transcription/types.ts +46 -0
  91. package/src/video/index.ts +34 -0
  92. package/src/video/openrouter-video.ts +210 -0
  93. package/src/video/types.ts +72 -0
@@ -16,25 +16,38 @@
16
16
  * POST /v1/chat/completions → OpenAI chat-completions in/out
17
17
  * POST /v1/messages → Anthropic messages in/out
18
18
  * POST /v1/responses → OpenAI Responses in/out
19
+ * POST /v1/pi/stream → native pi-ai stream in/out
20
+ * POST /v1/systemone | /alpha/decisions → TypeSafe System One judgments (routes/systemone)
21
+ * POST /v1/images[/generations|/edits] → image generation, OpenAI/OpenRouter wire (routes/images)
22
+ * POST /v1/audio/speech → text-to-speech, raw audio out (routes/speech)
23
+ * POST /v1/audio/transcriptions → speech-to-text, multipart or JSON base64 in (routes/transcriptions)
24
+ *
25
+ * Chat routes live in this file; every other modality is a `routes/*` module
26
+ * built on the shared plumbing in `dispatch.ts`.
19
27
  */
20
28
 
21
29
  import { Effort } from "@oh-my-pi/pi-catalog/effort";
22
- import { extractHttpStatusFromError, logger } from "@oh-my-pi/pi-utils";
23
- import type { ApiKeyResolver } from "../auth-retry";
30
+ import { type ModelKind, modelKind } from "@oh-my-pi/pi-catalog/types";
31
+ import { logger } from "@oh-my-pi/pi-utils";
24
32
  import type { AuthStorage } from "../auth-storage";
25
- import * as AIError from "../error";
26
33
  import { classifyGatewayError } from "../error/gateway";
27
- import { isUsageLimitOutcome } from "../error/rate-limit";
28
34
  import * as anthropicMessages from "../providers/anthropic-messages-server";
29
35
  import * as openaiChat from "../providers/openai-chat-server";
30
36
  import * as openaiResponses from "../providers/openai-responses-server";
31
37
  import * as piNative from "../providers/pi-native-server";
32
38
  import { completeSimple, streamSimple } from "../stream";
33
- import type { Api, AssistantMessage, AssistantMessageEventStream, Context, Model, SimpleStreamOptions } from "../types";
34
- import type { ClientUsageIdentity } from "../usage";
39
+ import type { Api, AssistantMessageEventStream, Context, Model, SimpleStreamOptions } from "../types";
35
40
  import { deterministicUuid } from "../utils/deterministic-id";
36
41
  import { parseBind } from "../utils/parse-bind";
37
- import { extractProviderRetryHint } from "../utils/retry-after";
42
+ import {
43
+ type AuthGatewayBootOptions,
44
+ buildGatewayApiKeyResolver,
45
+ mirrorRequestAbort,
46
+ normalizeClientSessionKey,
47
+ recordGatewayUsage,
48
+ resolveGatewayAccount,
49
+ resolveGatewayApiKey,
50
+ } from "./dispatch";
38
51
  import {
39
52
  captureRequestHeaders,
40
53
  corsHeaders,
@@ -45,32 +58,21 @@ import {
45
58
  resolvePeer,
46
59
  withCors,
47
60
  } from "./http";
61
+ import { handleEmbeddings } from "./routes/embeddings";
62
+ import { handleImageEdits, handleImageGenerations } from "./routes/images";
63
+ import { handleRerank } from "./routes/rerank";
64
+ import { handleSpeech } from "./routes/speech";
65
+ import { handleSystemOne } from "./routes/systemone";
66
+ import { handleTranscriptions } from "./routes/transcriptions";
67
+ import { handleVideoContent, handleVideoPoll, handleVideoSubmit } from "./routes/video";
48
68
  import { AuthGatewaySessionStateStore } from "./session-state";
49
69
  import type {
50
70
  AuthGatewayServerHandle,
51
- AuthGatewayServerOptions,
52
71
  AuthGatewayFormatModule as FormatModule,
53
72
  AuthGatewayParsedRequest as ParsedFormatRequest,
54
73
  } from "./types";
55
74
  import { DEFAULT_AUTH_GATEWAY_BIND } from "./types";
56
75
 
57
- // ParsedFormatRequest / ParsedFormatOptions / FormatModule come from ./types.
58
-
59
- export type ModelResolver = (modelId: string) => Model<Api> | undefined;
60
-
61
- export interface AuthGatewayBootOptions extends AuthGatewayServerOptions {
62
- /** Source of credentials. Caller wires this to a broker-backed AuthStorage. */
63
- storage: AuthStorage;
64
- /**
65
- * Resolve a client-requested model id to a pi-ai Model. Caller supplies
66
- * this from a ModelRegistry (lives in `coding-agent` to avoid an inverse
67
- * dependency in `pi-ai`).
68
- */
69
- resolveModel: ModelResolver;
70
- /** Optional supplier for `/v1/models` listing. Returns the full model array. */
71
- listModels?: () => Iterable<Model<Api>>;
72
- }
73
-
74
76
  // `parseBind` lives in ../utils/parse-bind so the gateway and broker can't
75
77
  // drift on accepted inputs (e.g. empty hostname, IPv6 brackets).
76
78
 
@@ -130,40 +132,6 @@ function deriveSessionId(modelId: string, context: Context): string {
130
132
  return deterministicUuid(seed);
131
133
  }
132
134
 
133
- /**
134
- * The client's own session key, or `undefined` when it sent none. A blank key
135
- * counts as none: honouring it would collapse every caller that sends an empty
136
- * key into one shared credential-sticky, prefix-cache and provider-session
137
- * bucket.
138
- */
139
- function normalizeClientSessionKey(clientKey: string | undefined): string | undefined {
140
- return clientKey !== undefined && clientKey.trim().length > 0 ? clientKey : undefined;
141
- }
142
-
143
- /**
144
- * Stable identity of the account a request's credential belongs to.
145
- *
146
- * `markUsageLimitReached` and the auth-retry resolver switch a session to a
147
- * sibling credential, so the provider state retained for that session can
148
- * outlive the account that taught it. OAuth rows expose an account id / email
149
- * that survives token refresh — fingerprinting the bearer instead would look
150
- * like a rotation every time a token refreshes and discard the retained
151
- * lessons for nothing. Key-based rows fall back to a hash of the key, never
152
- * the key itself: this value is held for the lifetime of the entry.
153
- */
154
- function resolveGatewayAccount(storage: AuthStorage, provider: string, sessionId: string, apiKey: string): string {
155
- const identity = storage.getOAuthAccountIdentity(provider, sessionId);
156
- if (identity) {
157
- return `oauth:${JSON.stringify([
158
- identity.accountId ?? "",
159
- identity.email ?? "",
160
- identity.projectId ?? "",
161
- identity.orgId ?? "",
162
- ])}`;
163
- }
164
- return `key:${Bun.hash(apiKey).toString(36)}`;
165
- }
166
-
167
135
  function buildStreamOptions(parsed: ParsedFormatRequest, api: Api, signal: AbortSignal): SimpleStreamOptions {
168
136
  const opts: SimpleStreamOptions = { signal, cursorExternalToolExecutor: true };
169
137
  const { options } = parsed;
@@ -249,167 +217,26 @@ function buildStreamOptions(parsed: ParsedFormatRequest, api: Api, signal: Abort
249
217
  return opts;
250
218
  }
251
219
 
252
- /**
253
- * Hook fired by {@link streamSimple} when the upstream request fails in a
254
- * way that's rotatable — today that's HTTP 401 (credential is bad) and
255
- * usage-limit phrasing matched by {@link isUsageLimitError} (Codex's
256
- * `usage_limit_reached`, Anthropic's `usage_limit_reached`, Google's
257
- * `resource_exhausted`, …). The two cases need different storage actions:
258
- *
259
- * - **usage-limit** → {@link AuthStorage.markUsageLimitReached}. Marks just
260
- * the current session's credential as temporarily blocked (honouring
261
- * `retry-after` / `resets_at` hints when present) and returns `true` only
262
- * when a sibling credential is still available. Burning the credential
263
- * with `invalidateCredentialMatching` here would orphan accounts whose
264
- * reset window is several hours away — exactly the bug this helper exists
265
- * to avoid.
266
- * - **auth-failure** → {@link AuthStorage.invalidateCredentialMatching}.
267
- * Suspect/delete the row so it doesn't get re-picked next request.
268
- *
269
- * In both branches we return the next `getApiKey` result (sticky on the
270
- * same `sessionId`) so streamSimple can transparently retry the pre-emit
271
- * failure with a fresh credential. Returning `undefined` aborts the retry
272
- * and surfaces the original error to the caller.
273
- */
274
- async function refreshGatewayApiKeyAfterAuthError(
275
- storage: AuthStorage,
276
- model: Model<Api>,
277
- sessionId: string,
278
- provider: string,
279
- oldKey: string,
280
- error: unknown,
281
- signal: AbortSignal,
282
- format: string,
283
- peer: string,
284
- ): Promise<string | undefined> {
285
- const message = error instanceof Error ? error.message : String(error);
286
- const status = extractHttpStatusFromError(error);
287
- if (AIError.isUsageLimit(error) || isUsageLimitOutcome(status, message)) {
288
- const retryAfterMs = extractProviderRetryHint(provider, message);
289
- const { switched, retryAtMs } = await storage.markUsageLimitReached(provider, sessionId, {
290
- retryAfterMs,
291
- providerTimed: retryAfterMs !== undefined,
292
- baseUrl: model.baseUrl,
293
- modelId: model.id,
294
- apiKey: oldKey,
295
- signal,
296
- });
297
- logger.debug("auth-gateway retrying provider request after usage-limit block", {
298
- format,
299
- provider,
300
- peer,
301
- switched,
302
- retryAfterMs,
303
- retryAtMs,
304
- error: message,
305
- });
306
- if (!switched) return undefined;
307
- return storage.getApiKey(provider, sessionId, { modelId: model.id, signal });
308
- }
309
- await storage.invalidateCredentialMatching(provider, oldKey, { sessionId, signal });
310
- logger.debug("auth-gateway retrying provider request after credential invalidation", {
311
- format,
312
- provider,
313
- peer,
314
- error: message,
315
- });
316
- return storage.getApiKey(provider, sessionId, { modelId: model.id, signal });
317
- }
318
-
319
- /**
320
- * Build the {@link ApiKeyResolver} handed to `streamSimple` for a gateway
321
- * request. Drives the central a/b/c auth-retry policy server-side:
322
- *
323
- * - initial resolve → the credential already resolved for this request.
324
- * - step (b) `!lastChance` → force-refresh the SAME session-sticky credential
325
- * (a peer/broker may have rotated its token out from under our cached copy).
326
- * - step (c) `lastChance` → {@link refreshGatewayApiKeyAfterAuthError} switches
327
- * to a sibling (usage-limit block vs credential invalidation by error class).
328
- *
329
- * `lastKey` tracks the most recent bearer so the switch step invalidates the
330
- * credential that actually failed.
331
- */
332
- function buildGatewayApiKeyResolver(
333
- storage: AuthStorage,
334
- model: Model<Api>,
335
- sessionId: string,
336
- initialKey: string,
337
- requestSignal: AbortSignal,
338
- format: string,
339
- peer: string,
340
- onResolvedKey: (apiKey: string) => void,
341
- ): ApiKeyResolver {
342
- let lastKey = initialKey;
343
- return async ({ lastChance, error, signal }) => {
344
- const sig = signal ?? requestSignal;
345
- if (error === undefined) {
346
- lastKey = initialKey;
347
- return initialKey;
348
- }
349
- if (!lastChance) {
350
- const refreshed = await storage.getApiKey(model.provider, sessionId, {
351
- modelId: model.id,
352
- signal: sig,
353
- forceRefresh: true,
354
- });
355
- lastKey = refreshed ?? lastKey;
356
- if (refreshed) onResolvedKey(refreshed);
357
- return refreshed;
358
- }
359
- const next = await refreshGatewayApiKeyAfterAuthError(
360
- storage,
361
- model,
362
- sessionId,
363
- model.provider,
364
- lastKey,
365
- error,
366
- sig,
367
- format,
368
- peer,
369
- );
370
- lastKey = next ?? lastKey;
371
- if (next) onResolvedKey(next);
372
- return next;
373
- };
374
- }
375
-
376
220
  function clientClosedResponse(route: { module: FormatModule }): Response {
377
221
  return route.module.formatError(499, "request_aborted", "client closed request");
378
222
  }
379
223
 
380
- /**
381
- * Attribute one settled upstream request to the originating client via the
382
- * broker's observed-usage channel (`AuthStorage.recordObservedUsage`, batched
383
- * by the remote store). Error/aborted turns still record — the provider
384
- * billed whatever tokens the partial turn consumed; zero-usage messages
385
- * (pre-flight failures) are skipped.
386
- */
387
- function recordGatewayUsage(
388
- storage: AuthStorage,
389
- model: Model<Api>,
390
- client: ClientUsageIdentity,
391
- message: AssistantMessage,
392
- ): void {
393
- const usage = message.usage;
394
- if (usage.input + usage.output + usage.cacheRead + usage.cacheWrite === 0) return;
395
- storage.recordObservedUsage({
396
- provider: model.provider,
397
- model: model.id,
398
- at: message.timestamp || Date.now(),
399
- usage: { input: usage.input, output: usage.output, cacheRead: usage.cacheRead, cacheWrite: usage.cacheWrite },
400
- costUsd: usage.cost.total,
401
- client,
402
- });
403
- }
224
+ /** Route that serves each non-chat catalog kind the gateway advertises. */
225
+ const KIND_ROUTES: Partial<Record<ModelKind, string>> = {
226
+ judge: "POST /v1/systemone",
227
+ image: "POST /v1/images/generations",
228
+ tts: "POST /v1/audio/speech",
229
+ stt: "POST /v1/audio/transcriptions",
230
+ embedding: "POST /v1/embeddings",
231
+ rerank: "POST /v1/rerank",
232
+ video: "POST /v1/videos",
233
+ };
404
234
 
405
- function mirrorRequestAbort(req: Request): AbortController {
406
- const controller = new AbortController();
407
- if (req.signal.aborted) {
408
- controller.abort(req.signal.reason);
409
- } else {
410
- req.signal.addEventListener("abort", () => controller.abort(req.signal.reason), { once: true });
411
- }
412
- return controller;
235
+ /** Chat routes cannot drive a non-chat model; name the route that does, or `undefined` for chat models. */
236
+ function chatRouteRejection(model: Model<Api>): string | undefined {
237
+ const kind = modelKind(model);
238
+ const route = KIND_ROUTES[kind];
239
+ return route && `Model ${model.provider}/${model.id} is a ${kind} model; use ${route}`;
413
240
  }
414
241
 
415
242
  // (handlePassthrough removed — see note above.)
@@ -450,6 +277,8 @@ async function handleFormatEndpoint(
450
277
  if (!model) {
451
278
  return route.module.formatError(404, "invalid_request_error", `Unknown model: ${modelId}`);
452
279
  }
280
+ const kindRejection = chatRouteRejection(model);
281
+ if (kindRejection) return route.module.formatError(400, "invalid_request_error", kindRejection);
453
282
  const client = resolveClientIdentity(req.headers);
454
283
 
455
284
  // Parse the wire-format request BEFORE resolving the credential so we
@@ -510,28 +339,12 @@ async function handleFormatEndpoint(
510
339
  // expected to resolve the credential and pass it as `options.apiKey`.
511
340
  // For OAuth providers this returns the access token (refreshed via the
512
341
  // broker override on AuthStorage when needed).
513
- let apiKey: string | undefined;
514
- try {
515
- apiKey = await bootOpts.storage.getApiKey(model.provider, sessionId, {
516
- modelId: model.id,
517
- signal: controller.signal,
518
- });
519
- } catch (error) {
520
- if (controller.signal.aborted) return clientClosedResponse(route);
521
- const classified = classifyGatewayError(error);
522
- logger.warn("auth-gateway getApiKey threw", { provider: model.provider, peer, error: classified.message });
523
- return route.module.formatError(classified.status, classified.type, classified.message);
524
- }
342
+ const apiKey = await resolveGatewayApiKey(bootOpts.storage, model, sessionId, controller.signal, peer);
525
343
  if (controller.signal.aborted) return clientClosedResponse(route);
526
- if (!apiKey) {
527
- return route.module.formatError(
528
- 401,
529
- "authentication_error",
530
- `No credential available for provider ${model.provider}`,
531
- );
532
- }
344
+ if (typeof apiKey !== "string") return route.module.formatError(apiKey.status, apiKey.type, apiKey.message);
533
345
 
534
346
  const streamOpts = buildStreamOptions(parsed, model.api, controller.signal);
347
+ if (bootOpts.fetch) streamOpts.fetch = bootOpts.fetch;
535
348
  // Per-session provider learning (sticky strict-tools / fast-mode / thinking
536
349
  // fallbacks, Codex transport sessions). Owned by this gateway instance: the
537
350
  // map is non-serializable, so no client can supply it and every turn would
@@ -571,7 +384,7 @@ async function handleFormatEndpoint(
571
384
  try {
572
385
  if (controller.signal.aborted) return clientClosedResponse(route);
573
386
  const message = await completeSimple(model, parsed.context, streamOpts);
574
- recordGatewayUsage(bootOpts.storage, model, client, message);
387
+ recordGatewayUsage(bootOpts.storage, model, client, message.usage, message.timestamp || undefined);
575
388
  if (message.stopReason === "aborted" || message.stopReason === "error") {
576
389
  const errorMessage =
577
390
  message.errorMessage ??
@@ -591,7 +404,7 @@ async function handleFormatEndpoint(
591
404
  return json(
592
405
  200,
593
406
  route.module.encodeResponse(message, parsed.modelId),
594
- gatewayResponseHeaders(model, { requestId, message, startedAt }),
407
+ gatewayResponseHeaders(model, { requestId, costUsd: message.usage.cost.total, startedAt }),
595
408
  );
596
409
  } catch (error) {
597
410
  if (controller.signal.aborted) return clientClosedResponse(route);
@@ -626,7 +439,9 @@ async function handleFormatEndpoint(
626
439
  if (controller.signal.aborted) return clientClosedResponse(route);
627
440
  void events
628
441
  .result()
629
- .then(message => recordGatewayUsage(bootOpts.storage, model, client, message))
442
+ .then(message =>
443
+ recordGatewayUsage(bootOpts.storage, model, client, message.usage, message.timestamp || undefined),
444
+ )
630
445
  .catch(() => {})
631
446
  .finally(() => lease.release());
632
447
  streamOwnsLease = true;
@@ -705,6 +520,8 @@ async function handlePiNative(
705
520
  if (!model) {
706
521
  return piNative.formatError(404, "invalid_request_error", `Unknown model: ${parsed.modelId}`);
707
522
  }
523
+ const kindRejection = chatRouteRejection(model);
524
+ if (kindRejection) return piNative.formatError(400, "invalid_request_error", kindRejection);
708
525
  const client = resolveClientIdentity(req.headers);
709
526
  // Pi-native already parsed `streamOpts.sessionId` (when set by the
710
527
  // client); fall back to the derived key so credential-stickiness lines
@@ -715,26 +532,9 @@ async function handlePiNative(
715
532
  const sessionId = clientKey ?? deriveSessionId(parsed.modelId, parsed.context);
716
533
  parsed.options.sessionId = sessionId;
717
534
 
718
- let apiKey: string | undefined;
719
- try {
720
- apiKey = await bootOpts.storage.getApiKey(model.provider, sessionId, {
721
- modelId: model.id,
722
- signal: controller.signal,
723
- });
724
- } catch (error) {
725
- if (controller.signal.aborted) return aborted();
726
- const classified = classifyGatewayError(error);
727
- logger.warn("auth-gateway getApiKey threw", { provider: model.provider, peer, error: classified.message });
728
- return piNative.formatError(classified.status, classified.type, classified.message);
729
- }
535
+ const apiKey = await resolveGatewayApiKey(bootOpts.storage, model, sessionId, controller.signal, peer);
730
536
  if (controller.signal.aborted) return aborted();
731
- if (!apiKey) {
732
- return piNative.formatError(
733
- 401,
734
- "authentication_error",
735
- `No credential available for provider ${model.provider}`,
736
- );
737
- }
537
+ if (typeof apiKey !== "string") return piNative.formatError(apiKey.status, apiKey.type, apiKey.message);
738
538
 
739
539
  // Per-session provider learning, owned by this gateway instance. The map is
740
540
  // non-serializable, so `parseRequest` cannot accept one from the wire and
@@ -758,6 +558,7 @@ async function handlePiNative(
758
558
  cursorExternalToolExecutor: true,
759
559
  providerSessionState: lease.states,
760
560
  };
561
+ if (bootOpts.fetch) streamOpts.fetch = bootOpts.fetch;
761
562
  streamOpts.apiKey = buildGatewayApiKeyResolver(
762
563
  bootOpts.storage,
763
564
  model,
@@ -799,7 +600,7 @@ async function handlePiNative(
799
600
  try {
800
601
  if (controller.signal.aborted) return aborted();
801
602
  const message = await completeSimple(model, parsed.context, streamOpts);
802
- recordGatewayUsage(bootOpts.storage, model, client, message);
603
+ recordGatewayUsage(bootOpts.storage, model, client, message.usage, message.timestamp || undefined);
803
604
  if (message.stopReason === "aborted" || message.stopReason === "error") {
804
605
  const errorMessage =
805
606
  message.errorMessage ??
@@ -816,7 +617,11 @@ async function handlePiNative(
816
617
  const classified = classifyGatewayError(message.errorClassificationMessage ?? errorMessage);
817
618
  return piNative.formatError(classified.status, classified.type, errorMessage);
818
619
  }
819
- return json(200, { message }, gatewayResponseHeaders(model, { requestId, message, startedAt }));
620
+ return json(
621
+ 200,
622
+ { message },
623
+ gatewayResponseHeaders(model, { requestId, costUsd: message.usage.cost.total, startedAt }),
624
+ );
820
625
  } catch (error) {
821
626
  if (controller.signal.aborted) return aborted();
822
627
  const classified = classifyGatewayError(error);
@@ -846,7 +651,9 @@ async function handlePiNative(
846
651
  if (controller.signal.aborted) return aborted();
847
652
  void events
848
653
  .result()
849
- .then(message => recordGatewayUsage(bootOpts.storage, model, client, message))
654
+ .then(message =>
655
+ recordGatewayUsage(bootOpts.storage, model, client, message.usage, message.timestamp || undefined),
656
+ )
850
657
  .catch(() => {})
851
658
  .finally(() => lease.release());
852
659
  streamOwnsLease = true;
@@ -911,13 +718,16 @@ async function handleCredentialsCheck(storage: AuthStorage, signal: AbortSignal)
911
718
  * (omp's own proxy discovery, Zed's openai_compatible provider, ...) read to
912
719
  * size and capability-gate discovered models: `context_length`,
913
720
  * `max_output_tokens`, `input_modalities`, and `supports_tools` (only emitted
914
- * when the catalog explicitly reports `false`; absent means usable).
721
+ * when the catalog explicitly reports `false`; absent means usable). `kind` is
722
+ * emitted for non-chat rows (`judge`, `image`, `tts`, `stt`, `embedding`,
723
+ * `rerank`, `video`) so clients can keep them off chat routes; absent means chat.
915
724
  */
916
725
  interface ModelListRow {
917
726
  id: string;
918
727
  object: "model";
919
728
  owned_by: string;
920
729
  api: Api;
730
+ kind?: ModelKind;
921
731
  display_name: string;
922
732
  context_length?: number;
923
733
  max_output_tokens?: number;
@@ -940,6 +750,7 @@ function handleModelsList(opts: AuthGatewayBootOptions): Response {
940
750
  display_name: model.name,
941
751
  input_modalities: model.input,
942
752
  };
753
+ if (modelKind(model) !== "chat") row.kind = modelKind(model);
943
754
  if (model.contextWindow != null) row.context_length = model.contextWindow;
944
755
  if (model.maxTokens != null) row.max_output_tokens = model.maxTokens;
945
756
  if (model.supportsTools === false) row.supports_tools = false;
@@ -948,6 +759,9 @@ function handleModelsList(opts: AuthGatewayBootOptions): Response {
948
759
  return json(200, { object: "list", data });
949
760
  }
950
761
 
762
+ /** `GET /v1/videos/:id` (poll) and `GET /v1/videos/:id/content` (download); group 1 = id, group 2 = `/content`. */
763
+ const VIDEO_JOB_PATH = /^\/v1\/videos\/([^/]+)(\/content)?$/;
764
+
951
765
  export function startAuthGateway(opts: AuthGatewayBootOptions): AuthGatewayServerHandle {
952
766
  const bind = parseBind(opts.bind ?? DEFAULT_AUTH_GATEWAY_BIND);
953
767
  const tokens = new Set<string>(opts.bearerTokens);
@@ -1004,6 +818,55 @@ export function startAuthGateway(opts: AuthGatewayBootOptions): AuthGatewayServe
1004
818
  return withCors(await handlePiNative(opts, req, peer, sessionStates), req);
1005
819
  }
1006
820
 
821
+ // TypeSafe System One judgments (jev). TypeSafe SDKs and omp's own
822
+ // judge point `TYPESAFE_BASE_URL` at the gateway; OpenRouter SDKs
823
+ // reach the same handler through their Decisions path.
824
+ if (req.method === "POST" && (pathname === "/v1/systemone" || pathname === "/alpha/decisions")) {
825
+ return withCors(await handleSystemOne(opts, req, peer), req);
826
+ }
827
+
828
+ // Image generation: OpenAI `/v1/images/generations` + OpenRouter `/v1/images`
829
+ // (JSON), and OpenAI multipart / OpenRouter JSON edits.
830
+ if (req.method === "POST" && (pathname === "/v1/images/generations" || pathname === "/v1/images")) {
831
+ return withCors(await handleImageGenerations(opts, req, peer), req);
832
+ }
833
+ if (req.method === "POST" && pathname === "/v1/images/edits") {
834
+ return withCors(await handleImageEdits(opts, req, peer), req);
835
+ }
836
+
837
+ // Text-to-speech, OpenAI/OpenRouter wire; answers raw audio bytes.
838
+ if (req.method === "POST" && pathname === "/v1/audio/speech") {
839
+ return withCors(await handleSpeech(opts, req, peer), req);
840
+ }
841
+
842
+ // Speech-to-text, OpenAI multipart or OpenRouter JSON base64 wire.
843
+ if (req.method === "POST" && pathname === "/v1/audio/transcriptions") {
844
+ return withCors(await handleTranscriptions(opts, req, peer), req);
845
+ }
846
+
847
+ // Embeddings, OpenAI wire (OpenRouter is compatible).
848
+ if (req.method === "POST" && pathname === "/v1/embeddings") {
849
+ return withCors(await handleEmbeddings(opts, req, peer), req);
850
+ }
851
+
852
+ // Rerank, OpenRouter wire.
853
+ if (req.method === "POST" && pathname === "/v1/rerank") {
854
+ return withCors(await handleRerank(opts, req, peer), req);
855
+ }
856
+
857
+ // Video generation, OpenRouter's asynchronous wire: submit, then poll
858
+ // and download by the gateway-issued job id (stateless — the id
859
+ // encodes provider, model, and upstream job).
860
+ if (req.method === "POST" && pathname === "/v1/videos") {
861
+ return withCors(await handleVideoSubmit(opts, req, peer), req);
862
+ }
863
+ const videoJob = req.method === "GET" ? VIDEO_JOB_PATH.exec(pathname) : null;
864
+ if (videoJob) {
865
+ const gatewayId = decodeURIComponent(videoJob[1]);
866
+ const handler = videoJob[2] ? handleVideoContent : handleVideoPoll;
867
+ return withCors(await handler(opts, req, peer, gatewayId), req);
868
+ }
869
+
1007
870
  // Model catalog.
1008
871
  if (req.method === "GET" && pathname === "/v1/models") {
1009
872
  return withCors(handleModelsList(opts), req);
@@ -0,0 +1,17 @@
1
+ import type { Api, Model } from "@oh-my-pi/pi-catalog/types";
2
+ import * as AIError from "../error";
3
+ import { embedOpenAI, type EmbeddingOptions } from "./openai-embeddings";
4
+ import type { EmbeddingRequest, EmbeddingResult } from "./types";
5
+
6
+ export * from "./openai-embeddings";
7
+ export * from "./types";
8
+
9
+ /** Dispatch an embedding request through the transport selected by the catalog model. */
10
+ export function embed(
11
+ model: Model<Api>,
12
+ request: EmbeddingRequest,
13
+ options: EmbeddingOptions,
14
+ ): Promise<EmbeddingResult> {
15
+ if (model.api === "openai-embeddings") return embedOpenAI(model, request, options);
16
+ throw new AIError.ConfigurationError(`Unsupported embeddings API: ${model.api}`);
17
+ }