@oh-my-pi/pi-ai 18.2.7 → 18.2.8
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +16 -0
- package/dist/types/auth-gateway/dispatch.d.ts +80 -0
- package/dist/types/auth-gateway/http.d.ts +5 -6
- package/dist/types/auth-gateway/index.d.ts +1 -0
- package/dist/types/auth-gateway/routes/embeddings.d.ts +3 -0
- package/dist/types/auth-gateway/routes/images.d.ts +3 -0
- package/dist/types/auth-gateway/routes/rerank.d.ts +3 -0
- package/dist/types/auth-gateway/routes/speech.d.ts +2 -0
- package/dist/types/auth-gateway/routes/systemone.d.ts +2 -0
- package/dist/types/auth-gateway/routes/transcriptions.d.ts +3 -0
- package/dist/types/auth-gateway/routes/video.d.ts +7 -0
- package/dist/types/auth-gateway/server.d.ts +10 -16
- package/dist/types/embeddings/index.d.ts +7 -0
- package/dist/types/embeddings/openai-embeddings.d.ts +15 -0
- package/dist/types/embeddings/types.d.ts +15 -0
- package/dist/types/images/google-antigravity.d.ts +9 -0
- package/dist/types/images/google-generative-ai.d.ts +3 -0
- package/dist/types/images/index.d.ts +14 -0
- package/dist/types/images/openai-hosted.d.ts +3 -0
- package/dist/types/images/openai-images.d.ts +5 -0
- package/dist/types/images/openrouter-images.d.ts +3 -0
- package/dist/types/images/shared.d.ts +34 -0
- package/dist/types/images/types.d.ts +31 -0
- package/dist/types/index.d.ts +7 -1
- package/dist/types/judgment/typesafe.d.ts +2 -0
- package/dist/types/providers/embeddings-server.d.ts +32 -0
- package/dist/types/providers/images-server.d.ts +22 -0
- package/dist/types/providers/rerank-server.d.ts +35 -0
- package/dist/types/providers/speech-server.d.ts +8 -0
- package/dist/types/providers/systemone-server.d.ts +26 -0
- package/dist/types/providers/transcriptions-server.d.ts +32 -0
- package/dist/types/providers/video-server.d.ts +39 -0
- package/dist/types/rerank/index.d.ts +7 -0
- package/dist/types/rerank/openrouter-rerank.d.ts +15 -0
- package/dist/types/rerank/types.d.ts +17 -0
- package/dist/types/speech/index.d.ts +13 -0
- package/dist/types/speech/openai-speech.d.ts +3 -0
- package/dist/types/speech/transport.d.ts +7 -0
- package/dist/types/speech/types.d.ts +24 -0
- package/dist/types/speech/xai-tts.d.ts +7 -0
- package/dist/types/transcription/index.d.ts +7 -0
- package/dist/types/transcription/openai-transcriptions.d.ts +15 -0
- package/dist/types/transcription/types.d.ts +41 -0
- package/dist/types/video/index.d.ts +11 -0
- package/dist/types/video/openrouter-video.d.ts +19 -0
- package/dist/types/video/types.d.ts +62 -0
- package/package.json +30 -6
- package/src/auth-gateway/dispatch.ts +273 -0
- package/src/auth-gateway/http.ts +6 -7
- package/src/auth-gateway/index.ts +1 -0
- package/src/auth-gateway/routes/embeddings.ts +98 -0
- package/src/auth-gateway/routes/images.ts +131 -0
- package/src/auth-gateway/routes/rerank.ts +87 -0
- package/src/auth-gateway/routes/speech.ts +101 -0
- package/src/auth-gateway/routes/systemone.ts +116 -0
- package/src/auth-gateway/routes/transcriptions.ts +98 -0
- package/src/auth-gateway/routes/video.ts +243 -0
- package/src/auth-gateway/server.ts +123 -260
- package/src/embeddings/index.ts +17 -0
- package/src/embeddings/openai-embeddings.ts +141 -0
- package/src/embeddings/types.ts +14 -0
- package/src/error/rate-limit.ts +1 -1
- package/src/images/google-antigravity.ts +180 -0
- package/src/images/google-generative-ai.ts +92 -0
- package/src/images/index.ts +59 -0
- package/src/images/openai-hosted.ts +185 -0
- package/src/images/openai-images.ts +110 -0
- package/src/images/openrouter-images.ts +33 -0
- package/src/images/shared.ts +193 -0
- package/src/images/types.ts +36 -0
- package/src/index.ts +7 -1
- package/src/judgment/typesafe.ts +5 -0
- package/src/providers/embeddings-server.ts +151 -0
- package/src/providers/images-server.ts +159 -0
- package/src/providers/rerank-server.ts +166 -0
- package/src/providers/speech-server.ts +53 -0
- package/src/providers/systemone-server.ts +73 -0
- package/src/providers/transcriptions-server.ts +243 -0
- package/src/providers/video-server.ts +286 -0
- package/src/rerank/index.ts +13 -0
- package/src/rerank/openrouter-rerank.ts +136 -0
- package/src/rerank/types.ts +20 -0
- package/src/speech/index.ts +35 -0
- package/src/speech/openai-speech.ts +26 -0
- package/src/speech/transport.ts +66 -0
- package/src/speech/types.ts +37 -0
- package/src/speech/xai-tts.ts +41 -0
- package/src/transcription/index.ts +17 -0
- package/src/transcription/openai-transcriptions.ts +133 -0
- package/src/transcription/types.ts +46 -0
- package/src/video/index.ts +34 -0
- package/src/video/openrouter-video.ts +210 -0
- package/src/video/types.ts +72 -0
|
@@ -16,25 +16,38 @@
|
|
|
16
16
|
* POST /v1/chat/completions → OpenAI chat-completions in/out
|
|
17
17
|
* POST /v1/messages → Anthropic messages in/out
|
|
18
18
|
* POST /v1/responses → OpenAI Responses in/out
|
|
19
|
+
* POST /v1/pi/stream → native pi-ai stream in/out
|
|
20
|
+
* POST /v1/systemone | /alpha/decisions → TypeSafe System One judgments (routes/systemone)
|
|
21
|
+
* POST /v1/images[/generations|/edits] → image generation, OpenAI/OpenRouter wire (routes/images)
|
|
22
|
+
* POST /v1/audio/speech → text-to-speech, raw audio out (routes/speech)
|
|
23
|
+
* POST /v1/audio/transcriptions → speech-to-text, multipart or JSON base64 in (routes/transcriptions)
|
|
24
|
+
*
|
|
25
|
+
* Chat routes live in this file; every other modality is a `routes/*` module
|
|
26
|
+
* built on the shared plumbing in `dispatch.ts`.
|
|
19
27
|
*/
|
|
20
28
|
|
|
21
29
|
import { Effort } from "@oh-my-pi/pi-catalog/effort";
|
|
22
|
-
import {
|
|
23
|
-
import
|
|
30
|
+
import { type ModelKind, modelKind } from "@oh-my-pi/pi-catalog/types";
|
|
31
|
+
import { logger } from "@oh-my-pi/pi-utils";
|
|
24
32
|
import type { AuthStorage } from "../auth-storage";
|
|
25
|
-
import * as AIError from "../error";
|
|
26
33
|
import { classifyGatewayError } from "../error/gateway";
|
|
27
|
-
import { isUsageLimitOutcome } from "../error/rate-limit";
|
|
28
34
|
import * as anthropicMessages from "../providers/anthropic-messages-server";
|
|
29
35
|
import * as openaiChat from "../providers/openai-chat-server";
|
|
30
36
|
import * as openaiResponses from "../providers/openai-responses-server";
|
|
31
37
|
import * as piNative from "../providers/pi-native-server";
|
|
32
38
|
import { completeSimple, streamSimple } from "../stream";
|
|
33
|
-
import type { Api,
|
|
34
|
-
import type { ClientUsageIdentity } from "../usage";
|
|
39
|
+
import type { Api, AssistantMessageEventStream, Context, Model, SimpleStreamOptions } from "../types";
|
|
35
40
|
import { deterministicUuid } from "../utils/deterministic-id";
|
|
36
41
|
import { parseBind } from "../utils/parse-bind";
|
|
37
|
-
import {
|
|
42
|
+
import {
|
|
43
|
+
type AuthGatewayBootOptions,
|
|
44
|
+
buildGatewayApiKeyResolver,
|
|
45
|
+
mirrorRequestAbort,
|
|
46
|
+
normalizeClientSessionKey,
|
|
47
|
+
recordGatewayUsage,
|
|
48
|
+
resolveGatewayAccount,
|
|
49
|
+
resolveGatewayApiKey,
|
|
50
|
+
} from "./dispatch";
|
|
38
51
|
import {
|
|
39
52
|
captureRequestHeaders,
|
|
40
53
|
corsHeaders,
|
|
@@ -45,32 +58,21 @@ import {
|
|
|
45
58
|
resolvePeer,
|
|
46
59
|
withCors,
|
|
47
60
|
} from "./http";
|
|
61
|
+
import { handleEmbeddings } from "./routes/embeddings";
|
|
62
|
+
import { handleImageEdits, handleImageGenerations } from "./routes/images";
|
|
63
|
+
import { handleRerank } from "./routes/rerank";
|
|
64
|
+
import { handleSpeech } from "./routes/speech";
|
|
65
|
+
import { handleSystemOne } from "./routes/systemone";
|
|
66
|
+
import { handleTranscriptions } from "./routes/transcriptions";
|
|
67
|
+
import { handleVideoContent, handleVideoPoll, handleVideoSubmit } from "./routes/video";
|
|
48
68
|
import { AuthGatewaySessionStateStore } from "./session-state";
|
|
49
69
|
import type {
|
|
50
70
|
AuthGatewayServerHandle,
|
|
51
|
-
AuthGatewayServerOptions,
|
|
52
71
|
AuthGatewayFormatModule as FormatModule,
|
|
53
72
|
AuthGatewayParsedRequest as ParsedFormatRequest,
|
|
54
73
|
} from "./types";
|
|
55
74
|
import { DEFAULT_AUTH_GATEWAY_BIND } from "./types";
|
|
56
75
|
|
|
57
|
-
// ParsedFormatRequest / ParsedFormatOptions / FormatModule come from ./types.
|
|
58
|
-
|
|
59
|
-
export type ModelResolver = (modelId: string) => Model<Api> | undefined;
|
|
60
|
-
|
|
61
|
-
export interface AuthGatewayBootOptions extends AuthGatewayServerOptions {
|
|
62
|
-
/** Source of credentials. Caller wires this to a broker-backed AuthStorage. */
|
|
63
|
-
storage: AuthStorage;
|
|
64
|
-
/**
|
|
65
|
-
* Resolve a client-requested model id to a pi-ai Model. Caller supplies
|
|
66
|
-
* this from a ModelRegistry (lives in `coding-agent` to avoid an inverse
|
|
67
|
-
* dependency in `pi-ai`).
|
|
68
|
-
*/
|
|
69
|
-
resolveModel: ModelResolver;
|
|
70
|
-
/** Optional supplier for `/v1/models` listing. Returns the full model array. */
|
|
71
|
-
listModels?: () => Iterable<Model<Api>>;
|
|
72
|
-
}
|
|
73
|
-
|
|
74
76
|
// `parseBind` lives in ../utils/parse-bind so the gateway and broker can't
|
|
75
77
|
// drift on accepted inputs (e.g. empty hostname, IPv6 brackets).
|
|
76
78
|
|
|
@@ -130,40 +132,6 @@ function deriveSessionId(modelId: string, context: Context): string {
|
|
|
130
132
|
return deterministicUuid(seed);
|
|
131
133
|
}
|
|
132
134
|
|
|
133
|
-
/**
|
|
134
|
-
* The client's own session key, or `undefined` when it sent none. A blank key
|
|
135
|
-
* counts as none: honouring it would collapse every caller that sends an empty
|
|
136
|
-
* key into one shared credential-sticky, prefix-cache and provider-session
|
|
137
|
-
* bucket.
|
|
138
|
-
*/
|
|
139
|
-
function normalizeClientSessionKey(clientKey: string | undefined): string | undefined {
|
|
140
|
-
return clientKey !== undefined && clientKey.trim().length > 0 ? clientKey : undefined;
|
|
141
|
-
}
|
|
142
|
-
|
|
143
|
-
/**
|
|
144
|
-
* Stable identity of the account a request's credential belongs to.
|
|
145
|
-
*
|
|
146
|
-
* `markUsageLimitReached` and the auth-retry resolver switch a session to a
|
|
147
|
-
* sibling credential, so the provider state retained for that session can
|
|
148
|
-
* outlive the account that taught it. OAuth rows expose an account id / email
|
|
149
|
-
* that survives token refresh — fingerprinting the bearer instead would look
|
|
150
|
-
* like a rotation every time a token refreshes and discard the retained
|
|
151
|
-
* lessons for nothing. Key-based rows fall back to a hash of the key, never
|
|
152
|
-
* the key itself: this value is held for the lifetime of the entry.
|
|
153
|
-
*/
|
|
154
|
-
function resolveGatewayAccount(storage: AuthStorage, provider: string, sessionId: string, apiKey: string): string {
|
|
155
|
-
const identity = storage.getOAuthAccountIdentity(provider, sessionId);
|
|
156
|
-
if (identity) {
|
|
157
|
-
return `oauth:${JSON.stringify([
|
|
158
|
-
identity.accountId ?? "",
|
|
159
|
-
identity.email ?? "",
|
|
160
|
-
identity.projectId ?? "",
|
|
161
|
-
identity.orgId ?? "",
|
|
162
|
-
])}`;
|
|
163
|
-
}
|
|
164
|
-
return `key:${Bun.hash(apiKey).toString(36)}`;
|
|
165
|
-
}
|
|
166
|
-
|
|
167
135
|
function buildStreamOptions(parsed: ParsedFormatRequest, api: Api, signal: AbortSignal): SimpleStreamOptions {
|
|
168
136
|
const opts: SimpleStreamOptions = { signal, cursorExternalToolExecutor: true };
|
|
169
137
|
const { options } = parsed;
|
|
@@ -249,167 +217,26 @@ function buildStreamOptions(parsed: ParsedFormatRequest, api: Api, signal: Abort
|
|
|
249
217
|
return opts;
|
|
250
218
|
}
|
|
251
219
|
|
|
252
|
-
/**
|
|
253
|
-
* Hook fired by {@link streamSimple} when the upstream request fails in a
|
|
254
|
-
* way that's rotatable — today that's HTTP 401 (credential is bad) and
|
|
255
|
-
* usage-limit phrasing matched by {@link isUsageLimitError} (Codex's
|
|
256
|
-
* `usage_limit_reached`, Anthropic's `usage_limit_reached`, Google's
|
|
257
|
-
* `resource_exhausted`, …). The two cases need different storage actions:
|
|
258
|
-
*
|
|
259
|
-
* - **usage-limit** → {@link AuthStorage.markUsageLimitReached}. Marks just
|
|
260
|
-
* the current session's credential as temporarily blocked (honouring
|
|
261
|
-
* `retry-after` / `resets_at` hints when present) and returns `true` only
|
|
262
|
-
* when a sibling credential is still available. Burning the credential
|
|
263
|
-
* with `invalidateCredentialMatching` here would orphan accounts whose
|
|
264
|
-
* reset window is several hours away — exactly the bug this helper exists
|
|
265
|
-
* to avoid.
|
|
266
|
-
* - **auth-failure** → {@link AuthStorage.invalidateCredentialMatching}.
|
|
267
|
-
* Suspect/delete the row so it doesn't get re-picked next request.
|
|
268
|
-
*
|
|
269
|
-
* In both branches we return the next `getApiKey` result (sticky on the
|
|
270
|
-
* same `sessionId`) so streamSimple can transparently retry the pre-emit
|
|
271
|
-
* failure with a fresh credential. Returning `undefined` aborts the retry
|
|
272
|
-
* and surfaces the original error to the caller.
|
|
273
|
-
*/
|
|
274
|
-
async function refreshGatewayApiKeyAfterAuthError(
|
|
275
|
-
storage: AuthStorage,
|
|
276
|
-
model: Model<Api>,
|
|
277
|
-
sessionId: string,
|
|
278
|
-
provider: string,
|
|
279
|
-
oldKey: string,
|
|
280
|
-
error: unknown,
|
|
281
|
-
signal: AbortSignal,
|
|
282
|
-
format: string,
|
|
283
|
-
peer: string,
|
|
284
|
-
): Promise<string | undefined> {
|
|
285
|
-
const message = error instanceof Error ? error.message : String(error);
|
|
286
|
-
const status = extractHttpStatusFromError(error);
|
|
287
|
-
if (AIError.isUsageLimit(error) || isUsageLimitOutcome(status, message)) {
|
|
288
|
-
const retryAfterMs = extractProviderRetryHint(provider, message);
|
|
289
|
-
const { switched, retryAtMs } = await storage.markUsageLimitReached(provider, sessionId, {
|
|
290
|
-
retryAfterMs,
|
|
291
|
-
providerTimed: retryAfterMs !== undefined,
|
|
292
|
-
baseUrl: model.baseUrl,
|
|
293
|
-
modelId: model.id,
|
|
294
|
-
apiKey: oldKey,
|
|
295
|
-
signal,
|
|
296
|
-
});
|
|
297
|
-
logger.debug("auth-gateway retrying provider request after usage-limit block", {
|
|
298
|
-
format,
|
|
299
|
-
provider,
|
|
300
|
-
peer,
|
|
301
|
-
switched,
|
|
302
|
-
retryAfterMs,
|
|
303
|
-
retryAtMs,
|
|
304
|
-
error: message,
|
|
305
|
-
});
|
|
306
|
-
if (!switched) return undefined;
|
|
307
|
-
return storage.getApiKey(provider, sessionId, { modelId: model.id, signal });
|
|
308
|
-
}
|
|
309
|
-
await storage.invalidateCredentialMatching(provider, oldKey, { sessionId, signal });
|
|
310
|
-
logger.debug("auth-gateway retrying provider request after credential invalidation", {
|
|
311
|
-
format,
|
|
312
|
-
provider,
|
|
313
|
-
peer,
|
|
314
|
-
error: message,
|
|
315
|
-
});
|
|
316
|
-
return storage.getApiKey(provider, sessionId, { modelId: model.id, signal });
|
|
317
|
-
}
|
|
318
|
-
|
|
319
|
-
/**
|
|
320
|
-
* Build the {@link ApiKeyResolver} handed to `streamSimple` for a gateway
|
|
321
|
-
* request. Drives the central a/b/c auth-retry policy server-side:
|
|
322
|
-
*
|
|
323
|
-
* - initial resolve → the credential already resolved for this request.
|
|
324
|
-
* - step (b) `!lastChance` → force-refresh the SAME session-sticky credential
|
|
325
|
-
* (a peer/broker may have rotated its token out from under our cached copy).
|
|
326
|
-
* - step (c) `lastChance` → {@link refreshGatewayApiKeyAfterAuthError} switches
|
|
327
|
-
* to a sibling (usage-limit block vs credential invalidation by error class).
|
|
328
|
-
*
|
|
329
|
-
* `lastKey` tracks the most recent bearer so the switch step invalidates the
|
|
330
|
-
* credential that actually failed.
|
|
331
|
-
*/
|
|
332
|
-
function buildGatewayApiKeyResolver(
|
|
333
|
-
storage: AuthStorage,
|
|
334
|
-
model: Model<Api>,
|
|
335
|
-
sessionId: string,
|
|
336
|
-
initialKey: string,
|
|
337
|
-
requestSignal: AbortSignal,
|
|
338
|
-
format: string,
|
|
339
|
-
peer: string,
|
|
340
|
-
onResolvedKey: (apiKey: string) => void,
|
|
341
|
-
): ApiKeyResolver {
|
|
342
|
-
let lastKey = initialKey;
|
|
343
|
-
return async ({ lastChance, error, signal }) => {
|
|
344
|
-
const sig = signal ?? requestSignal;
|
|
345
|
-
if (error === undefined) {
|
|
346
|
-
lastKey = initialKey;
|
|
347
|
-
return initialKey;
|
|
348
|
-
}
|
|
349
|
-
if (!lastChance) {
|
|
350
|
-
const refreshed = await storage.getApiKey(model.provider, sessionId, {
|
|
351
|
-
modelId: model.id,
|
|
352
|
-
signal: sig,
|
|
353
|
-
forceRefresh: true,
|
|
354
|
-
});
|
|
355
|
-
lastKey = refreshed ?? lastKey;
|
|
356
|
-
if (refreshed) onResolvedKey(refreshed);
|
|
357
|
-
return refreshed;
|
|
358
|
-
}
|
|
359
|
-
const next = await refreshGatewayApiKeyAfterAuthError(
|
|
360
|
-
storage,
|
|
361
|
-
model,
|
|
362
|
-
sessionId,
|
|
363
|
-
model.provider,
|
|
364
|
-
lastKey,
|
|
365
|
-
error,
|
|
366
|
-
sig,
|
|
367
|
-
format,
|
|
368
|
-
peer,
|
|
369
|
-
);
|
|
370
|
-
lastKey = next ?? lastKey;
|
|
371
|
-
if (next) onResolvedKey(next);
|
|
372
|
-
return next;
|
|
373
|
-
};
|
|
374
|
-
}
|
|
375
|
-
|
|
376
220
|
function clientClosedResponse(route: { module: FormatModule }): Response {
|
|
377
221
|
return route.module.formatError(499, "request_aborted", "client closed request");
|
|
378
222
|
}
|
|
379
223
|
|
|
380
|
-
/**
|
|
381
|
-
|
|
382
|
-
|
|
383
|
-
|
|
384
|
-
|
|
385
|
-
|
|
386
|
-
|
|
387
|
-
|
|
388
|
-
|
|
389
|
-
|
|
390
|
-
client: ClientUsageIdentity,
|
|
391
|
-
message: AssistantMessage,
|
|
392
|
-
): void {
|
|
393
|
-
const usage = message.usage;
|
|
394
|
-
if (usage.input + usage.output + usage.cacheRead + usage.cacheWrite === 0) return;
|
|
395
|
-
storage.recordObservedUsage({
|
|
396
|
-
provider: model.provider,
|
|
397
|
-
model: model.id,
|
|
398
|
-
at: message.timestamp || Date.now(),
|
|
399
|
-
usage: { input: usage.input, output: usage.output, cacheRead: usage.cacheRead, cacheWrite: usage.cacheWrite },
|
|
400
|
-
costUsd: usage.cost.total,
|
|
401
|
-
client,
|
|
402
|
-
});
|
|
403
|
-
}
|
|
224
|
+
/** Route that serves each non-chat catalog kind the gateway advertises. */
|
|
225
|
+
const KIND_ROUTES: Partial<Record<ModelKind, string>> = {
|
|
226
|
+
judge: "POST /v1/systemone",
|
|
227
|
+
image: "POST /v1/images/generations",
|
|
228
|
+
tts: "POST /v1/audio/speech",
|
|
229
|
+
stt: "POST /v1/audio/transcriptions",
|
|
230
|
+
embedding: "POST /v1/embeddings",
|
|
231
|
+
rerank: "POST /v1/rerank",
|
|
232
|
+
video: "POST /v1/videos",
|
|
233
|
+
};
|
|
404
234
|
|
|
405
|
-
|
|
406
|
-
|
|
407
|
-
|
|
408
|
-
|
|
409
|
-
}
|
|
410
|
-
req.signal.addEventListener("abort", () => controller.abort(req.signal.reason), { once: true });
|
|
411
|
-
}
|
|
412
|
-
return controller;
|
|
235
|
+
/** Chat routes cannot drive a non-chat model; name the route that does, or `undefined` for chat models. */
|
|
236
|
+
function chatRouteRejection(model: Model<Api>): string | undefined {
|
|
237
|
+
const kind = modelKind(model);
|
|
238
|
+
const route = KIND_ROUTES[kind];
|
|
239
|
+
return route && `Model ${model.provider}/${model.id} is a ${kind} model; use ${route}`;
|
|
413
240
|
}
|
|
414
241
|
|
|
415
242
|
// (handlePassthrough removed — see note above.)
|
|
@@ -450,6 +277,8 @@ async function handleFormatEndpoint(
|
|
|
450
277
|
if (!model) {
|
|
451
278
|
return route.module.formatError(404, "invalid_request_error", `Unknown model: ${modelId}`);
|
|
452
279
|
}
|
|
280
|
+
const kindRejection = chatRouteRejection(model);
|
|
281
|
+
if (kindRejection) return route.module.formatError(400, "invalid_request_error", kindRejection);
|
|
453
282
|
const client = resolveClientIdentity(req.headers);
|
|
454
283
|
|
|
455
284
|
// Parse the wire-format request BEFORE resolving the credential so we
|
|
@@ -510,28 +339,12 @@ async function handleFormatEndpoint(
|
|
|
510
339
|
// expected to resolve the credential and pass it as `options.apiKey`.
|
|
511
340
|
// For OAuth providers this returns the access token (refreshed via the
|
|
512
341
|
// broker override on AuthStorage when needed).
|
|
513
|
-
|
|
514
|
-
try {
|
|
515
|
-
apiKey = await bootOpts.storage.getApiKey(model.provider, sessionId, {
|
|
516
|
-
modelId: model.id,
|
|
517
|
-
signal: controller.signal,
|
|
518
|
-
});
|
|
519
|
-
} catch (error) {
|
|
520
|
-
if (controller.signal.aborted) return clientClosedResponse(route);
|
|
521
|
-
const classified = classifyGatewayError(error);
|
|
522
|
-
logger.warn("auth-gateway getApiKey threw", { provider: model.provider, peer, error: classified.message });
|
|
523
|
-
return route.module.formatError(classified.status, classified.type, classified.message);
|
|
524
|
-
}
|
|
342
|
+
const apiKey = await resolveGatewayApiKey(bootOpts.storage, model, sessionId, controller.signal, peer);
|
|
525
343
|
if (controller.signal.aborted) return clientClosedResponse(route);
|
|
526
|
-
if (
|
|
527
|
-
return route.module.formatError(
|
|
528
|
-
401,
|
|
529
|
-
"authentication_error",
|
|
530
|
-
`No credential available for provider ${model.provider}`,
|
|
531
|
-
);
|
|
532
|
-
}
|
|
344
|
+
if (typeof apiKey !== "string") return route.module.formatError(apiKey.status, apiKey.type, apiKey.message);
|
|
533
345
|
|
|
534
346
|
const streamOpts = buildStreamOptions(parsed, model.api, controller.signal);
|
|
347
|
+
if (bootOpts.fetch) streamOpts.fetch = bootOpts.fetch;
|
|
535
348
|
// Per-session provider learning (sticky strict-tools / fast-mode / thinking
|
|
536
349
|
// fallbacks, Codex transport sessions). Owned by this gateway instance: the
|
|
537
350
|
// map is non-serializable, so no client can supply it and every turn would
|
|
@@ -571,7 +384,7 @@ async function handleFormatEndpoint(
|
|
|
571
384
|
try {
|
|
572
385
|
if (controller.signal.aborted) return clientClosedResponse(route);
|
|
573
386
|
const message = await completeSimple(model, parsed.context, streamOpts);
|
|
574
|
-
recordGatewayUsage(bootOpts.storage, model, client, message);
|
|
387
|
+
recordGatewayUsage(bootOpts.storage, model, client, message.usage, message.timestamp || undefined);
|
|
575
388
|
if (message.stopReason === "aborted" || message.stopReason === "error") {
|
|
576
389
|
const errorMessage =
|
|
577
390
|
message.errorMessage ??
|
|
@@ -591,7 +404,7 @@ async function handleFormatEndpoint(
|
|
|
591
404
|
return json(
|
|
592
405
|
200,
|
|
593
406
|
route.module.encodeResponse(message, parsed.modelId),
|
|
594
|
-
gatewayResponseHeaders(model, { requestId, message, startedAt }),
|
|
407
|
+
gatewayResponseHeaders(model, { requestId, costUsd: message.usage.cost.total, startedAt }),
|
|
595
408
|
);
|
|
596
409
|
} catch (error) {
|
|
597
410
|
if (controller.signal.aborted) return clientClosedResponse(route);
|
|
@@ -626,7 +439,9 @@ async function handleFormatEndpoint(
|
|
|
626
439
|
if (controller.signal.aborted) return clientClosedResponse(route);
|
|
627
440
|
void events
|
|
628
441
|
.result()
|
|
629
|
-
.then(message =>
|
|
442
|
+
.then(message =>
|
|
443
|
+
recordGatewayUsage(bootOpts.storage, model, client, message.usage, message.timestamp || undefined),
|
|
444
|
+
)
|
|
630
445
|
.catch(() => {})
|
|
631
446
|
.finally(() => lease.release());
|
|
632
447
|
streamOwnsLease = true;
|
|
@@ -705,6 +520,8 @@ async function handlePiNative(
|
|
|
705
520
|
if (!model) {
|
|
706
521
|
return piNative.formatError(404, "invalid_request_error", `Unknown model: ${parsed.modelId}`);
|
|
707
522
|
}
|
|
523
|
+
const kindRejection = chatRouteRejection(model);
|
|
524
|
+
if (kindRejection) return piNative.formatError(400, "invalid_request_error", kindRejection);
|
|
708
525
|
const client = resolveClientIdentity(req.headers);
|
|
709
526
|
// Pi-native already parsed `streamOpts.sessionId` (when set by the
|
|
710
527
|
// client); fall back to the derived key so credential-stickiness lines
|
|
@@ -715,26 +532,9 @@ async function handlePiNative(
|
|
|
715
532
|
const sessionId = clientKey ?? deriveSessionId(parsed.modelId, parsed.context);
|
|
716
533
|
parsed.options.sessionId = sessionId;
|
|
717
534
|
|
|
718
|
-
|
|
719
|
-
try {
|
|
720
|
-
apiKey = await bootOpts.storage.getApiKey(model.provider, sessionId, {
|
|
721
|
-
modelId: model.id,
|
|
722
|
-
signal: controller.signal,
|
|
723
|
-
});
|
|
724
|
-
} catch (error) {
|
|
725
|
-
if (controller.signal.aborted) return aborted();
|
|
726
|
-
const classified = classifyGatewayError(error);
|
|
727
|
-
logger.warn("auth-gateway getApiKey threw", { provider: model.provider, peer, error: classified.message });
|
|
728
|
-
return piNative.formatError(classified.status, classified.type, classified.message);
|
|
729
|
-
}
|
|
535
|
+
const apiKey = await resolveGatewayApiKey(bootOpts.storage, model, sessionId, controller.signal, peer);
|
|
730
536
|
if (controller.signal.aborted) return aborted();
|
|
731
|
-
if (
|
|
732
|
-
return piNative.formatError(
|
|
733
|
-
401,
|
|
734
|
-
"authentication_error",
|
|
735
|
-
`No credential available for provider ${model.provider}`,
|
|
736
|
-
);
|
|
737
|
-
}
|
|
537
|
+
if (typeof apiKey !== "string") return piNative.formatError(apiKey.status, apiKey.type, apiKey.message);
|
|
738
538
|
|
|
739
539
|
// Per-session provider learning, owned by this gateway instance. The map is
|
|
740
540
|
// non-serializable, so `parseRequest` cannot accept one from the wire and
|
|
@@ -758,6 +558,7 @@ async function handlePiNative(
|
|
|
758
558
|
cursorExternalToolExecutor: true,
|
|
759
559
|
providerSessionState: lease.states,
|
|
760
560
|
};
|
|
561
|
+
if (bootOpts.fetch) streamOpts.fetch = bootOpts.fetch;
|
|
761
562
|
streamOpts.apiKey = buildGatewayApiKeyResolver(
|
|
762
563
|
bootOpts.storage,
|
|
763
564
|
model,
|
|
@@ -799,7 +600,7 @@ async function handlePiNative(
|
|
|
799
600
|
try {
|
|
800
601
|
if (controller.signal.aborted) return aborted();
|
|
801
602
|
const message = await completeSimple(model, parsed.context, streamOpts);
|
|
802
|
-
recordGatewayUsage(bootOpts.storage, model, client, message);
|
|
603
|
+
recordGatewayUsage(bootOpts.storage, model, client, message.usage, message.timestamp || undefined);
|
|
803
604
|
if (message.stopReason === "aborted" || message.stopReason === "error") {
|
|
804
605
|
const errorMessage =
|
|
805
606
|
message.errorMessage ??
|
|
@@ -816,7 +617,11 @@ async function handlePiNative(
|
|
|
816
617
|
const classified = classifyGatewayError(message.errorClassificationMessage ?? errorMessage);
|
|
817
618
|
return piNative.formatError(classified.status, classified.type, errorMessage);
|
|
818
619
|
}
|
|
819
|
-
return json(
|
|
620
|
+
return json(
|
|
621
|
+
200,
|
|
622
|
+
{ message },
|
|
623
|
+
gatewayResponseHeaders(model, { requestId, costUsd: message.usage.cost.total, startedAt }),
|
|
624
|
+
);
|
|
820
625
|
} catch (error) {
|
|
821
626
|
if (controller.signal.aborted) return aborted();
|
|
822
627
|
const classified = classifyGatewayError(error);
|
|
@@ -846,7 +651,9 @@ async function handlePiNative(
|
|
|
846
651
|
if (controller.signal.aborted) return aborted();
|
|
847
652
|
void events
|
|
848
653
|
.result()
|
|
849
|
-
.then(message =>
|
|
654
|
+
.then(message =>
|
|
655
|
+
recordGatewayUsage(bootOpts.storage, model, client, message.usage, message.timestamp || undefined),
|
|
656
|
+
)
|
|
850
657
|
.catch(() => {})
|
|
851
658
|
.finally(() => lease.release());
|
|
852
659
|
streamOwnsLease = true;
|
|
@@ -911,13 +718,16 @@ async function handleCredentialsCheck(storage: AuthStorage, signal: AbortSignal)
|
|
|
911
718
|
* (omp's own proxy discovery, Zed's openai_compatible provider, ...) read to
|
|
912
719
|
* size and capability-gate discovered models: `context_length`,
|
|
913
720
|
* `max_output_tokens`, `input_modalities`, and `supports_tools` (only emitted
|
|
914
|
-
* when the catalog explicitly reports `false`; absent means usable).
|
|
721
|
+
* when the catalog explicitly reports `false`; absent means usable). `kind` is
|
|
722
|
+
* emitted for non-chat rows (`judge`, `image`, `tts`, `stt`, `embedding`,
|
|
723
|
+
* `rerank`, `video`) so clients can keep them off chat routes; absent means chat.
|
|
915
724
|
*/
|
|
916
725
|
interface ModelListRow {
|
|
917
726
|
id: string;
|
|
918
727
|
object: "model";
|
|
919
728
|
owned_by: string;
|
|
920
729
|
api: Api;
|
|
730
|
+
kind?: ModelKind;
|
|
921
731
|
display_name: string;
|
|
922
732
|
context_length?: number;
|
|
923
733
|
max_output_tokens?: number;
|
|
@@ -940,6 +750,7 @@ function handleModelsList(opts: AuthGatewayBootOptions): Response {
|
|
|
940
750
|
display_name: model.name,
|
|
941
751
|
input_modalities: model.input,
|
|
942
752
|
};
|
|
753
|
+
if (modelKind(model) !== "chat") row.kind = modelKind(model);
|
|
943
754
|
if (model.contextWindow != null) row.context_length = model.contextWindow;
|
|
944
755
|
if (model.maxTokens != null) row.max_output_tokens = model.maxTokens;
|
|
945
756
|
if (model.supportsTools === false) row.supports_tools = false;
|
|
@@ -948,6 +759,9 @@ function handleModelsList(opts: AuthGatewayBootOptions): Response {
|
|
|
948
759
|
return json(200, { object: "list", data });
|
|
949
760
|
}
|
|
950
761
|
|
|
762
|
+
/** `GET /v1/videos/:id` (poll) and `GET /v1/videos/:id/content` (download); group 1 = id, group 2 = `/content`. */
|
|
763
|
+
const VIDEO_JOB_PATH = /^\/v1\/videos\/([^/]+)(\/content)?$/;
|
|
764
|
+
|
|
951
765
|
export function startAuthGateway(opts: AuthGatewayBootOptions): AuthGatewayServerHandle {
|
|
952
766
|
const bind = parseBind(opts.bind ?? DEFAULT_AUTH_GATEWAY_BIND);
|
|
953
767
|
const tokens = new Set<string>(opts.bearerTokens);
|
|
@@ -1004,6 +818,55 @@ export function startAuthGateway(opts: AuthGatewayBootOptions): AuthGatewayServe
|
|
|
1004
818
|
return withCors(await handlePiNative(opts, req, peer, sessionStates), req);
|
|
1005
819
|
}
|
|
1006
820
|
|
|
821
|
+
// TypeSafe System One judgments (jev). TypeSafe SDKs and omp's own
|
|
822
|
+
// judge point `TYPESAFE_BASE_URL` at the gateway; OpenRouter SDKs
|
|
823
|
+
// reach the same handler through their Decisions path.
|
|
824
|
+
if (req.method === "POST" && (pathname === "/v1/systemone" || pathname === "/alpha/decisions")) {
|
|
825
|
+
return withCors(await handleSystemOne(opts, req, peer), req);
|
|
826
|
+
}
|
|
827
|
+
|
|
828
|
+
// Image generation: OpenAI `/v1/images/generations` + OpenRouter `/v1/images`
|
|
829
|
+
// (JSON), and OpenAI multipart / OpenRouter JSON edits.
|
|
830
|
+
if (req.method === "POST" && (pathname === "/v1/images/generations" || pathname === "/v1/images")) {
|
|
831
|
+
return withCors(await handleImageGenerations(opts, req, peer), req);
|
|
832
|
+
}
|
|
833
|
+
if (req.method === "POST" && pathname === "/v1/images/edits") {
|
|
834
|
+
return withCors(await handleImageEdits(opts, req, peer), req);
|
|
835
|
+
}
|
|
836
|
+
|
|
837
|
+
// Text-to-speech, OpenAI/OpenRouter wire; answers raw audio bytes.
|
|
838
|
+
if (req.method === "POST" && pathname === "/v1/audio/speech") {
|
|
839
|
+
return withCors(await handleSpeech(opts, req, peer), req);
|
|
840
|
+
}
|
|
841
|
+
|
|
842
|
+
// Speech-to-text, OpenAI multipart or OpenRouter JSON base64 wire.
|
|
843
|
+
if (req.method === "POST" && pathname === "/v1/audio/transcriptions") {
|
|
844
|
+
return withCors(await handleTranscriptions(opts, req, peer), req);
|
|
845
|
+
}
|
|
846
|
+
|
|
847
|
+
// Embeddings, OpenAI wire (OpenRouter is compatible).
|
|
848
|
+
if (req.method === "POST" && pathname === "/v1/embeddings") {
|
|
849
|
+
return withCors(await handleEmbeddings(opts, req, peer), req);
|
|
850
|
+
}
|
|
851
|
+
|
|
852
|
+
// Rerank, OpenRouter wire.
|
|
853
|
+
if (req.method === "POST" && pathname === "/v1/rerank") {
|
|
854
|
+
return withCors(await handleRerank(opts, req, peer), req);
|
|
855
|
+
}
|
|
856
|
+
|
|
857
|
+
// Video generation, OpenRouter's asynchronous wire: submit, then poll
|
|
858
|
+
// and download by the gateway-issued job id (stateless — the id
|
|
859
|
+
// encodes provider, model, and upstream job).
|
|
860
|
+
if (req.method === "POST" && pathname === "/v1/videos") {
|
|
861
|
+
return withCors(await handleVideoSubmit(opts, req, peer), req);
|
|
862
|
+
}
|
|
863
|
+
const videoJob = req.method === "GET" ? VIDEO_JOB_PATH.exec(pathname) : null;
|
|
864
|
+
if (videoJob) {
|
|
865
|
+
const gatewayId = decodeURIComponent(videoJob[1]);
|
|
866
|
+
const handler = videoJob[2] ? handleVideoContent : handleVideoPoll;
|
|
867
|
+
return withCors(await handler(opts, req, peer, gatewayId), req);
|
|
868
|
+
}
|
|
869
|
+
|
|
1007
870
|
// Model catalog.
|
|
1008
871
|
if (req.method === "GET" && pathname === "/v1/models") {
|
|
1009
872
|
return withCors(handleModelsList(opts), req);
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
import type { Api, Model } from "@oh-my-pi/pi-catalog/types";
|
|
2
|
+
import * as AIError from "../error";
|
|
3
|
+
import { embedOpenAI, type EmbeddingOptions } from "./openai-embeddings";
|
|
4
|
+
import type { EmbeddingRequest, EmbeddingResult } from "./types";
|
|
5
|
+
|
|
6
|
+
export * from "./openai-embeddings";
|
|
7
|
+
export * from "./types";
|
|
8
|
+
|
|
9
|
+
/** Dispatch an embedding request through the transport selected by the catalog model. */
|
|
10
|
+
export function embed(
|
|
11
|
+
model: Model<Api>,
|
|
12
|
+
request: EmbeddingRequest,
|
|
13
|
+
options: EmbeddingOptions,
|
|
14
|
+
): Promise<EmbeddingResult> {
|
|
15
|
+
if (model.api === "openai-embeddings") return embedOpenAI(model, request, options);
|
|
16
|
+
throw new AIError.ConfigurationError(`Unsupported embeddings API: ${model.api}`);
|
|
17
|
+
}
|