@oh-my-pi/pi-ai 18.2.7 → 18.2.9
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +49 -18
- package/dist/types/auth-gateway/dispatch.d.ts +80 -0
- package/dist/types/auth-gateway/http.d.ts +5 -6
- package/dist/types/auth-gateway/index.d.ts +1 -0
- package/dist/types/auth-gateway/routes/embeddings.d.ts +3 -0
- package/dist/types/auth-gateway/routes/images.d.ts +3 -0
- package/dist/types/auth-gateway/routes/rerank.d.ts +3 -0
- package/dist/types/auth-gateway/routes/speech.d.ts +2 -0
- package/dist/types/auth-gateway/routes/systemone.d.ts +2 -0
- package/dist/types/auth-gateway/routes/transcriptions.d.ts +3 -0
- package/dist/types/auth-gateway/routes/video.d.ts +7 -0
- package/dist/types/auth-gateway/server.d.ts +10 -16
- package/dist/types/auth-gateway/types.d.ts +5 -0
- package/dist/types/auth-storage.d.ts +35 -32
- package/dist/types/embeddings/index.d.ts +7 -0
- package/dist/types/embeddings/openai-embeddings.d.ts +15 -0
- package/dist/types/embeddings/types.d.ts +15 -0
- package/dist/types/images/google-antigravity.d.ts +9 -0
- package/dist/types/images/google-generative-ai.d.ts +3 -0
- package/dist/types/images/index.d.ts +14 -0
- package/dist/types/images/openai-hosted.d.ts +3 -0
- package/dist/types/images/openai-images.d.ts +5 -0
- package/dist/types/images/openrouter-images.d.ts +3 -0
- package/dist/types/images/shared.d.ts +34 -0
- package/dist/types/images/types.d.ts +31 -0
- package/dist/types/index.d.ts +8 -1
- package/dist/types/judgment/typesafe.d.ts +2 -0
- package/dist/types/providers/amazon-bedrock.d.ts +7 -0
- package/dist/types/providers/claude-code-fingerprint.d.ts +22 -3
- package/dist/types/providers/embeddings-server.d.ts +32 -0
- package/dist/types/providers/google-gemini-cli.d.ts +0 -2
- package/dist/types/providers/images-server.d.ts +22 -0
- package/dist/types/providers/openai-chat-server-schema.d.ts +2 -2
- package/dist/types/providers/rerank-server.d.ts +35 -0
- package/dist/types/providers/speech-server.d.ts +8 -0
- package/dist/types/providers/systemone-server.d.ts +26 -0
- package/dist/types/providers/transcriptions-server.d.ts +32 -0
- package/dist/types/providers/video-server.d.ts +39 -0
- package/dist/types/rerank/index.d.ts +7 -0
- package/dist/types/rerank/openrouter-rerank.d.ts +15 -0
- package/dist/types/rerank/types.d.ts +17 -0
- package/dist/types/speech/index.d.ts +13 -0
- package/dist/types/speech/openai-speech.d.ts +3 -0
- package/dist/types/speech/transport.d.ts +7 -0
- package/dist/types/speech/types.d.ts +24 -0
- package/dist/types/speech/xai-tts.d.ts +7 -0
- package/dist/types/transcription/index.d.ts +7 -0
- package/dist/types/transcription/openai-transcriptions.d.ts +15 -0
- package/dist/types/transcription/types.d.ts +41 -0
- package/dist/types/usage/claude-api.d.ts +22 -0
- package/dist/types/usage/claude-reset.d.ts +44 -0
- package/dist/types/usage.d.ts +111 -5
- package/dist/types/utils/schema/json-schema-validator.d.ts +5 -2
- package/dist/types/utils/tool-call-loop-guard.d.ts +1 -1
- package/dist/types/video/index.d.ts +11 -0
- package/dist/types/video/openrouter-video.d.ts +19 -0
- package/dist/types/video/types.d.ts +62 -0
- package/package.json +30 -6
- package/src/auth/sqlite-credential-store.ts +44 -1
- package/src/auth-broker/remote-store.ts +6 -6
- package/src/auth-broker/wire-schemas.ts +14 -0
- package/src/auth-gateway/dispatch.ts +273 -0
- package/src/auth-gateway/http.ts +6 -7
- package/src/auth-gateway/index.ts +1 -0
- package/src/auth-gateway/routes/embeddings.ts +98 -0
- package/src/auth-gateway/routes/images.ts +131 -0
- package/src/auth-gateway/routes/rerank.ts +87 -0
- package/src/auth-gateway/routes/speech.ts +101 -0
- package/src/auth-gateway/routes/systemone.ts +116 -0
- package/src/auth-gateway/routes/transcriptions.ts +98 -0
- package/src/auth-gateway/routes/video.ts +243 -0
- package/src/auth-gateway/server.ts +127 -260
- package/src/auth-gateway/types.ts +5 -0
- package/src/auth-storage.ts +263 -140
- package/src/embeddings/index.ts +17 -0
- package/src/embeddings/openai-embeddings.ts +141 -0
- package/src/embeddings/types.ts +14 -0
- package/src/error/flags.ts +10 -0
- package/src/error/rate-limit.ts +1 -1
- package/src/images/google-antigravity.ts +180 -0
- package/src/images/google-generative-ai.ts +92 -0
- package/src/images/index.ts +59 -0
- package/src/images/openai-hosted.ts +185 -0
- package/src/images/openai-images.ts +110 -0
- package/src/images/openrouter-images.ts +33 -0
- package/src/images/shared.ts +193 -0
- package/src/images/types.ts +36 -0
- package/src/index.ts +8 -1
- package/src/judgment/typesafe.ts +5 -0
- package/src/providers/amazon-bedrock.ts +55 -5
- package/src/providers/anthropic.ts +45 -11
- package/src/providers/aws-credentials.ts +124 -11
- package/src/providers/claude-code-fingerprint.ts +55 -3
- package/src/providers/embeddings-server.ts +151 -0
- package/src/providers/gitlab-duo.ts +20 -4
- package/src/providers/google-gemini-cli.ts +0 -8
- package/src/providers/google-shared.ts +1 -18
- package/src/providers/images-server.ts +159 -0
- package/src/providers/openai-chat-server-schema.ts +1 -1
- package/src/providers/openai-chat-server.ts +3 -1
- package/src/providers/openai-codex-responses.ts +27 -5
- package/src/providers/openai-completions.ts +122 -19
- package/src/providers/pi-native-server.ts +1 -0
- package/src/providers/rerank-server.ts +166 -0
- package/src/providers/speech-server.ts +53 -0
- package/src/providers/systemone-server.ts +73 -0
- package/src/providers/transcriptions-server.ts +243 -0
- package/src/providers/video-server.ts +286 -0
- package/src/registry/oauth/anthropic.ts +2 -3
- package/src/rerank/index.ts +13 -0
- package/src/rerank/openrouter-rerank.ts +136 -0
- package/src/rerank/types.ts +20 -0
- package/src/speech/index.ts +35 -0
- package/src/speech/openai-speech.ts +26 -0
- package/src/speech/transport.ts +66 -0
- package/src/speech/types.ts +37 -0
- package/src/speech/xai-tts.ts +41 -0
- package/src/stream.ts +13 -4
- package/src/transcription/index.ts +17 -0
- package/src/transcription/openai-transcriptions.ts +133 -0
- package/src/transcription/types.ts +46 -0
- package/src/usage/alibaba-token-plan.ts +7 -1
- package/src/usage/claude-api.ts +66 -0
- package/src/usage/claude-reset.ts +638 -0
- package/src/usage/claude.ts +37 -59
- package/src/usage/kimi.ts +32 -1
- package/src/usage.ts +52 -5
- package/src/utils/schema/json-schema-validator.ts +23 -10
- package/src/utils/tool-call-loop-guard.ts +2 -2
- package/src/utils/validation.ts +145 -50
- package/src/video/index.ts +34 -0
- package/src/video/openrouter-video.ts +210 -0
- package/src/video/types.ts +72 -0
|
@@ -16,25 +16,38 @@
|
|
|
16
16
|
* POST /v1/chat/completions → OpenAI chat-completions in/out
|
|
17
17
|
* POST /v1/messages → Anthropic messages in/out
|
|
18
18
|
* POST /v1/responses → OpenAI Responses in/out
|
|
19
|
+
* POST /v1/pi/stream → native pi-ai stream in/out
|
|
20
|
+
* POST /v1/systemone | /alpha/decisions → TypeSafe System One judgments (routes/systemone)
|
|
21
|
+
* POST /v1/images[/generations|/edits] → image generation, OpenAI/OpenRouter wire (routes/images)
|
|
22
|
+
* POST /v1/audio/speech → text-to-speech, raw audio out (routes/speech)
|
|
23
|
+
* POST /v1/audio/transcriptions → speech-to-text, multipart or JSON base64 in (routes/transcriptions)
|
|
24
|
+
*
|
|
25
|
+
* Chat routes live in this file; every other modality is a `routes/*` module
|
|
26
|
+
* built on the shared plumbing in `dispatch.ts`.
|
|
19
27
|
*/
|
|
20
28
|
|
|
21
29
|
import { Effort } from "@oh-my-pi/pi-catalog/effort";
|
|
22
|
-
import {
|
|
23
|
-
import
|
|
30
|
+
import { type ModelKind, modelKind } from "@oh-my-pi/pi-catalog/types";
|
|
31
|
+
import { logger } from "@oh-my-pi/pi-utils";
|
|
24
32
|
import type { AuthStorage } from "../auth-storage";
|
|
25
|
-
import * as AIError from "../error";
|
|
26
33
|
import { classifyGatewayError } from "../error/gateway";
|
|
27
|
-
import { isUsageLimitOutcome } from "../error/rate-limit";
|
|
28
34
|
import * as anthropicMessages from "../providers/anthropic-messages-server";
|
|
29
35
|
import * as openaiChat from "../providers/openai-chat-server";
|
|
30
36
|
import * as openaiResponses from "../providers/openai-responses-server";
|
|
31
37
|
import * as piNative from "../providers/pi-native-server";
|
|
32
38
|
import { completeSimple, streamSimple } from "../stream";
|
|
33
|
-
import type { Api,
|
|
34
|
-
import type { ClientUsageIdentity } from "../usage";
|
|
39
|
+
import type { Api, AssistantMessageEventStream, Context, Model, SimpleStreamOptions } from "../types";
|
|
35
40
|
import { deterministicUuid } from "../utils/deterministic-id";
|
|
36
41
|
import { parseBind } from "../utils/parse-bind";
|
|
37
|
-
import {
|
|
42
|
+
import {
|
|
43
|
+
type AuthGatewayBootOptions,
|
|
44
|
+
buildGatewayApiKeyResolver,
|
|
45
|
+
mirrorRequestAbort,
|
|
46
|
+
normalizeClientSessionKey,
|
|
47
|
+
recordGatewayUsage,
|
|
48
|
+
resolveGatewayAccount,
|
|
49
|
+
resolveGatewayApiKey,
|
|
50
|
+
} from "./dispatch";
|
|
38
51
|
import {
|
|
39
52
|
captureRequestHeaders,
|
|
40
53
|
corsHeaders,
|
|
@@ -45,32 +58,21 @@ import {
|
|
|
45
58
|
resolvePeer,
|
|
46
59
|
withCors,
|
|
47
60
|
} from "./http";
|
|
61
|
+
import { handleEmbeddings } from "./routes/embeddings";
|
|
62
|
+
import { handleImageEdits, handleImageGenerations } from "./routes/images";
|
|
63
|
+
import { handleRerank } from "./routes/rerank";
|
|
64
|
+
import { handleSpeech } from "./routes/speech";
|
|
65
|
+
import { handleSystemOne } from "./routes/systemone";
|
|
66
|
+
import { handleTranscriptions } from "./routes/transcriptions";
|
|
67
|
+
import { handleVideoContent, handleVideoPoll, handleVideoSubmit } from "./routes/video";
|
|
48
68
|
import { AuthGatewaySessionStateStore } from "./session-state";
|
|
49
69
|
import type {
|
|
50
70
|
AuthGatewayServerHandle,
|
|
51
|
-
AuthGatewayServerOptions,
|
|
52
71
|
AuthGatewayFormatModule as FormatModule,
|
|
53
72
|
AuthGatewayParsedRequest as ParsedFormatRequest,
|
|
54
73
|
} from "./types";
|
|
55
74
|
import { DEFAULT_AUTH_GATEWAY_BIND } from "./types";
|
|
56
75
|
|
|
57
|
-
// ParsedFormatRequest / ParsedFormatOptions / FormatModule come from ./types.
|
|
58
|
-
|
|
59
|
-
export type ModelResolver = (modelId: string) => Model<Api> | undefined;
|
|
60
|
-
|
|
61
|
-
export interface AuthGatewayBootOptions extends AuthGatewayServerOptions {
|
|
62
|
-
/** Source of credentials. Caller wires this to a broker-backed AuthStorage. */
|
|
63
|
-
storage: AuthStorage;
|
|
64
|
-
/**
|
|
65
|
-
* Resolve a client-requested model id to a pi-ai Model. Caller supplies
|
|
66
|
-
* this from a ModelRegistry (lives in `coding-agent` to avoid an inverse
|
|
67
|
-
* dependency in `pi-ai`).
|
|
68
|
-
*/
|
|
69
|
-
resolveModel: ModelResolver;
|
|
70
|
-
/** Optional supplier for `/v1/models` listing. Returns the full model array. */
|
|
71
|
-
listModels?: () => Iterable<Model<Api>>;
|
|
72
|
-
}
|
|
73
|
-
|
|
74
76
|
// `parseBind` lives in ../utils/parse-bind so the gateway and broker can't
|
|
75
77
|
// drift on accepted inputs (e.g. empty hostname, IPv6 brackets).
|
|
76
78
|
|
|
@@ -130,40 +132,6 @@ function deriveSessionId(modelId: string, context: Context): string {
|
|
|
130
132
|
return deterministicUuid(seed);
|
|
131
133
|
}
|
|
132
134
|
|
|
133
|
-
/**
|
|
134
|
-
* The client's own session key, or `undefined` when it sent none. A blank key
|
|
135
|
-
* counts as none: honouring it would collapse every caller that sends an empty
|
|
136
|
-
* key into one shared credential-sticky, prefix-cache and provider-session
|
|
137
|
-
* bucket.
|
|
138
|
-
*/
|
|
139
|
-
function normalizeClientSessionKey(clientKey: string | undefined): string | undefined {
|
|
140
|
-
return clientKey !== undefined && clientKey.trim().length > 0 ? clientKey : undefined;
|
|
141
|
-
}
|
|
142
|
-
|
|
143
|
-
/**
|
|
144
|
-
* Stable identity of the account a request's credential belongs to.
|
|
145
|
-
*
|
|
146
|
-
* `markUsageLimitReached` and the auth-retry resolver switch a session to a
|
|
147
|
-
* sibling credential, so the provider state retained for that session can
|
|
148
|
-
* outlive the account that taught it. OAuth rows expose an account id / email
|
|
149
|
-
* that survives token refresh — fingerprinting the bearer instead would look
|
|
150
|
-
* like a rotation every time a token refreshes and discard the retained
|
|
151
|
-
* lessons for nothing. Key-based rows fall back to a hash of the key, never
|
|
152
|
-
* the key itself: this value is held for the lifetime of the entry.
|
|
153
|
-
*/
|
|
154
|
-
function resolveGatewayAccount(storage: AuthStorage, provider: string, sessionId: string, apiKey: string): string {
|
|
155
|
-
const identity = storage.getOAuthAccountIdentity(provider, sessionId);
|
|
156
|
-
if (identity) {
|
|
157
|
-
return `oauth:${JSON.stringify([
|
|
158
|
-
identity.accountId ?? "",
|
|
159
|
-
identity.email ?? "",
|
|
160
|
-
identity.projectId ?? "",
|
|
161
|
-
identity.orgId ?? "",
|
|
162
|
-
])}`;
|
|
163
|
-
}
|
|
164
|
-
return `key:${Bun.hash(apiKey).toString(36)}`;
|
|
165
|
-
}
|
|
166
|
-
|
|
167
135
|
function buildStreamOptions(parsed: ParsedFormatRequest, api: Api, signal: AbortSignal): SimpleStreamOptions {
|
|
168
136
|
const opts: SimpleStreamOptions = { signal, cursorExternalToolExecutor: true };
|
|
169
137
|
const { options } = parsed;
|
|
@@ -193,6 +161,10 @@ function buildStreamOptions(parsed: ParsedFormatRequest, api: Api, signal: Abort
|
|
|
193
161
|
}
|
|
194
162
|
if (options.reasoning !== undefined) opts.reasoning = options.reasoning;
|
|
195
163
|
if (options.disableReasoning !== undefined) opts.disableReasoning = options.disableReasoning;
|
|
164
|
+
if (options.forceReasoningOff !== undefined) {
|
|
165
|
+
opts.disableReasoning = options.forceReasoningOff;
|
|
166
|
+
opts.forceReasoningOff = options.forceReasoningOff;
|
|
167
|
+
}
|
|
196
168
|
if (options.hideThinkingSummary !== undefined) opts.hideThinkingSummary = options.hideThinkingSummary;
|
|
197
169
|
if (options.taskBudget !== undefined) opts.taskBudget = options.taskBudget;
|
|
198
170
|
if (options.anthropicPrefixMismatchBehavior !== undefined) {
|
|
@@ -249,167 +221,26 @@ function buildStreamOptions(parsed: ParsedFormatRequest, api: Api, signal: Abort
|
|
|
249
221
|
return opts;
|
|
250
222
|
}
|
|
251
223
|
|
|
252
|
-
/**
|
|
253
|
-
* Hook fired by {@link streamSimple} when the upstream request fails in a
|
|
254
|
-
* way that's rotatable — today that's HTTP 401 (credential is bad) and
|
|
255
|
-
* usage-limit phrasing matched by {@link isUsageLimitError} (Codex's
|
|
256
|
-
* `usage_limit_reached`, Anthropic's `usage_limit_reached`, Google's
|
|
257
|
-
* `resource_exhausted`, …). The two cases need different storage actions:
|
|
258
|
-
*
|
|
259
|
-
* - **usage-limit** → {@link AuthStorage.markUsageLimitReached}. Marks just
|
|
260
|
-
* the current session's credential as temporarily blocked (honouring
|
|
261
|
-
* `retry-after` / `resets_at` hints when present) and returns `true` only
|
|
262
|
-
* when a sibling credential is still available. Burning the credential
|
|
263
|
-
* with `invalidateCredentialMatching` here would orphan accounts whose
|
|
264
|
-
* reset window is several hours away — exactly the bug this helper exists
|
|
265
|
-
* to avoid.
|
|
266
|
-
* - **auth-failure** → {@link AuthStorage.invalidateCredentialMatching}.
|
|
267
|
-
* Suspect/delete the row so it doesn't get re-picked next request.
|
|
268
|
-
*
|
|
269
|
-
* In both branches we return the next `getApiKey` result (sticky on the
|
|
270
|
-
* same `sessionId`) so streamSimple can transparently retry the pre-emit
|
|
271
|
-
* failure with a fresh credential. Returning `undefined` aborts the retry
|
|
272
|
-
* and surfaces the original error to the caller.
|
|
273
|
-
*/
|
|
274
|
-
async function refreshGatewayApiKeyAfterAuthError(
|
|
275
|
-
storage: AuthStorage,
|
|
276
|
-
model: Model<Api>,
|
|
277
|
-
sessionId: string,
|
|
278
|
-
provider: string,
|
|
279
|
-
oldKey: string,
|
|
280
|
-
error: unknown,
|
|
281
|
-
signal: AbortSignal,
|
|
282
|
-
format: string,
|
|
283
|
-
peer: string,
|
|
284
|
-
): Promise<string | undefined> {
|
|
285
|
-
const message = error instanceof Error ? error.message : String(error);
|
|
286
|
-
const status = extractHttpStatusFromError(error);
|
|
287
|
-
if (AIError.isUsageLimit(error) || isUsageLimitOutcome(status, message)) {
|
|
288
|
-
const retryAfterMs = extractProviderRetryHint(provider, message);
|
|
289
|
-
const { switched, retryAtMs } = await storage.markUsageLimitReached(provider, sessionId, {
|
|
290
|
-
retryAfterMs,
|
|
291
|
-
providerTimed: retryAfterMs !== undefined,
|
|
292
|
-
baseUrl: model.baseUrl,
|
|
293
|
-
modelId: model.id,
|
|
294
|
-
apiKey: oldKey,
|
|
295
|
-
signal,
|
|
296
|
-
});
|
|
297
|
-
logger.debug("auth-gateway retrying provider request after usage-limit block", {
|
|
298
|
-
format,
|
|
299
|
-
provider,
|
|
300
|
-
peer,
|
|
301
|
-
switched,
|
|
302
|
-
retryAfterMs,
|
|
303
|
-
retryAtMs,
|
|
304
|
-
error: message,
|
|
305
|
-
});
|
|
306
|
-
if (!switched) return undefined;
|
|
307
|
-
return storage.getApiKey(provider, sessionId, { modelId: model.id, signal });
|
|
308
|
-
}
|
|
309
|
-
await storage.invalidateCredentialMatching(provider, oldKey, { sessionId, signal });
|
|
310
|
-
logger.debug("auth-gateway retrying provider request after credential invalidation", {
|
|
311
|
-
format,
|
|
312
|
-
provider,
|
|
313
|
-
peer,
|
|
314
|
-
error: message,
|
|
315
|
-
});
|
|
316
|
-
return storage.getApiKey(provider, sessionId, { modelId: model.id, signal });
|
|
317
|
-
}
|
|
318
|
-
|
|
319
|
-
/**
|
|
320
|
-
* Build the {@link ApiKeyResolver} handed to `streamSimple` for a gateway
|
|
321
|
-
* request. Drives the central a/b/c auth-retry policy server-side:
|
|
322
|
-
*
|
|
323
|
-
* - initial resolve → the credential already resolved for this request.
|
|
324
|
-
* - step (b) `!lastChance` → force-refresh the SAME session-sticky credential
|
|
325
|
-
* (a peer/broker may have rotated its token out from under our cached copy).
|
|
326
|
-
* - step (c) `lastChance` → {@link refreshGatewayApiKeyAfterAuthError} switches
|
|
327
|
-
* to a sibling (usage-limit block vs credential invalidation by error class).
|
|
328
|
-
*
|
|
329
|
-
* `lastKey` tracks the most recent bearer so the switch step invalidates the
|
|
330
|
-
* credential that actually failed.
|
|
331
|
-
*/
|
|
332
|
-
function buildGatewayApiKeyResolver(
|
|
333
|
-
storage: AuthStorage,
|
|
334
|
-
model: Model<Api>,
|
|
335
|
-
sessionId: string,
|
|
336
|
-
initialKey: string,
|
|
337
|
-
requestSignal: AbortSignal,
|
|
338
|
-
format: string,
|
|
339
|
-
peer: string,
|
|
340
|
-
onResolvedKey: (apiKey: string) => void,
|
|
341
|
-
): ApiKeyResolver {
|
|
342
|
-
let lastKey = initialKey;
|
|
343
|
-
return async ({ lastChance, error, signal }) => {
|
|
344
|
-
const sig = signal ?? requestSignal;
|
|
345
|
-
if (error === undefined) {
|
|
346
|
-
lastKey = initialKey;
|
|
347
|
-
return initialKey;
|
|
348
|
-
}
|
|
349
|
-
if (!lastChance) {
|
|
350
|
-
const refreshed = await storage.getApiKey(model.provider, sessionId, {
|
|
351
|
-
modelId: model.id,
|
|
352
|
-
signal: sig,
|
|
353
|
-
forceRefresh: true,
|
|
354
|
-
});
|
|
355
|
-
lastKey = refreshed ?? lastKey;
|
|
356
|
-
if (refreshed) onResolvedKey(refreshed);
|
|
357
|
-
return refreshed;
|
|
358
|
-
}
|
|
359
|
-
const next = await refreshGatewayApiKeyAfterAuthError(
|
|
360
|
-
storage,
|
|
361
|
-
model,
|
|
362
|
-
sessionId,
|
|
363
|
-
model.provider,
|
|
364
|
-
lastKey,
|
|
365
|
-
error,
|
|
366
|
-
sig,
|
|
367
|
-
format,
|
|
368
|
-
peer,
|
|
369
|
-
);
|
|
370
|
-
lastKey = next ?? lastKey;
|
|
371
|
-
if (next) onResolvedKey(next);
|
|
372
|
-
return next;
|
|
373
|
-
};
|
|
374
|
-
}
|
|
375
|
-
|
|
376
224
|
function clientClosedResponse(route: { module: FormatModule }): Response {
|
|
377
225
|
return route.module.formatError(499, "request_aborted", "client closed request");
|
|
378
226
|
}
|
|
379
227
|
|
|
380
|
-
/**
|
|
381
|
-
|
|
382
|
-
|
|
383
|
-
|
|
384
|
-
|
|
385
|
-
|
|
386
|
-
|
|
387
|
-
|
|
388
|
-
|
|
389
|
-
|
|
390
|
-
client: ClientUsageIdentity,
|
|
391
|
-
message: AssistantMessage,
|
|
392
|
-
): void {
|
|
393
|
-
const usage = message.usage;
|
|
394
|
-
if (usage.input + usage.output + usage.cacheRead + usage.cacheWrite === 0) return;
|
|
395
|
-
storage.recordObservedUsage({
|
|
396
|
-
provider: model.provider,
|
|
397
|
-
model: model.id,
|
|
398
|
-
at: message.timestamp || Date.now(),
|
|
399
|
-
usage: { input: usage.input, output: usage.output, cacheRead: usage.cacheRead, cacheWrite: usage.cacheWrite },
|
|
400
|
-
costUsd: usage.cost.total,
|
|
401
|
-
client,
|
|
402
|
-
});
|
|
403
|
-
}
|
|
228
|
+
/** Route that serves each non-chat catalog kind the gateway advertises. */
|
|
229
|
+
const KIND_ROUTES: Partial<Record<ModelKind, string>> = {
|
|
230
|
+
judge: "POST /v1/systemone",
|
|
231
|
+
image: "POST /v1/images/generations",
|
|
232
|
+
tts: "POST /v1/audio/speech",
|
|
233
|
+
stt: "POST /v1/audio/transcriptions",
|
|
234
|
+
embedding: "POST /v1/embeddings",
|
|
235
|
+
rerank: "POST /v1/rerank",
|
|
236
|
+
video: "POST /v1/videos",
|
|
237
|
+
};
|
|
404
238
|
|
|
405
|
-
|
|
406
|
-
|
|
407
|
-
|
|
408
|
-
|
|
409
|
-
}
|
|
410
|
-
req.signal.addEventListener("abort", () => controller.abort(req.signal.reason), { once: true });
|
|
411
|
-
}
|
|
412
|
-
return controller;
|
|
239
|
+
/** Chat routes cannot drive a non-chat model; name the route that does, or `undefined` for chat models. */
|
|
240
|
+
function chatRouteRejection(model: Model<Api>): string | undefined {
|
|
241
|
+
const kind = modelKind(model);
|
|
242
|
+
const route = KIND_ROUTES[kind];
|
|
243
|
+
return route && `Model ${model.provider}/${model.id} is a ${kind} model; use ${route}`;
|
|
413
244
|
}
|
|
414
245
|
|
|
415
246
|
// (handlePassthrough removed — see note above.)
|
|
@@ -450,6 +281,8 @@ async function handleFormatEndpoint(
|
|
|
450
281
|
if (!model) {
|
|
451
282
|
return route.module.formatError(404, "invalid_request_error", `Unknown model: ${modelId}`);
|
|
452
283
|
}
|
|
284
|
+
const kindRejection = chatRouteRejection(model);
|
|
285
|
+
if (kindRejection) return route.module.formatError(400, "invalid_request_error", kindRejection);
|
|
453
286
|
const client = resolveClientIdentity(req.headers);
|
|
454
287
|
|
|
455
288
|
// Parse the wire-format request BEFORE resolving the credential so we
|
|
@@ -510,28 +343,12 @@ async function handleFormatEndpoint(
|
|
|
510
343
|
// expected to resolve the credential and pass it as `options.apiKey`.
|
|
511
344
|
// For OAuth providers this returns the access token (refreshed via the
|
|
512
345
|
// broker override on AuthStorage when needed).
|
|
513
|
-
|
|
514
|
-
try {
|
|
515
|
-
apiKey = await bootOpts.storage.getApiKey(model.provider, sessionId, {
|
|
516
|
-
modelId: model.id,
|
|
517
|
-
signal: controller.signal,
|
|
518
|
-
});
|
|
519
|
-
} catch (error) {
|
|
520
|
-
if (controller.signal.aborted) return clientClosedResponse(route);
|
|
521
|
-
const classified = classifyGatewayError(error);
|
|
522
|
-
logger.warn("auth-gateway getApiKey threw", { provider: model.provider, peer, error: classified.message });
|
|
523
|
-
return route.module.formatError(classified.status, classified.type, classified.message);
|
|
524
|
-
}
|
|
346
|
+
const apiKey = await resolveGatewayApiKey(bootOpts.storage, model, sessionId, controller.signal, peer);
|
|
525
347
|
if (controller.signal.aborted) return clientClosedResponse(route);
|
|
526
|
-
if (
|
|
527
|
-
return route.module.formatError(
|
|
528
|
-
401,
|
|
529
|
-
"authentication_error",
|
|
530
|
-
`No credential available for provider ${model.provider}`,
|
|
531
|
-
);
|
|
532
|
-
}
|
|
348
|
+
if (typeof apiKey !== "string") return route.module.formatError(apiKey.status, apiKey.type, apiKey.message);
|
|
533
349
|
|
|
534
350
|
const streamOpts = buildStreamOptions(parsed, model.api, controller.signal);
|
|
351
|
+
if (bootOpts.fetch) streamOpts.fetch = bootOpts.fetch;
|
|
535
352
|
// Per-session provider learning (sticky strict-tools / fast-mode / thinking
|
|
536
353
|
// fallbacks, Codex transport sessions). Owned by this gateway instance: the
|
|
537
354
|
// map is non-serializable, so no client can supply it and every turn would
|
|
@@ -571,7 +388,7 @@ async function handleFormatEndpoint(
|
|
|
571
388
|
try {
|
|
572
389
|
if (controller.signal.aborted) return clientClosedResponse(route);
|
|
573
390
|
const message = await completeSimple(model, parsed.context, streamOpts);
|
|
574
|
-
recordGatewayUsage(bootOpts.storage, model, client, message);
|
|
391
|
+
recordGatewayUsage(bootOpts.storage, model, client, message.usage, message.timestamp || undefined);
|
|
575
392
|
if (message.stopReason === "aborted" || message.stopReason === "error") {
|
|
576
393
|
const errorMessage =
|
|
577
394
|
message.errorMessage ??
|
|
@@ -591,7 +408,7 @@ async function handleFormatEndpoint(
|
|
|
591
408
|
return json(
|
|
592
409
|
200,
|
|
593
410
|
route.module.encodeResponse(message, parsed.modelId),
|
|
594
|
-
gatewayResponseHeaders(model, { requestId, message, startedAt }),
|
|
411
|
+
gatewayResponseHeaders(model, { requestId, costUsd: message.usage.cost.total, startedAt }),
|
|
595
412
|
);
|
|
596
413
|
} catch (error) {
|
|
597
414
|
if (controller.signal.aborted) return clientClosedResponse(route);
|
|
@@ -626,7 +443,9 @@ async function handleFormatEndpoint(
|
|
|
626
443
|
if (controller.signal.aborted) return clientClosedResponse(route);
|
|
627
444
|
void events
|
|
628
445
|
.result()
|
|
629
|
-
.then(message =>
|
|
446
|
+
.then(message =>
|
|
447
|
+
recordGatewayUsage(bootOpts.storage, model, client, message.usage, message.timestamp || undefined),
|
|
448
|
+
)
|
|
630
449
|
.catch(() => {})
|
|
631
450
|
.finally(() => lease.release());
|
|
632
451
|
streamOwnsLease = true;
|
|
@@ -705,6 +524,8 @@ async function handlePiNative(
|
|
|
705
524
|
if (!model) {
|
|
706
525
|
return piNative.formatError(404, "invalid_request_error", `Unknown model: ${parsed.modelId}`);
|
|
707
526
|
}
|
|
527
|
+
const kindRejection = chatRouteRejection(model);
|
|
528
|
+
if (kindRejection) return piNative.formatError(400, "invalid_request_error", kindRejection);
|
|
708
529
|
const client = resolveClientIdentity(req.headers);
|
|
709
530
|
// Pi-native already parsed `streamOpts.sessionId` (when set by the
|
|
710
531
|
// client); fall back to the derived key so credential-stickiness lines
|
|
@@ -715,26 +536,9 @@ async function handlePiNative(
|
|
|
715
536
|
const sessionId = clientKey ?? deriveSessionId(parsed.modelId, parsed.context);
|
|
716
537
|
parsed.options.sessionId = sessionId;
|
|
717
538
|
|
|
718
|
-
|
|
719
|
-
try {
|
|
720
|
-
apiKey = await bootOpts.storage.getApiKey(model.provider, sessionId, {
|
|
721
|
-
modelId: model.id,
|
|
722
|
-
signal: controller.signal,
|
|
723
|
-
});
|
|
724
|
-
} catch (error) {
|
|
725
|
-
if (controller.signal.aborted) return aborted();
|
|
726
|
-
const classified = classifyGatewayError(error);
|
|
727
|
-
logger.warn("auth-gateway getApiKey threw", { provider: model.provider, peer, error: classified.message });
|
|
728
|
-
return piNative.formatError(classified.status, classified.type, classified.message);
|
|
729
|
-
}
|
|
539
|
+
const apiKey = await resolveGatewayApiKey(bootOpts.storage, model, sessionId, controller.signal, peer);
|
|
730
540
|
if (controller.signal.aborted) return aborted();
|
|
731
|
-
if (
|
|
732
|
-
return piNative.formatError(
|
|
733
|
-
401,
|
|
734
|
-
"authentication_error",
|
|
735
|
-
`No credential available for provider ${model.provider}`,
|
|
736
|
-
);
|
|
737
|
-
}
|
|
541
|
+
if (typeof apiKey !== "string") return piNative.formatError(apiKey.status, apiKey.type, apiKey.message);
|
|
738
542
|
|
|
739
543
|
// Per-session provider learning, owned by this gateway instance. The map is
|
|
740
544
|
// non-serializable, so `parseRequest` cannot accept one from the wire and
|
|
@@ -758,6 +562,7 @@ async function handlePiNative(
|
|
|
758
562
|
cursorExternalToolExecutor: true,
|
|
759
563
|
providerSessionState: lease.states,
|
|
760
564
|
};
|
|
565
|
+
if (bootOpts.fetch) streamOpts.fetch = bootOpts.fetch;
|
|
761
566
|
streamOpts.apiKey = buildGatewayApiKeyResolver(
|
|
762
567
|
bootOpts.storage,
|
|
763
568
|
model,
|
|
@@ -799,7 +604,7 @@ async function handlePiNative(
|
|
|
799
604
|
try {
|
|
800
605
|
if (controller.signal.aborted) return aborted();
|
|
801
606
|
const message = await completeSimple(model, parsed.context, streamOpts);
|
|
802
|
-
recordGatewayUsage(bootOpts.storage, model, client, message);
|
|
607
|
+
recordGatewayUsage(bootOpts.storage, model, client, message.usage, message.timestamp || undefined);
|
|
803
608
|
if (message.stopReason === "aborted" || message.stopReason === "error") {
|
|
804
609
|
const errorMessage =
|
|
805
610
|
message.errorMessage ??
|
|
@@ -816,7 +621,11 @@ async function handlePiNative(
|
|
|
816
621
|
const classified = classifyGatewayError(message.errorClassificationMessage ?? errorMessage);
|
|
817
622
|
return piNative.formatError(classified.status, classified.type, errorMessage);
|
|
818
623
|
}
|
|
819
|
-
return json(
|
|
624
|
+
return json(
|
|
625
|
+
200,
|
|
626
|
+
{ message },
|
|
627
|
+
gatewayResponseHeaders(model, { requestId, costUsd: message.usage.cost.total, startedAt }),
|
|
628
|
+
);
|
|
820
629
|
} catch (error) {
|
|
821
630
|
if (controller.signal.aborted) return aborted();
|
|
822
631
|
const classified = classifyGatewayError(error);
|
|
@@ -846,7 +655,9 @@ async function handlePiNative(
|
|
|
846
655
|
if (controller.signal.aborted) return aborted();
|
|
847
656
|
void events
|
|
848
657
|
.result()
|
|
849
|
-
.then(message =>
|
|
658
|
+
.then(message =>
|
|
659
|
+
recordGatewayUsage(bootOpts.storage, model, client, message.usage, message.timestamp || undefined),
|
|
660
|
+
)
|
|
850
661
|
.catch(() => {})
|
|
851
662
|
.finally(() => lease.release());
|
|
852
663
|
streamOwnsLease = true;
|
|
@@ -911,13 +722,16 @@ async function handleCredentialsCheck(storage: AuthStorage, signal: AbortSignal)
|
|
|
911
722
|
* (omp's own proxy discovery, Zed's openai_compatible provider, ...) read to
|
|
912
723
|
* size and capability-gate discovered models: `context_length`,
|
|
913
724
|
* `max_output_tokens`, `input_modalities`, and `supports_tools` (only emitted
|
|
914
|
-
* when the catalog explicitly reports `false`; absent means usable).
|
|
725
|
+
* when the catalog explicitly reports `false`; absent means usable). `kind` is
|
|
726
|
+
* emitted for non-chat rows (`judge`, `image`, `tts`, `stt`, `embedding`,
|
|
727
|
+
* `rerank`, `video`) so clients can keep them off chat routes; absent means chat.
|
|
915
728
|
*/
|
|
916
729
|
interface ModelListRow {
|
|
917
730
|
id: string;
|
|
918
731
|
object: "model";
|
|
919
732
|
owned_by: string;
|
|
920
733
|
api: Api;
|
|
734
|
+
kind?: ModelKind;
|
|
921
735
|
display_name: string;
|
|
922
736
|
context_length?: number;
|
|
923
737
|
max_output_tokens?: number;
|
|
@@ -940,6 +754,7 @@ function handleModelsList(opts: AuthGatewayBootOptions): Response {
|
|
|
940
754
|
display_name: model.name,
|
|
941
755
|
input_modalities: model.input,
|
|
942
756
|
};
|
|
757
|
+
if (modelKind(model) !== "chat") row.kind = modelKind(model);
|
|
943
758
|
if (model.contextWindow != null) row.context_length = model.contextWindow;
|
|
944
759
|
if (model.maxTokens != null) row.max_output_tokens = model.maxTokens;
|
|
945
760
|
if (model.supportsTools === false) row.supports_tools = false;
|
|
@@ -948,6 +763,9 @@ function handleModelsList(opts: AuthGatewayBootOptions): Response {
|
|
|
948
763
|
return json(200, { object: "list", data });
|
|
949
764
|
}
|
|
950
765
|
|
|
766
|
+
/** `GET /v1/videos/:id` (poll) and `GET /v1/videos/:id/content` (download); group 1 = id, group 2 = `/content`. */
|
|
767
|
+
const VIDEO_JOB_PATH = /^\/v1\/videos\/([^/]+)(\/content)?$/;
|
|
768
|
+
|
|
951
769
|
export function startAuthGateway(opts: AuthGatewayBootOptions): AuthGatewayServerHandle {
|
|
952
770
|
const bind = parseBind(opts.bind ?? DEFAULT_AUTH_GATEWAY_BIND);
|
|
953
771
|
const tokens = new Set<string>(opts.bearerTokens);
|
|
@@ -1004,6 +822,55 @@ export function startAuthGateway(opts: AuthGatewayBootOptions): AuthGatewayServe
|
|
|
1004
822
|
return withCors(await handlePiNative(opts, req, peer, sessionStates), req);
|
|
1005
823
|
}
|
|
1006
824
|
|
|
825
|
+
// TypeSafe System One judgments (jev). TypeSafe SDKs and omp's own
|
|
826
|
+
// judge point `TYPESAFE_BASE_URL` at the gateway; OpenRouter SDKs
|
|
827
|
+
// reach the same handler through their Decisions path.
|
|
828
|
+
if (req.method === "POST" && (pathname === "/v1/systemone" || pathname === "/alpha/decisions")) {
|
|
829
|
+
return withCors(await handleSystemOne(opts, req, peer), req);
|
|
830
|
+
}
|
|
831
|
+
|
|
832
|
+
// Image generation: OpenAI `/v1/images/generations` + OpenRouter `/v1/images`
|
|
833
|
+
// (JSON), and OpenAI multipart / OpenRouter JSON edits.
|
|
834
|
+
if (req.method === "POST" && (pathname === "/v1/images/generations" || pathname === "/v1/images")) {
|
|
835
|
+
return withCors(await handleImageGenerations(opts, req, peer), req);
|
|
836
|
+
}
|
|
837
|
+
if (req.method === "POST" && pathname === "/v1/images/edits") {
|
|
838
|
+
return withCors(await handleImageEdits(opts, req, peer), req);
|
|
839
|
+
}
|
|
840
|
+
|
|
841
|
+
// Text-to-speech, OpenAI/OpenRouter wire; answers raw audio bytes.
|
|
842
|
+
if (req.method === "POST" && pathname === "/v1/audio/speech") {
|
|
843
|
+
return withCors(await handleSpeech(opts, req, peer), req);
|
|
844
|
+
}
|
|
845
|
+
|
|
846
|
+
// Speech-to-text, OpenAI multipart or OpenRouter JSON base64 wire.
|
|
847
|
+
if (req.method === "POST" && pathname === "/v1/audio/transcriptions") {
|
|
848
|
+
return withCors(await handleTranscriptions(opts, req, peer), req);
|
|
849
|
+
}
|
|
850
|
+
|
|
851
|
+
// Embeddings, OpenAI wire (OpenRouter is compatible).
|
|
852
|
+
if (req.method === "POST" && pathname === "/v1/embeddings") {
|
|
853
|
+
return withCors(await handleEmbeddings(opts, req, peer), req);
|
|
854
|
+
}
|
|
855
|
+
|
|
856
|
+
// Rerank, OpenRouter wire.
|
|
857
|
+
if (req.method === "POST" && pathname === "/v1/rerank") {
|
|
858
|
+
return withCors(await handleRerank(opts, req, peer), req);
|
|
859
|
+
}
|
|
860
|
+
|
|
861
|
+
// Video generation, OpenRouter's asynchronous wire: submit, then poll
|
|
862
|
+
// and download by the gateway-issued job id (stateless — the id
|
|
863
|
+
// encodes provider, model, and upstream job).
|
|
864
|
+
if (req.method === "POST" && pathname === "/v1/videos") {
|
|
865
|
+
return withCors(await handleVideoSubmit(opts, req, peer), req);
|
|
866
|
+
}
|
|
867
|
+
const videoJob = req.method === "GET" ? VIDEO_JOB_PATH.exec(pathname) : null;
|
|
868
|
+
if (videoJob) {
|
|
869
|
+
const gatewayId = decodeURIComponent(videoJob[1]);
|
|
870
|
+
const handler = videoJob[2] ? handleVideoContent : handleVideoPoll;
|
|
871
|
+
return withCors(await handler(opts, req, peer, gatewayId), req);
|
|
872
|
+
}
|
|
873
|
+
|
|
1007
874
|
// Model catalog.
|
|
1008
875
|
if (req.method === "GET" && pathname === "/v1/models") {
|
|
1009
876
|
return withCors(handleModelsList(opts), req);
|
|
@@ -59,6 +59,11 @@ export interface AuthGatewayParsedRequestOptions {
|
|
|
59
59
|
reasoning?: Effort;
|
|
60
60
|
/** Force-disable reasoning (Anthropic `thinking: { type: "disabled" }`). */
|
|
61
61
|
disableReasoning?: boolean;
|
|
62
|
+
/**
|
|
63
|
+
* Preserve an explicit wire-level reasoning-off request through providers
|
|
64
|
+
* that distinguish it from the generic disable hint.
|
|
65
|
+
*/
|
|
66
|
+
forceReasoningOff?: boolean;
|
|
62
67
|
/**
|
|
63
68
|
* Explicit Anthropic `thinking.budget_tokens`. Mirrors Rust's
|
|
64
69
|
* `resolve_thinking_budget`: pins onto whichever effort the client
|