@oh-my-pi/pi-ai 18.2.7 → 18.2.9
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +49 -18
- package/dist/types/auth-gateway/dispatch.d.ts +80 -0
- package/dist/types/auth-gateway/http.d.ts +5 -6
- package/dist/types/auth-gateway/index.d.ts +1 -0
- package/dist/types/auth-gateway/routes/embeddings.d.ts +3 -0
- package/dist/types/auth-gateway/routes/images.d.ts +3 -0
- package/dist/types/auth-gateway/routes/rerank.d.ts +3 -0
- package/dist/types/auth-gateway/routes/speech.d.ts +2 -0
- package/dist/types/auth-gateway/routes/systemone.d.ts +2 -0
- package/dist/types/auth-gateway/routes/transcriptions.d.ts +3 -0
- package/dist/types/auth-gateway/routes/video.d.ts +7 -0
- package/dist/types/auth-gateway/server.d.ts +10 -16
- package/dist/types/auth-gateway/types.d.ts +5 -0
- package/dist/types/auth-storage.d.ts +35 -32
- package/dist/types/embeddings/index.d.ts +7 -0
- package/dist/types/embeddings/openai-embeddings.d.ts +15 -0
- package/dist/types/embeddings/types.d.ts +15 -0
- package/dist/types/images/google-antigravity.d.ts +9 -0
- package/dist/types/images/google-generative-ai.d.ts +3 -0
- package/dist/types/images/index.d.ts +14 -0
- package/dist/types/images/openai-hosted.d.ts +3 -0
- package/dist/types/images/openai-images.d.ts +5 -0
- package/dist/types/images/openrouter-images.d.ts +3 -0
- package/dist/types/images/shared.d.ts +34 -0
- package/dist/types/images/types.d.ts +31 -0
- package/dist/types/index.d.ts +8 -1
- package/dist/types/judgment/typesafe.d.ts +2 -0
- package/dist/types/providers/amazon-bedrock.d.ts +7 -0
- package/dist/types/providers/claude-code-fingerprint.d.ts +22 -3
- package/dist/types/providers/embeddings-server.d.ts +32 -0
- package/dist/types/providers/google-gemini-cli.d.ts +0 -2
- package/dist/types/providers/images-server.d.ts +22 -0
- package/dist/types/providers/openai-chat-server-schema.d.ts +2 -2
- package/dist/types/providers/rerank-server.d.ts +35 -0
- package/dist/types/providers/speech-server.d.ts +8 -0
- package/dist/types/providers/systemone-server.d.ts +26 -0
- package/dist/types/providers/transcriptions-server.d.ts +32 -0
- package/dist/types/providers/video-server.d.ts +39 -0
- package/dist/types/rerank/index.d.ts +7 -0
- package/dist/types/rerank/openrouter-rerank.d.ts +15 -0
- package/dist/types/rerank/types.d.ts +17 -0
- package/dist/types/speech/index.d.ts +13 -0
- package/dist/types/speech/openai-speech.d.ts +3 -0
- package/dist/types/speech/transport.d.ts +7 -0
- package/dist/types/speech/types.d.ts +24 -0
- package/dist/types/speech/xai-tts.d.ts +7 -0
- package/dist/types/transcription/index.d.ts +7 -0
- package/dist/types/transcription/openai-transcriptions.d.ts +15 -0
- package/dist/types/transcription/types.d.ts +41 -0
- package/dist/types/usage/claude-api.d.ts +22 -0
- package/dist/types/usage/claude-reset.d.ts +44 -0
- package/dist/types/usage.d.ts +111 -5
- package/dist/types/utils/schema/json-schema-validator.d.ts +5 -2
- package/dist/types/utils/tool-call-loop-guard.d.ts +1 -1
- package/dist/types/video/index.d.ts +11 -0
- package/dist/types/video/openrouter-video.d.ts +19 -0
- package/dist/types/video/types.d.ts +62 -0
- package/package.json +30 -6
- package/src/auth/sqlite-credential-store.ts +44 -1
- package/src/auth-broker/remote-store.ts +6 -6
- package/src/auth-broker/wire-schemas.ts +14 -0
- package/src/auth-gateway/dispatch.ts +273 -0
- package/src/auth-gateway/http.ts +6 -7
- package/src/auth-gateway/index.ts +1 -0
- package/src/auth-gateway/routes/embeddings.ts +98 -0
- package/src/auth-gateway/routes/images.ts +131 -0
- package/src/auth-gateway/routes/rerank.ts +87 -0
- package/src/auth-gateway/routes/speech.ts +101 -0
- package/src/auth-gateway/routes/systemone.ts +116 -0
- package/src/auth-gateway/routes/transcriptions.ts +98 -0
- package/src/auth-gateway/routes/video.ts +243 -0
- package/src/auth-gateway/server.ts +127 -260
- package/src/auth-gateway/types.ts +5 -0
- package/src/auth-storage.ts +263 -140
- package/src/embeddings/index.ts +17 -0
- package/src/embeddings/openai-embeddings.ts +141 -0
- package/src/embeddings/types.ts +14 -0
- package/src/error/flags.ts +10 -0
- package/src/error/rate-limit.ts +1 -1
- package/src/images/google-antigravity.ts +180 -0
- package/src/images/google-generative-ai.ts +92 -0
- package/src/images/index.ts +59 -0
- package/src/images/openai-hosted.ts +185 -0
- package/src/images/openai-images.ts +110 -0
- package/src/images/openrouter-images.ts +33 -0
- package/src/images/shared.ts +193 -0
- package/src/images/types.ts +36 -0
- package/src/index.ts +8 -1
- package/src/judgment/typesafe.ts +5 -0
- package/src/providers/amazon-bedrock.ts +55 -5
- package/src/providers/anthropic.ts +45 -11
- package/src/providers/aws-credentials.ts +124 -11
- package/src/providers/claude-code-fingerprint.ts +55 -3
- package/src/providers/embeddings-server.ts +151 -0
- package/src/providers/gitlab-duo.ts +20 -4
- package/src/providers/google-gemini-cli.ts +0 -8
- package/src/providers/google-shared.ts +1 -18
- package/src/providers/images-server.ts +159 -0
- package/src/providers/openai-chat-server-schema.ts +1 -1
- package/src/providers/openai-chat-server.ts +3 -1
- package/src/providers/openai-codex-responses.ts +27 -5
- package/src/providers/openai-completions.ts +122 -19
- package/src/providers/pi-native-server.ts +1 -0
- package/src/providers/rerank-server.ts +166 -0
- package/src/providers/speech-server.ts +53 -0
- package/src/providers/systemone-server.ts +73 -0
- package/src/providers/transcriptions-server.ts +243 -0
- package/src/providers/video-server.ts +286 -0
- package/src/registry/oauth/anthropic.ts +2 -3
- package/src/rerank/index.ts +13 -0
- package/src/rerank/openrouter-rerank.ts +136 -0
- package/src/rerank/types.ts +20 -0
- package/src/speech/index.ts +35 -0
- package/src/speech/openai-speech.ts +26 -0
- package/src/speech/transport.ts +66 -0
- package/src/speech/types.ts +37 -0
- package/src/speech/xai-tts.ts +41 -0
- package/src/stream.ts +13 -4
- package/src/transcription/index.ts +17 -0
- package/src/transcription/openai-transcriptions.ts +133 -0
- package/src/transcription/types.ts +46 -0
- package/src/usage/alibaba-token-plan.ts +7 -1
- package/src/usage/claude-api.ts +66 -0
- package/src/usage/claude-reset.ts +638 -0
- package/src/usage/claude.ts +37 -59
- package/src/usage/kimi.ts +32 -1
- package/src/usage.ts +52 -5
- package/src/utils/schema/json-schema-validator.ts +23 -10
- package/src/utils/tool-call-loop-guard.ts +2 -2
- package/src/utils/validation.ts +145 -50
- package/src/video/index.ts +34 -0
- package/src/video/openrouter-video.ts +210 -0
- package/src/video/types.ts +72 -0
|
@@ -0,0 +1,273 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Per-request plumbing shared by every auth-gateway route.
|
|
3
|
+
*
|
|
4
|
+
* A route module (`server.ts` chat/pi-native handlers, `routes/*.ts` for
|
|
5
|
+
* judgments, images, speech, transcription, …) owns only its wire format:
|
|
6
|
+
* parse the body, pick a model, call the pi-ai client, encode the reply.
|
|
7
|
+
* Everything credential-shaped lives here so each route drives the same
|
|
8
|
+
* broker-backed rotation policy and the same usage ledger.
|
|
9
|
+
*/
|
|
10
|
+
import { extractHttpStatusFromError, logger } from "@oh-my-pi/pi-utils";
|
|
11
|
+
import type { ApiKeyResolver } from "../auth-retry";
|
|
12
|
+
import type { AuthStorage } from "../auth-storage";
|
|
13
|
+
import * as AIError from "../error";
|
|
14
|
+
import { classifyGatewayError, type GatewayErrorClassification } from "../error/gateway";
|
|
15
|
+
import { isUsageLimitOutcome } from "../error/rate-limit";
|
|
16
|
+
import type { Api, FetchImpl, Model, Usage } from "../types";
|
|
17
|
+
import type { ClientUsageIdentity } from "../usage";
|
|
18
|
+
import { extractProviderRetryHint } from "../utils/retry-after";
|
|
19
|
+
import type { AuthGatewayServerOptions } from "./types";
|
|
20
|
+
|
|
21
|
+
export type ModelResolver = (modelId: string) => Model<Api> | undefined;
|
|
22
|
+
|
|
23
|
+
export interface AuthGatewayBootOptions extends AuthGatewayServerOptions {
|
|
24
|
+
/** Source of credentials. Caller wires this to a broker-backed AuthStorage. */
|
|
25
|
+
storage: AuthStorage;
|
|
26
|
+
/**
|
|
27
|
+
* Resolve a client-requested model id to a pi-ai Model. Caller supplies
|
|
28
|
+
* this from a ModelRegistry (lives in `coding-agent` to avoid an inverse
|
|
29
|
+
* dependency in `pi-ai`).
|
|
30
|
+
*/
|
|
31
|
+
resolveModel: ModelResolver;
|
|
32
|
+
/** Optional supplier for `/v1/models` listing. Returns the full model array. */
|
|
33
|
+
listModels?: () => Iterable<Model<Api>>;
|
|
34
|
+
/** Upstream transport for every provider call; defaults to global `fetch`. Test seam. */
|
|
35
|
+
fetch?: FetchImpl;
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
/**
|
|
39
|
+
* The client's own session key, or `undefined` when it sent none. A blank key
|
|
40
|
+
* counts as none: honouring it would collapse every caller that sends an empty
|
|
41
|
+
* key into one shared credential-sticky, prefix-cache and provider-session
|
|
42
|
+
* bucket.
|
|
43
|
+
*/
|
|
44
|
+
export function normalizeClientSessionKey(clientKey: string | undefined): string | undefined {
|
|
45
|
+
return clientKey !== undefined && clientKey.trim().length > 0 ? clientKey : undefined;
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
/**
|
|
49
|
+
* Stable identity of the account a request's credential belongs to.
|
|
50
|
+
*
|
|
51
|
+
* `markUsageLimitReached` and the auth-retry resolver switch a session to a
|
|
52
|
+
* sibling credential, so the provider state retained for that session can
|
|
53
|
+
* outlive the account that taught it. OAuth rows expose an account id / email
|
|
54
|
+
* that survives token refresh — fingerprinting the bearer instead would look
|
|
55
|
+
* like a rotation every time a token refreshes and discard the retained
|
|
56
|
+
* lessons for nothing. Key-based rows fall back to a hash of the key, never
|
|
57
|
+
* the key itself: this value is held for the lifetime of the entry.
|
|
58
|
+
*/
|
|
59
|
+
export function resolveGatewayAccount(
|
|
60
|
+
storage: AuthStorage,
|
|
61
|
+
provider: string,
|
|
62
|
+
sessionId: string,
|
|
63
|
+
apiKey: string,
|
|
64
|
+
): string {
|
|
65
|
+
const identity = storage.getOAuthAccountIdentity(provider, sessionId);
|
|
66
|
+
if (identity) {
|
|
67
|
+
return `oauth:${JSON.stringify([
|
|
68
|
+
identity.accountId ?? "",
|
|
69
|
+
identity.email ?? "",
|
|
70
|
+
identity.projectId ?? "",
|
|
71
|
+
identity.orgId ?? "",
|
|
72
|
+
])}`;
|
|
73
|
+
}
|
|
74
|
+
return `key:${Bun.hash(apiKey).toString(36)}`;
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
/**
|
|
78
|
+
* Resolve the credential for one request from broker-backed storage.
|
|
79
|
+
*
|
|
80
|
+
* pi-ai clients never consult `AuthStorage`; the gateway resolves the bearer
|
|
81
|
+
* (an OAuth access token refreshed through the broker when needed) and hands
|
|
82
|
+
* it to the client. Returns the key, or the error classification the route
|
|
83
|
+
* should encode in its own envelope: storage failures map through
|
|
84
|
+
* {@link classifyGatewayError}, a provider without any credential is a 401.
|
|
85
|
+
*/
|
|
86
|
+
export async function resolveGatewayApiKey(
|
|
87
|
+
storage: AuthStorage,
|
|
88
|
+
model: Model<Api>,
|
|
89
|
+
sessionId: string,
|
|
90
|
+
signal: AbortSignal,
|
|
91
|
+
peer: string,
|
|
92
|
+
): Promise<string | GatewayErrorClassification> {
|
|
93
|
+
let apiKey: string | undefined;
|
|
94
|
+
try {
|
|
95
|
+
apiKey = await storage.getApiKey(model.provider, sessionId, { modelId: model.id, signal });
|
|
96
|
+
} catch (error) {
|
|
97
|
+
const classified = classifyGatewayError(error);
|
|
98
|
+
logger.warn("auth-gateway getApiKey threw", { provider: model.provider, peer, error: classified.message });
|
|
99
|
+
return classified;
|
|
100
|
+
}
|
|
101
|
+
if (apiKey) return apiKey;
|
|
102
|
+
return {
|
|
103
|
+
status: 401,
|
|
104
|
+
type: "authentication_error",
|
|
105
|
+
message: `No credential available for provider ${model.provider}`,
|
|
106
|
+
};
|
|
107
|
+
}
|
|
108
|
+
|
|
109
|
+
/**
|
|
110
|
+
* Hook fired by a pi-ai client when the upstream request fails in a way
|
|
111
|
+
* that's rotatable — today that's HTTP 401 (credential is bad) and
|
|
112
|
+
* usage-limit phrasing matched by {@link isUsageLimitError} (Codex's
|
|
113
|
+
* `usage_limit_reached`, Anthropic's `usage_limit_reached`, Google's
|
|
114
|
+
* `resource_exhausted`, …). The two cases need different storage actions:
|
|
115
|
+
*
|
|
116
|
+
* - **usage-limit** → {@link AuthStorage.markUsageLimitReached}. Marks just
|
|
117
|
+
* the current session's credential as temporarily blocked (honouring
|
|
118
|
+
* `retry-after` / `resets_at` hints when present) and returns `true` only
|
|
119
|
+
* when a sibling credential is still available. Burning the credential
|
|
120
|
+
* with `invalidateCredentialMatching` here would orphan accounts whose
|
|
121
|
+
* reset window is several hours away — exactly the bug this helper exists
|
|
122
|
+
* to avoid.
|
|
123
|
+
* - **auth-failure** → {@link AuthStorage.invalidateCredentialMatching}.
|
|
124
|
+
* Suspect/delete the row so it doesn't get re-picked next request.
|
|
125
|
+
*
|
|
126
|
+
* In both branches we return the next `getApiKey` result (sticky on the
|
|
127
|
+
* same `sessionId`) so the client can transparently retry the pre-emit
|
|
128
|
+
* failure with a fresh credential. Returning `undefined` aborts the retry
|
|
129
|
+
* and surfaces the original error to the caller.
|
|
130
|
+
*/
|
|
131
|
+
async function refreshGatewayApiKeyAfterAuthError(
|
|
132
|
+
storage: AuthStorage,
|
|
133
|
+
model: Model<Api>,
|
|
134
|
+
sessionId: string,
|
|
135
|
+
provider: string,
|
|
136
|
+
oldKey: string,
|
|
137
|
+
error: unknown,
|
|
138
|
+
signal: AbortSignal,
|
|
139
|
+
format: string,
|
|
140
|
+
peer: string,
|
|
141
|
+
): Promise<string | undefined> {
|
|
142
|
+
const message = error instanceof Error ? error.message : String(error);
|
|
143
|
+
const status = extractHttpStatusFromError(error);
|
|
144
|
+
if (AIError.isUsageLimit(error) || isUsageLimitOutcome(status, message)) {
|
|
145
|
+
const retryAfterMs = extractProviderRetryHint(provider, message);
|
|
146
|
+
const { switched, retryAtMs } = await storage.markUsageLimitReached(provider, sessionId, {
|
|
147
|
+
retryAfterMs,
|
|
148
|
+
providerTimed: retryAfterMs !== undefined,
|
|
149
|
+
baseUrl: model.baseUrl,
|
|
150
|
+
modelId: model.id,
|
|
151
|
+
apiKey: oldKey,
|
|
152
|
+
signal,
|
|
153
|
+
});
|
|
154
|
+
logger.debug("auth-gateway retrying provider request after usage-limit block", {
|
|
155
|
+
format,
|
|
156
|
+
provider,
|
|
157
|
+
peer,
|
|
158
|
+
switched,
|
|
159
|
+
retryAfterMs,
|
|
160
|
+
retryAtMs,
|
|
161
|
+
error: message,
|
|
162
|
+
});
|
|
163
|
+
if (!switched) return undefined;
|
|
164
|
+
return storage.getApiKey(provider, sessionId, { modelId: model.id, signal });
|
|
165
|
+
}
|
|
166
|
+
await storage.invalidateCredentialMatching(provider, oldKey, { sessionId, signal });
|
|
167
|
+
logger.debug("auth-gateway retrying provider request after credential invalidation", {
|
|
168
|
+
format,
|
|
169
|
+
provider,
|
|
170
|
+
peer,
|
|
171
|
+
error: message,
|
|
172
|
+
});
|
|
173
|
+
return storage.getApiKey(provider, sessionId, { modelId: model.id, signal });
|
|
174
|
+
}
|
|
175
|
+
|
|
176
|
+
/**
|
|
177
|
+
* Build the {@link ApiKeyResolver} handed to a pi-ai client for a gateway
|
|
178
|
+
* request. Drives the central a/b/c auth-retry policy server-side:
|
|
179
|
+
*
|
|
180
|
+
* - initial resolve → the credential already resolved for this request.
|
|
181
|
+
* - step (b) `!lastChance` → force-refresh the SAME session-sticky credential
|
|
182
|
+
* (a peer/broker may have rotated its token out from under our cached copy).
|
|
183
|
+
* - step (c) `lastChance` → {@link refreshGatewayApiKeyAfterAuthError} switches
|
|
184
|
+
* to a sibling (usage-limit block vs credential invalidation by error class).
|
|
185
|
+
*
|
|
186
|
+
* `lastKey` tracks the most recent bearer so the switch step invalidates the
|
|
187
|
+
* credential that actually failed. `onResolvedKey` observes every rotation;
|
|
188
|
+
* routes that retain provider session state use it to re-key the account
|
|
189
|
+
* lease, one-shot routes pass `undefined`.
|
|
190
|
+
*/
|
|
191
|
+
export function buildGatewayApiKeyResolver(
|
|
192
|
+
storage: AuthStorage,
|
|
193
|
+
model: Model<Api>,
|
|
194
|
+
sessionId: string,
|
|
195
|
+
initialKey: string,
|
|
196
|
+
requestSignal: AbortSignal,
|
|
197
|
+
format: string,
|
|
198
|
+
peer: string,
|
|
199
|
+
onResolvedKey?: (apiKey: string) => void,
|
|
200
|
+
): ApiKeyResolver {
|
|
201
|
+
let lastKey = initialKey;
|
|
202
|
+
return async ({ lastChance, error, signal }) => {
|
|
203
|
+
const sig = signal ?? requestSignal;
|
|
204
|
+
if (error === undefined) {
|
|
205
|
+
lastKey = initialKey;
|
|
206
|
+
return initialKey;
|
|
207
|
+
}
|
|
208
|
+
if (!lastChance) {
|
|
209
|
+
const refreshed = await storage.getApiKey(model.provider, sessionId, {
|
|
210
|
+
modelId: model.id,
|
|
211
|
+
signal: sig,
|
|
212
|
+
forceRefresh: true,
|
|
213
|
+
});
|
|
214
|
+
lastKey = refreshed ?? lastKey;
|
|
215
|
+
if (refreshed) onResolvedKey?.(refreshed);
|
|
216
|
+
return refreshed;
|
|
217
|
+
}
|
|
218
|
+
const next = await refreshGatewayApiKeyAfterAuthError(
|
|
219
|
+
storage,
|
|
220
|
+
model,
|
|
221
|
+
sessionId,
|
|
222
|
+
model.provider,
|
|
223
|
+
lastKey,
|
|
224
|
+
error,
|
|
225
|
+
sig,
|
|
226
|
+
format,
|
|
227
|
+
peer,
|
|
228
|
+
);
|
|
229
|
+
lastKey = next ?? lastKey;
|
|
230
|
+
if (next) onResolvedKey?.(next);
|
|
231
|
+
return next;
|
|
232
|
+
};
|
|
233
|
+
}
|
|
234
|
+
|
|
235
|
+
/**
|
|
236
|
+
* Attribute one settled upstream request to the originating client via the
|
|
237
|
+
* broker's observed-usage channel (`AuthStorage.recordObservedUsage`, batched
|
|
238
|
+
* by the remote store). Error/aborted turns still record — the provider
|
|
239
|
+
* billed whatever tokens the partial turn consumed; zero-usage results
|
|
240
|
+
* (pre-flight failures) are skipped. `at` defaults to now.
|
|
241
|
+
*/
|
|
242
|
+
export function recordGatewayUsage(
|
|
243
|
+
storage: AuthStorage,
|
|
244
|
+
model: Model<Api>,
|
|
245
|
+
client: ClientUsageIdentity,
|
|
246
|
+
usage: Usage,
|
|
247
|
+
at?: number,
|
|
248
|
+
): void {
|
|
249
|
+
if (usage.input + usage.output + usage.cacheRead + usage.cacheWrite === 0) return;
|
|
250
|
+
storage.recordObservedUsage({
|
|
251
|
+
provider: model.provider,
|
|
252
|
+
model: model.id,
|
|
253
|
+
at,
|
|
254
|
+
usage: { input: usage.input, output: usage.output, cacheRead: usage.cacheRead, cacheWrite: usage.cacheWrite },
|
|
255
|
+
costUsd: usage.cost.total,
|
|
256
|
+
client,
|
|
257
|
+
});
|
|
258
|
+
}
|
|
259
|
+
|
|
260
|
+
/**
|
|
261
|
+
* An `AbortController` that follows the inbound request's abort signal. Routes
|
|
262
|
+
* abort it themselves when the response body is cancelled mid-stream, which
|
|
263
|
+
* `req.signal` alone does not observe.
|
|
264
|
+
*/
|
|
265
|
+
export function mirrorRequestAbort(req: Request): AbortController {
|
|
266
|
+
const controller = new AbortController();
|
|
267
|
+
if (req.signal.aborted) {
|
|
268
|
+
controller.abort(req.signal.reason);
|
|
269
|
+
} else {
|
|
270
|
+
req.signal.addEventListener("abort", () => controller.abort(req.signal.reason), { once: true });
|
|
271
|
+
}
|
|
272
|
+
return controller;
|
|
273
|
+
}
|
package/src/auth-gateway/http.ts
CHANGED
|
@@ -7,7 +7,7 @@
|
|
|
7
7
|
import { timingSafeEqual as nodeTimingSafeEqual } from "node:crypto";
|
|
8
8
|
import * as os from "node:os";
|
|
9
9
|
import { getInstallId } from "@oh-my-pi/pi-utils";
|
|
10
|
-
import type { Api,
|
|
10
|
+
import type { Api, Model } from "../types";
|
|
11
11
|
import type { ClientUsageIdentity } from "../usage";
|
|
12
12
|
|
|
13
13
|
const JSON_HEADERS = {
|
|
@@ -28,14 +28,13 @@ export function json(status: number, body: unknown, headers?: Record<string, str
|
|
|
28
28
|
* `request-id` (surfaced as `_request_id` by the OpenAI and Anthropic SDKs,
|
|
29
29
|
* matches the gateway log line), LiteLLM's model-resolution and cost headers,
|
|
30
30
|
* and OpenAI's `openai-processing-ms`. Model/request-id headers are always
|
|
31
|
-
* present; `
|
|
32
|
-
*
|
|
33
|
-
*
|
|
34
|
-
* only the identity headers.
|
|
31
|
+
* present; `costUsd` — known only once a non-streaming response has settled —
|
|
32
|
+
* adds the computed cost, and `startedAt` the wall time. Streaming responses
|
|
33
|
+
* send headers before usage exists, so they carry only the identity headers.
|
|
35
34
|
*/
|
|
36
35
|
export function gatewayResponseHeaders(
|
|
37
36
|
model: Model<Api>,
|
|
38
|
-
info: { requestId: string;
|
|
37
|
+
info: { requestId: string; costUsd?: number; startedAt?: number },
|
|
39
38
|
): Record<string, string> {
|
|
40
39
|
const headers: Record<string, string> = {
|
|
41
40
|
"x-request-id": info.requestId,
|
|
@@ -43,7 +42,7 @@ export function gatewayResponseHeaders(
|
|
|
43
42
|
"x-litellm-model-id": model.id,
|
|
44
43
|
};
|
|
45
44
|
if (model.baseUrl) headers["x-litellm-model-api-base"] = model.baseUrl;
|
|
46
|
-
if (info.
|
|
45
|
+
if (info.costUsd !== undefined) headers["x-litellm-response-cost"] = info.costUsd.toString();
|
|
47
46
|
if (info.startedAt !== undefined) {
|
|
48
47
|
const elapsed = (performance.now() - info.startedAt).toFixed(0);
|
|
49
48
|
headers["x-litellm-response-duration-ms"] = elapsed;
|
|
@@ -0,0 +1,98 @@
|
|
|
1
|
+
import { logger } from "@oh-my-pi/pi-utils";
|
|
2
|
+
import { embed } from "../../embeddings";
|
|
3
|
+
import { classifyGatewayError } from "../../error/gateway";
|
|
4
|
+
import * as embeddings from "../../providers/embeddings-server";
|
|
5
|
+
import { deterministicUuid } from "../../utils/deterministic-id";
|
|
6
|
+
import {
|
|
7
|
+
type AuthGatewayBootOptions,
|
|
8
|
+
buildGatewayApiKeyResolver,
|
|
9
|
+
mirrorRequestAbort,
|
|
10
|
+
recordGatewayUsage,
|
|
11
|
+
resolveGatewayApiKey,
|
|
12
|
+
} from "../dispatch";
|
|
13
|
+
import { gatewayResponseHeaders, json, resolveClientIdentity } from "../http";
|
|
14
|
+
|
|
15
|
+
/** OpenAI-compatible `POST /v1/embeddings` gateway handler. */
|
|
16
|
+
export async function handleEmbeddings(
|
|
17
|
+
bootOpts: AuthGatewayBootOptions,
|
|
18
|
+
req: Request,
|
|
19
|
+
peer: string,
|
|
20
|
+
): Promise<Response> {
|
|
21
|
+
const startedAt = performance.now();
|
|
22
|
+
const requestId = crypto.randomUUID();
|
|
23
|
+
const controller = mirrorRequestAbort(req);
|
|
24
|
+
const aborted = (): Response => embeddings.formatError(499, "request_aborted", "client closed request");
|
|
25
|
+
if (controller.signal.aborted) return aborted();
|
|
26
|
+
|
|
27
|
+
let parsed: embeddings.EmbeddingsParsedRequest;
|
|
28
|
+
try {
|
|
29
|
+
parsed = await embeddings.parseRequest(req);
|
|
30
|
+
} catch (error) {
|
|
31
|
+
if (controller.signal.aborted) return aborted();
|
|
32
|
+
const message = error instanceof Error ? error.message : String(error);
|
|
33
|
+
const status = error instanceof embeddings.EmbeddingsWireError ? error.status : 400;
|
|
34
|
+
return embeddings.formatError(status, "invalid_request_error", message);
|
|
35
|
+
}
|
|
36
|
+
if (controller.signal.aborted) return aborted();
|
|
37
|
+
|
|
38
|
+
const model = bootOpts.resolveModel(parsed.modelId);
|
|
39
|
+
if (!model) {
|
|
40
|
+
return embeddings.formatError(404, "invalid_request_error", `Unknown model: ${parsed.modelId}`);
|
|
41
|
+
}
|
|
42
|
+
if (model.api !== "openai-embeddings") {
|
|
43
|
+
return embeddings.formatError(
|
|
44
|
+
400,
|
|
45
|
+
"invalid_request_error",
|
|
46
|
+
`Model ${parsed.modelId} does not support embeddings`,
|
|
47
|
+
);
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
const client = resolveClientIdentity(req.headers);
|
|
51
|
+
const sessionId = deterministicUuid(`embeddings\u0000${model.provider}/${model.id}`);
|
|
52
|
+
const apiKey = await resolveGatewayApiKey(bootOpts.storage, model, sessionId, controller.signal, peer);
|
|
53
|
+
if (controller.signal.aborted) return aborted();
|
|
54
|
+
if (typeof apiKey !== "string") {
|
|
55
|
+
return embeddings.formatError(apiKey.status, apiKey.type, apiKey.message);
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
logger.info("auth-gateway request", {
|
|
59
|
+
requestId,
|
|
60
|
+
format: "embeddings",
|
|
61
|
+
model: parsed.modelId,
|
|
62
|
+
resolvedProvider: model.provider,
|
|
63
|
+
resolvedModel: model.id,
|
|
64
|
+
stream: false,
|
|
65
|
+
peer,
|
|
66
|
+
});
|
|
67
|
+
|
|
68
|
+
try {
|
|
69
|
+
const result = await embed(model, parsed.request, {
|
|
70
|
+
apiKey: buildGatewayApiKeyResolver(
|
|
71
|
+
bootOpts.storage,
|
|
72
|
+
model,
|
|
73
|
+
sessionId,
|
|
74
|
+
apiKey,
|
|
75
|
+
controller.signal,
|
|
76
|
+
"embeddings",
|
|
77
|
+
peer,
|
|
78
|
+
),
|
|
79
|
+
fetch: bootOpts.fetch,
|
|
80
|
+
signal: controller.signal,
|
|
81
|
+
});
|
|
82
|
+
recordGatewayUsage(bootOpts.storage, model, client, result.usage);
|
|
83
|
+
return json(
|
|
84
|
+
200,
|
|
85
|
+
embeddings.encodeResponse(result, parsed.modelId),
|
|
86
|
+
gatewayResponseHeaders(model, { requestId, costUsd: result.usage.cost.total, startedAt }),
|
|
87
|
+
);
|
|
88
|
+
} catch (error) {
|
|
89
|
+
if (controller.signal.aborted) return aborted();
|
|
90
|
+
const classified = classifyGatewayError(error);
|
|
91
|
+
logger.warn("auth-gateway embeddings failed", {
|
|
92
|
+
format: "embeddings",
|
|
93
|
+
error: classified.message,
|
|
94
|
+
peer,
|
|
95
|
+
});
|
|
96
|
+
return embeddings.formatError(classified.status, classified.type, classified.message);
|
|
97
|
+
}
|
|
98
|
+
}
|
|
@@ -0,0 +1,131 @@
|
|
|
1
|
+
import { calculateCost } from "@oh-my-pi/pi-catalog/models";
|
|
2
|
+
import { logger } from "@oh-my-pi/pi-utils";
|
|
3
|
+
import { classifyGatewayError } from "../../error/gateway";
|
|
4
|
+
import { generateImage } from "../../images";
|
|
5
|
+
import * as imagesServer from "../../providers/images-server";
|
|
6
|
+
import { deterministicUuid } from "../../utils/deterministic-id";
|
|
7
|
+
import {
|
|
8
|
+
type AuthGatewayBootOptions,
|
|
9
|
+
buildGatewayApiKeyResolver,
|
|
10
|
+
mirrorRequestAbort,
|
|
11
|
+
recordGatewayUsage,
|
|
12
|
+
resolveGatewayApiKey,
|
|
13
|
+
} from "../dispatch";
|
|
14
|
+
import { gatewayResponseHeaders, json, resolveClientIdentity } from "../http";
|
|
15
|
+
|
|
16
|
+
function isGatewayImageApi(api: string): boolean {
|
|
17
|
+
return (
|
|
18
|
+
api === "openai-images" ||
|
|
19
|
+
api === "openrouter-images" ||
|
|
20
|
+
api === "google-generative-ai" ||
|
|
21
|
+
api === "google-gemini-cli"
|
|
22
|
+
);
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
async function handleImages(
|
|
26
|
+
bootOpts: AuthGatewayBootOptions,
|
|
27
|
+
req: Request,
|
|
28
|
+
peer: string,
|
|
29
|
+
kind: imagesServer.ImageRequestKind,
|
|
30
|
+
): Promise<Response> {
|
|
31
|
+
const startedAt = performance.now();
|
|
32
|
+
const requestId = crypto.randomUUID();
|
|
33
|
+
const controller = mirrorRequestAbort(req);
|
|
34
|
+
const aborted = (): Response => imagesServer.formatError(499, "request_aborted", "client closed request");
|
|
35
|
+
if (controller.signal.aborted) return aborted();
|
|
36
|
+
|
|
37
|
+
let body: unknown | FormData;
|
|
38
|
+
try {
|
|
39
|
+
body =
|
|
40
|
+
kind === "edits" && req.headers.get("content-type")?.includes("multipart/form-data")
|
|
41
|
+
? await req.formData()
|
|
42
|
+
: await req.json();
|
|
43
|
+
} catch (error) {
|
|
44
|
+
if (controller.signal.aborted) return aborted();
|
|
45
|
+
return imagesServer.formatError(400, "invalid_request_error", `Invalid request body: ${String(error)}`);
|
|
46
|
+
}
|
|
47
|
+
if (controller.signal.aborted) return aborted();
|
|
48
|
+
|
|
49
|
+
let parsed: imagesServer.ImagesParsedRequest;
|
|
50
|
+
try {
|
|
51
|
+
parsed = await imagesServer.parseRequest(body, kind);
|
|
52
|
+
} catch (error) {
|
|
53
|
+
const message = error instanceof Error ? error.message : String(error);
|
|
54
|
+
return imagesServer.formatError(400, "invalid_request_error", message);
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
const model = bootOpts.resolveModel(parsed.modelId);
|
|
58
|
+
if (!model) {
|
|
59
|
+
return imagesServer.formatError(404, "invalid_request_error", `Unknown model: ${parsed.modelId}`);
|
|
60
|
+
}
|
|
61
|
+
if (model.api === "openai-responses" || model.api === "openai-codex-responses") {
|
|
62
|
+
return imagesServer.formatError(
|
|
63
|
+
400,
|
|
64
|
+
"invalid_request_error",
|
|
65
|
+
`Model ${parsed.modelId} requires a hosted image carrier, which the auth gateway cannot resolve`,
|
|
66
|
+
);
|
|
67
|
+
}
|
|
68
|
+
if (!isGatewayImageApi(model.api)) {
|
|
69
|
+
return imagesServer.formatError(
|
|
70
|
+
400,
|
|
71
|
+
"invalid_request_error",
|
|
72
|
+
`Model ${parsed.modelId} does not support image generation`,
|
|
73
|
+
);
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
const client = resolveClientIdentity(req.headers);
|
|
77
|
+
const sessionId = deterministicUuid(`images\u0000${model.provider}/${model.id}`);
|
|
78
|
+
const apiKey = await resolveGatewayApiKey(bootOpts.storage, model, sessionId, controller.signal, peer);
|
|
79
|
+
if (controller.signal.aborted) return aborted();
|
|
80
|
+
if (typeof apiKey !== "string") return imagesServer.formatError(apiKey.status, apiKey.type, apiKey.message);
|
|
81
|
+
|
|
82
|
+
logger.info("auth-gateway request", {
|
|
83
|
+
requestId,
|
|
84
|
+
format: "images",
|
|
85
|
+
model: parsed.modelId,
|
|
86
|
+
resolvedProvider: model.provider,
|
|
87
|
+
resolvedModel: model.id,
|
|
88
|
+
stream: false,
|
|
89
|
+
peer,
|
|
90
|
+
});
|
|
91
|
+
|
|
92
|
+
try {
|
|
93
|
+
const result = await generateImage(model, parsed.request, {
|
|
94
|
+
apiKey: buildGatewayApiKeyResolver(
|
|
95
|
+
bootOpts.storage,
|
|
96
|
+
model,
|
|
97
|
+
sessionId,
|
|
98
|
+
apiKey,
|
|
99
|
+
controller.signal,
|
|
100
|
+
"images",
|
|
101
|
+
peer,
|
|
102
|
+
),
|
|
103
|
+
fetch: bootOpts.fetch,
|
|
104
|
+
signal: controller.signal,
|
|
105
|
+
});
|
|
106
|
+
if (result.usage.cost.total === 0) calculateCost(model, result.usage);
|
|
107
|
+
recordGatewayUsage(bootOpts.storage, model, client, result.usage);
|
|
108
|
+
return json(
|
|
109
|
+
200,
|
|
110
|
+
imagesServer.encodeResponse(result, parsed.modelId),
|
|
111
|
+
gatewayResponseHeaders(model, { requestId, costUsd: result.usage.cost.total, startedAt }),
|
|
112
|
+
);
|
|
113
|
+
} catch (error) {
|
|
114
|
+
if (controller.signal.aborted) return aborted();
|
|
115
|
+
const classified = classifyGatewayError(error);
|
|
116
|
+
logger.warn("auth-gateway image generation failed", { format: "images", error: classified.message, peer });
|
|
117
|
+
return imagesServer.formatError(classified.status, classified.type, classified.message);
|
|
118
|
+
}
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
export function handleImageGenerations(
|
|
122
|
+
bootOpts: AuthGatewayBootOptions,
|
|
123
|
+
req: Request,
|
|
124
|
+
peer: string,
|
|
125
|
+
): Promise<Response> {
|
|
126
|
+
return handleImages(bootOpts, req, peer, "generations");
|
|
127
|
+
}
|
|
128
|
+
|
|
129
|
+
export function handleImageEdits(bootOpts: AuthGatewayBootOptions, req: Request, peer: string): Promise<Response> {
|
|
130
|
+
return handleImages(bootOpts, req, peer, "edits");
|
|
131
|
+
}
|
|
@@ -0,0 +1,87 @@
|
|
|
1
|
+
import { logger } from "@oh-my-pi/pi-utils";
|
|
2
|
+
import { classifyGatewayError } from "../../error/gateway";
|
|
3
|
+
import * as rerankWire from "../../providers/rerank-server";
|
|
4
|
+
import { rerank } from "../../rerank";
|
|
5
|
+
import { deterministicUuid } from "../../utils/deterministic-id";
|
|
6
|
+
import {
|
|
7
|
+
type AuthGatewayBootOptions,
|
|
8
|
+
buildGatewayApiKeyResolver,
|
|
9
|
+
mirrorRequestAbort,
|
|
10
|
+
recordGatewayUsage,
|
|
11
|
+
resolveGatewayApiKey,
|
|
12
|
+
} from "../dispatch";
|
|
13
|
+
import { gatewayResponseHeaders, json, resolveClientIdentity } from "../http";
|
|
14
|
+
|
|
15
|
+
/** OpenRouter-compatible `POST /v1/rerank` gateway handler. */
|
|
16
|
+
export async function handleRerank(bootOpts: AuthGatewayBootOptions, req: Request, peer: string): Promise<Response> {
|
|
17
|
+
const startedAt = performance.now();
|
|
18
|
+
const requestId = crypto.randomUUID();
|
|
19
|
+
const controller = mirrorRequestAbort(req);
|
|
20
|
+
const aborted = (): Response => rerankWire.formatError(499, "request_aborted", "client closed request");
|
|
21
|
+
if (controller.signal.aborted) return aborted();
|
|
22
|
+
|
|
23
|
+
let parsed: rerankWire.RerankParsedRequest;
|
|
24
|
+
try {
|
|
25
|
+
parsed = await rerankWire.parseRequest(req);
|
|
26
|
+
} catch (error) {
|
|
27
|
+
if (controller.signal.aborted) return aborted();
|
|
28
|
+
const message = error instanceof Error ? error.message : String(error);
|
|
29
|
+
const status = error instanceof rerankWire.RerankWireError ? error.status : 400;
|
|
30
|
+
return rerankWire.formatError(status, "invalid_request_error", message);
|
|
31
|
+
}
|
|
32
|
+
if (controller.signal.aborted) return aborted();
|
|
33
|
+
|
|
34
|
+
const model = bootOpts.resolveModel(parsed.modelId);
|
|
35
|
+
if (!model) return rerankWire.formatError(404, "invalid_request_error", `Unknown model: ${parsed.modelId}`);
|
|
36
|
+
if (model.api !== "openrouter-rerank") {
|
|
37
|
+
return rerankWire.formatError(400, "invalid_request_error", `Model ${parsed.modelId} does not support reranking`);
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
const client = resolveClientIdentity(req.headers);
|
|
41
|
+
const sessionId = deterministicUuid(`rerank\u0000${model.provider}/${model.id}`);
|
|
42
|
+
const apiKey = await resolveGatewayApiKey(bootOpts.storage, model, sessionId, controller.signal, peer);
|
|
43
|
+
if (controller.signal.aborted) return aborted();
|
|
44
|
+
if (typeof apiKey !== "string") return rerankWire.formatError(apiKey.status, apiKey.type, apiKey.message);
|
|
45
|
+
|
|
46
|
+
logger.info("auth-gateway request", {
|
|
47
|
+
requestId,
|
|
48
|
+
format: "rerank",
|
|
49
|
+
model: parsed.modelId,
|
|
50
|
+
resolvedProvider: model.provider,
|
|
51
|
+
resolvedModel: model.id,
|
|
52
|
+
stream: false,
|
|
53
|
+
peer,
|
|
54
|
+
});
|
|
55
|
+
|
|
56
|
+
try {
|
|
57
|
+
const result = await rerank(model, parsed.request, {
|
|
58
|
+
apiKey: buildGatewayApiKeyResolver(
|
|
59
|
+
bootOpts.storage,
|
|
60
|
+
model,
|
|
61
|
+
sessionId,
|
|
62
|
+
apiKey,
|
|
63
|
+
controller.signal,
|
|
64
|
+
"rerank",
|
|
65
|
+
peer,
|
|
66
|
+
),
|
|
67
|
+
fetch: bootOpts.fetch,
|
|
68
|
+
signal: controller.signal,
|
|
69
|
+
});
|
|
70
|
+
recordGatewayUsage(bootOpts.storage, model, client, result.usage);
|
|
71
|
+
return json(
|
|
72
|
+
200,
|
|
73
|
+
rerankWire.encodeResponse(
|
|
74
|
+
result,
|
|
75
|
+
parsed.modelId,
|
|
76
|
+
parsed.originalDocuments,
|
|
77
|
+
parsed.request.returnDocuments ?? false,
|
|
78
|
+
),
|
|
79
|
+
gatewayResponseHeaders(model, { requestId, costUsd: result.usage.cost.total, startedAt }),
|
|
80
|
+
);
|
|
81
|
+
} catch (error) {
|
|
82
|
+
if (controller.signal.aborted) return aborted();
|
|
83
|
+
const classified = classifyGatewayError(error);
|
|
84
|
+
logger.warn("auth-gateway rerank failed", { format: "rerank", error: classified.message, peer });
|
|
85
|
+
return rerankWire.formatError(classified.status, classified.type, classified.message);
|
|
86
|
+
}
|
|
87
|
+
}
|