@volter/twin-ai-gateway 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (42) hide show
  1. package/LICENSE +202 -0
  2. package/README.md +176 -0
  3. package/dist/src/ai-gateway-capabilities.d.ts +4 -0
  4. package/dist/src/ai-gateway-capabilities.js +772 -0
  5. package/dist/src/ai-gateway-conformance.d.ts +11 -0
  6. package/dist/src/ai-gateway-conformance.js +72 -0
  7. package/dist/src/ai-gateway-connector.d.ts +50 -0
  8. package/dist/src/ai-gateway-connector.js +97 -0
  9. package/dist/src/ai-gateway-models.d.ts +27 -0
  10. package/dist/src/ai-gateway-models.js +65 -0
  11. package/dist/src/ai-gateway-perform-harness.d.ts +5 -0
  12. package/dist/src/ai-gateway-perform-harness.js +17 -0
  13. package/dist/src/ai-gateway-scenario.d.ts +36 -0
  14. package/dist/src/ai-gateway-scenario.js +125 -0
  15. package/dist/src/ai-gateway-server.d.ts +16 -0
  16. package/dist/src/ai-gateway-server.js +107 -0
  17. package/dist/src/ai-gateway-stub.d.ts +20 -0
  18. package/dist/src/ai-gateway-stub.js +124 -0
  19. package/dist/src/ai-gateway-twin.d.ts +2 -0
  20. package/dist/src/ai-gateway-twin.js +994 -0
  21. package/dist/src/ai-gateway-types.d.ts +94 -0
  22. package/dist/src/ai-gateway-types.js +1 -0
  23. package/dist/src/ai-gateway-v3.d.ts +5 -0
  24. package/dist/src/ai-gateway-v3.js +367 -0
  25. package/dist/src/cli.d.ts +2 -0
  26. package/dist/src/cli.js +24 -0
  27. package/dist/src/index.d.ts +12 -0
  28. package/dist/src/index.js +54 -0
  29. package/package.json +66 -0
  30. package/src/ai-gateway-capabilities.ts +894 -0
  31. package/src/ai-gateway-conformance.ts +77 -0
  32. package/src/ai-gateway-connector.ts +100 -0
  33. package/src/ai-gateway-models.ts +105 -0
  34. package/src/ai-gateway-perform-harness.ts +17 -0
  35. package/src/ai-gateway-scenario.ts +137 -0
  36. package/src/ai-gateway-server.ts +122 -0
  37. package/src/ai-gateway-stub.ts +115 -0
  38. package/src/ai-gateway-twin.ts +1068 -0
  39. package/src/ai-gateway-types.ts +96 -0
  40. package/src/ai-gateway-v3.ts +372 -0
  41. package/src/cli.ts +23 -0
  42. package/src/index.ts +67 -0
@@ -0,0 +1,1068 @@
1
+ // Vercel AI Gateway twin — request handler. Routes the gateway's OpenAI-compatible surface
2
+ // (POST /v1/chat/completions incl. SSE streaming/tools/structured outputs/reasoning/provider
3
+ // routing, POST /v1/embeddings, GET /v1/models[...]) plus the gateway REST surface
4
+ // (GET /v1/credits, GET /v1/generation, GET /v1/report) AND the AI SDK gateway protocol
5
+ // (`@ai-sdk/gateway` / the `ai` package's `createGateway`: GET /v3/ai/config,
6
+ // POST /v3/ai/language-model — see ai-gateway-v3.ts) onto the shared @volter/world-core kernel.
7
+ // Model output is a deterministic, clearly-labeled stub; the protocol envelope — including
8
+ // providerMetadata.gateway routing/cost metadata — is vendor-faithful per Vercel's docs
9
+ // (docs/ai-gateway: "OpenAI Chat Completions API", "Advanced Configuration",
10
+ // "Provider Filtering, Ordering & Sorting", "REST API Reference") and, for /v3/ai, per the
11
+ // shipped @ai-sdk/gateway SDK itself.
12
+ import { applyTwinWrite, projectResources } from '@volter/world-core';
13
+ import { AI_GATEWAY_CATALOG, AI_GATEWAY_MODELS, findAiGatewayModel, type AiGatewayCatalogEntry } from './ai-gateway-models.ts';
14
+ import {
15
+ buildToolCall,
16
+ contentToText,
17
+ countPromptTokens,
18
+ estimateTokens,
19
+ lastUserText,
20
+ normalizeTool,
21
+ stableHash,
22
+ stubAssistantText,
23
+ synthesizeJsonSchemaValue,
24
+ } from './ai-gateway-stub.ts';
25
+ import { type AiGatewayScenarioRespond, realizeAiGatewayRespond, type ScriptedResult } from './ai-gateway-scenario.ts';
26
+ import { handleV3LanguageModel, v3Config } from './ai-gateway-v3.ts';
27
+ import type {
28
+ AiGatewayRequest,
29
+ AiGatewayResponse,
30
+ ChatMessage,
31
+ GatewayRouting,
32
+ GenerationRecord,
33
+ ModelAttempt,
34
+ SseEvent,
35
+ } from './ai-gateway-types.ts';
36
+
37
+ const SERVICE = 'ai-gateway';
38
+ const FIXED_NOW = '1970-01-01T00:00:00.000Z';
39
+ const STARTING_CREDITS = 100;
40
+
41
+ function nowEpoch(occurredAt?: string): number {
42
+ return Math.floor(Date.parse(occurredAt ?? FIXED_NOW) / 1000);
43
+ }
44
+
45
+ function nowIso(occurredAt?: string): string {
46
+ return new Date(Date.parse(occurredAt ?? FIXED_NOW)).toISOString();
47
+ }
48
+
49
+ function parseJson(body?: string): Record<string, unknown> {
50
+ if (!body?.trim()) return {};
51
+ try {
52
+ const parsed = JSON.parse(body);
53
+ return parsed && typeof parsed === 'object' && !Array.isArray(parsed) ? parsed : {};
54
+ } catch {
55
+ return {};
56
+ }
57
+ }
58
+
59
+ // Vendor error envelope: { error: { message, type, param?, code? } } — the documented AI
60
+ // Gateway error response format (OpenAI-compatible docs, "Error handling").
61
+ function error(status: number, message: string, opts: { type?: string; param?: string; code?: string } = {}): AiGatewayResponse {
62
+ return {
63
+ status,
64
+ body: {
65
+ error: {
66
+ message,
67
+ type: opts.type ?? 'invalid_request_error',
68
+ ...(opts.param !== undefined ? { param: opts.param } : {}),
69
+ ...(opts.code !== undefined ? { code: opts.code } : {}),
70
+ },
71
+ },
72
+ };
73
+ }
74
+
75
+ function modelNotFound(model: string): AiGatewayResponse {
76
+ return error(404, `The model '${model}' does not exist or you do not have access to it.`, { code: 'model_not_found', param: 'model' });
77
+ }
78
+
79
+ function readOnlyRejected(): AiGatewayResponse {
80
+ return error(405, 'This AI Gateway twin was started read-only; local writes are disabled.', { code: 'read_only' });
81
+ }
82
+
83
+ function rowsOfType(type: string, root?: string): Record<string, unknown>[] {
84
+ return projectResources(SERVICE, root)
85
+ .filter((r) => r.type === type)
86
+ .map((r) => {
87
+ const { type: _type, updatedAt: _updatedAt, ...rest } = r;
88
+ return rest as Record<string, unknown>;
89
+ });
90
+ }
91
+
92
+ function generationRows(root?: string): GenerationRecord[] {
93
+ return rowsOfType('generation', root) as unknown as GenerationRecord[];
94
+ }
95
+
96
+ // ── gateway routing options (providerOptions.gateway + top-level shorthands) ─────────────
97
+
98
+ type GatewayOptions = {
99
+ order?: string[];
100
+ only?: string[];
101
+ sort?: 'cost' | 'ttft' | 'tps';
102
+ models?: string[];
103
+ byok?: Record<string, unknown>;
104
+ caching?: 'auto';
105
+ };
106
+
107
+ function parseStringArray(raw: unknown, name: string): { value?: string[] } | { error: AiGatewayResponse } {
108
+ if (raw === undefined) return {};
109
+ if (!Array.isArray(raw) || raw.some((s) => typeof s !== 'string' || !s)) {
110
+ return { error: error(400, `${name} must be an array of non-empty strings`, { param: name }) };
111
+ }
112
+ return { value: raw as string[] };
113
+ }
114
+
115
+ const SORT_OPTIONS = new Set(['cost', 'ttft', 'tps']);
116
+
117
+ function parseGatewayOptions(params: Record<string, unknown>): { value: GatewayOptions } | { error: AiGatewayResponse } {
118
+ const out: GatewayOptions = {};
119
+
120
+ const po = params.providerOptions;
121
+ if (po !== undefined && (!po || typeof po !== 'object' || Array.isArray(po))) {
122
+ return { error: error(400, 'providerOptions must be an object', { param: 'providerOptions' }) };
123
+ }
124
+ const gw = po ? (po as Record<string, unknown>).gateway : undefined;
125
+ if (gw !== undefined && (!gw || typeof gw !== 'object' || Array.isArray(gw))) {
126
+ return { error: error(400, 'providerOptions.gateway must be an object', { param: 'providerOptions.gateway' }) };
127
+ }
128
+ const g = (gw ?? {}) as Record<string, unknown>;
129
+
130
+ for (const [key, name] of [['order', 'providerOptions.gateway.order'], ['only', 'providerOptions.gateway.only']] as const) {
131
+ const parsed = parseStringArray(g[key], name);
132
+ if ('error' in parsed) return parsed;
133
+ if (parsed.value) out[key] = parsed.value;
134
+ }
135
+
136
+ if (g.sort !== undefined) {
137
+ if (typeof g.sort !== 'string' || !SORT_OPTIONS.has(g.sort)) {
138
+ return { error: error(400, "providerOptions.gateway.sort must be one of 'cost', 'ttft', 'tps'", { param: 'providerOptions.gateway.sort' }) };
139
+ }
140
+ out.sort = g.sort as GatewayOptions['sort'];
141
+ }
142
+
143
+ // Model fallbacks: top-level `models` (Option 1) or providerOptions.gateway.models (Option 2).
144
+ const topModels = parseStringArray(params.models, 'models');
145
+ if ('error' in topModels) return topModels;
146
+ const gwModels = parseStringArray(g.models, 'providerOptions.gateway.models');
147
+ if ('error' in gwModels) return gwModels;
148
+ if (topModels.value && gwModels.value && JSON.stringify(topModels.value) !== JSON.stringify(gwModels.value)) {
149
+ return { error: error(400, 'models and providerOptions.gateway.models must resolve to the same value') };
150
+ }
151
+ const models = topModels.value ?? gwModels.value;
152
+ if (models) out.models = models;
153
+
154
+ // Top-level `provider` shorthand — documented to support `sort`, equivalent to
155
+ // providerOptions.gateway.sort; conflicting values fail the request.
156
+ if (params.provider !== undefined) {
157
+ const p = params.provider;
158
+ if (!p || typeof p !== 'object' || Array.isArray(p)) {
159
+ return { error: error(400, 'provider must be an object', { param: 'provider' }) };
160
+ }
161
+ const sort = (p as Record<string, unknown>).sort;
162
+ if (sort !== undefined) {
163
+ if (typeof sort !== 'string' || !SORT_OPTIONS.has(sort)) {
164
+ return { error: error(400, "provider.sort must be one of 'cost', 'ttft', 'tps'", { param: 'provider.sort' }) };
165
+ }
166
+ if (out.sort !== undefined && out.sort !== sort) {
167
+ return { error: error(400, 'provider.sort and providerOptions.gateway.sort must resolve to the same value') };
168
+ }
169
+ out.sort = sort as GatewayOptions['sort'];
170
+ }
171
+ }
172
+
173
+ if (g.byok !== undefined) {
174
+ const byok = g.byok;
175
+ if (!byok || typeof byok !== 'object' || Array.isArray(byok)) {
176
+ return { error: error(400, 'providerOptions.gateway.byok must be a record of provider slug to credential arrays', { param: 'providerOptions.gateway.byok' }) };
177
+ }
178
+ for (const [slug, creds] of Object.entries(byok as Record<string, unknown>)) {
179
+ if (!Array.isArray(creds) || creds.length === 0 || creds.some((c) => !c || typeof c !== 'object' || Array.isArray(c))) {
180
+ return { error: error(400, `providerOptions.gateway.byok.${slug} must be a non-empty array of credential objects`, { param: 'providerOptions.gateway.byok' }) };
181
+ }
182
+ }
183
+ out.byok = byok as Record<string, unknown>;
184
+ }
185
+
186
+ if (g.caching !== undefined) {
187
+ if (g.caching !== 'auto') {
188
+ return { error: error(400, "providerOptions.gateway.caching must be 'auto'", { param: 'providerOptions.gateway.caching' }) };
189
+ }
190
+ out.caching = 'auto';
191
+ }
192
+
193
+ return { value: out };
194
+ }
195
+
196
+ // Deterministic per-provider routing metric (twin simulation of the gateway's live cost/latency/
197
+ // throughput telemetry — stable across runs, seeded from the provider slug).
198
+ function providerMetric(provider: string, option: 'cost' | 'ttft' | 'tps'): number {
199
+ const seed = parseInt(stableHash(`${option}:${provider}`).slice(0, 6), 16);
200
+ if (option === 'cost') return Number((0.001 + (seed % 500) / 100000).toFixed(6));
201
+ if (option === 'ttft') return 200 + (seed % 800);
202
+ return 20 + (seed % 120); // tps
203
+ }
204
+
205
+ type ResolvedRouting = { entry: AiGatewayCatalogEntry; routing: GatewayRouting; credentialType: 'system' | 'byok' };
206
+
207
+ function resolveRouting(modelId: string, opts: GatewayOptions, occurredAt?: string): ResolvedRouting | { response: AiGatewayResponse } {
208
+ const startMs = Date.parse(occurredAt ?? FIXED_NOW);
209
+ const modelAttempts: ModelAttempt[] = [];
210
+
211
+ // Model fallback chain: primary first, then the `models` fallbacks in order. In the twin the
212
+ // deterministic failure mode for a model attempt is "not in the catalog".
213
+ const chain = [modelId, ...(opts.models ?? [])];
214
+ let entry: AiGatewayCatalogEntry | undefined;
215
+ let resolvedModelId = modelId;
216
+ for (const candidate of chain) {
217
+ const found = findAiGatewayModel(candidate);
218
+ if (found) {
219
+ entry = found;
220
+ resolvedModelId = candidate;
221
+ break;
222
+ }
223
+ modelAttempts.push({ modelId: candidate, canonicalSlug: candidate, success: false, providerAttemptCount: 0, providerAttempts: [] });
224
+ }
225
+ if (!entry) return { response: modelNotFound(modelId) };
226
+
227
+ let allowed = [...entry.providers];
228
+ if (opts.only) {
229
+ allowed = allowed.filter((p) => opts.only!.includes(p));
230
+ if (allowed.length === 0) {
231
+ return {
232
+ response: error(
233
+ 400,
234
+ `No allowed providers are available for model '${resolvedModelId}'. Allowed providers: ${opts.only.join(', ')}. Providers serving this model: ${entry.providers.join(', ')}.`,
235
+ { code: 'no_available_providers' },
236
+ ),
237
+ };
238
+ }
239
+ }
240
+
241
+ let sortMetrics: Record<string, number | null> | undefined;
242
+ if (opts.sort) {
243
+ const metrics: Record<string, number | null> = {};
244
+ for (const p of allowed) metrics[p] = providerMetric(p, opts.sort);
245
+ const dir = opts.sort === 'tps' ? -1 : 1; // tps: highest first; cost/ttft: lowest first
246
+ allowed = [...allowed].sort((a, b) => dir * ((metrics[a] ?? 0) - (metrics[b] ?? 0)) || a.localeCompare(b));
247
+ sortMetrics = metrics;
248
+ }
249
+ if (opts.order) {
250
+ // Documented combination semantics: `order` providers are promoted to the front, the rest
251
+ // keep the (possibly sorted) order — and `executionOrder` below reports the FINAL attempted
252
+ // order, promotion included.
253
+ const promoted = opts.order.filter((p) => allowed.includes(p));
254
+ allowed = [...promoted, ...allowed.filter((p) => !promoted.includes(p))];
255
+ }
256
+ const sortMeta: GatewayRouting['sort'] = opts.sort && sortMetrics
257
+ ? { option: opts.sort, executionOrder: [...allowed], metrics: sortMetrics, deprioritizedProviders: [] }
258
+ : undefined;
259
+
260
+ const resolvedProvider = allowed[0]!;
261
+ const credentialType: 'system' | 'byok' = opts.byok && Object.prototype.hasOwnProperty.call(opts.byok, resolvedProvider) ? 'byok' : 'system';
262
+ const providerApiModelId = resolvedModelId.slice(resolvedModelId.indexOf('/') + 1);
263
+
264
+ const routing: GatewayRouting = {
265
+ originalModelId: modelId,
266
+ resolvedProvider,
267
+ resolvedProviderApiModelId: providerApiModelId,
268
+ fallbacksAvailable: allowed.slice(1),
269
+ planningReasoning: `${credentialType === 'byok' ? 'BYOK' : 'System'} credentials planned for: ${resolvedProvider}. Total execution order: ${allowed.map((p) => `${p}(${opts.byok && Object.prototype.hasOwnProperty.call(opts.byok, p) ? 'byok' : 'system'})`).join(', ')}`,
270
+ canonicalSlug: resolvedModelId,
271
+ finalProvider: resolvedProvider,
272
+ modelAttemptCount: modelAttempts.length + 1,
273
+ modelAttempts: [
274
+ ...modelAttempts,
275
+ {
276
+ modelId: `${resolvedProvider}:${providerApiModelId}`,
277
+ canonicalSlug: resolvedModelId,
278
+ success: true,
279
+ providerAttemptCount: 1,
280
+ providerAttempts: [{
281
+ provider: resolvedProvider,
282
+ providerApiModelId,
283
+ credentialType,
284
+ success: true,
285
+ startTime: startMs,
286
+ endTime: startMs + 1000,
287
+ }],
288
+ },
289
+ ],
290
+ totalProviderAttemptCount: 1,
291
+ ...(sortMeta ? { sort: sortMeta } : {}),
292
+ };
293
+ return { entry, routing, credentialType };
294
+ }
295
+
296
+ // ── chat completions ──────────────────────────────────────────────────────────────────────
297
+
298
+ type ReasoningConfig = { enabled?: boolean; max_tokens?: number; effort?: string; exclude?: boolean };
299
+
300
+ type ChatArgs = {
301
+ model: string;
302
+ messages: ChatMessage[];
303
+ stream: boolean;
304
+ n: number;
305
+ maxTokens?: number;
306
+ tools?: unknown[];
307
+ toolChoice?: unknown;
308
+ responseFormat?: Record<string, unknown>;
309
+ reasoning?: ReasoningConfig;
310
+ cachedTokens: number;
311
+ gateway: GatewayOptions;
312
+ };
313
+
314
+ const MESSAGE_ROLES = new Set(['system', 'developer', 'user', 'assistant', 'tool']);
315
+ const CONTENT_PART_TYPES = new Set(['text', 'image_url', 'file']);
316
+
317
+ function validateMessages(raw: unknown): { messages: ChatMessage[] } | { response: AiGatewayResponse } {
318
+ if (!Array.isArray(raw) || raw.length === 0) {
319
+ return { response: error(400, 'messages must be a non-empty array', { param: 'messages', code: 'missing_parameter' }) };
320
+ }
321
+ for (const m of raw) {
322
+ if (!m || typeof m !== 'object' || typeof (m as ChatMessage).role !== 'string') {
323
+ return { response: error(400, 'each message must include a role', { param: 'messages' }) };
324
+ }
325
+ const message = m as ChatMessage;
326
+ if (!MESSAGE_ROLES.has(message.role)) {
327
+ return { response: error(400, `unsupported message role: ${message.role}`, { param: 'messages' }) };
328
+ }
329
+ if (message.role === 'tool' && (typeof message.tool_call_id !== 'string' || !message.tool_call_id)) {
330
+ return { response: error(400, "messages with role 'tool' must include tool_call_id", { param: 'messages' }) };
331
+ }
332
+ const hasContent = typeof message.content === 'string'
333
+ ? true
334
+ : Array.isArray(message.content) ? message.content.length > 0 : message.tool_calls !== undefined;
335
+ if (!hasContent) {
336
+ return { response: error(400, 'each message must include content or tool_calls', { param: 'messages' }) };
337
+ }
338
+ if (Array.isArray(message.content)) {
339
+ for (const part of message.content) {
340
+ const type = (part as { type?: unknown })?.type;
341
+ if (typeof type !== 'string' || !CONTENT_PART_TYPES.has(type)) {
342
+ return { response: error(400, `unsupported content part type: ${String(type)} (supported: text, image_url, file)`, { param: 'messages' }) };
343
+ }
344
+ if (type === 'image_url') {
345
+ const url = (part as { image_url?: { url?: unknown } }).image_url?.url;
346
+ if (typeof url !== 'string' || !url) return { response: error(400, 'image_url parts must include image_url.url', { param: 'messages' }) };
347
+ }
348
+ if (type === 'file') {
349
+ const file = (part as { file?: { data?: unknown } }).file;
350
+ if (!file || typeof file !== 'object' || typeof (file as { data?: unknown }).data !== 'string') {
351
+ return { response: error(400, 'file parts must include file.data', { param: 'messages' }) };
352
+ }
353
+ }
354
+ }
355
+ }
356
+ }
357
+ return { messages: raw as ChatMessage[] };
358
+ }
359
+
360
+ function validateSampling(params: Record<string, unknown>): AiGatewayResponse | null {
361
+ const inRange = (v: unknown, min: number, max: number) => typeof v === 'number' && v >= min && v <= max;
362
+ if (params.temperature !== undefined && !inRange(params.temperature, 0, 2)) {
363
+ return error(400, 'temperature must be a number between 0 and 2', { param: 'temperature' });
364
+ }
365
+ if (params.top_p !== undefined && !inRange(params.top_p, 0, 1)) {
366
+ return error(400, 'top_p must be a number between 0 and 1', { param: 'top_p' });
367
+ }
368
+ if (params.frequency_penalty !== undefined && !inRange(params.frequency_penalty, -2, 2)) {
369
+ return error(400, 'frequency_penalty must be a number between -2 and 2', { param: 'frequency_penalty' });
370
+ }
371
+ if (params.presence_penalty !== undefined && !inRange(params.presence_penalty, -2, 2)) {
372
+ return error(400, 'presence_penalty must be a number between -2 and 2', { param: 'presence_penalty' });
373
+ }
374
+ if (params.stop !== undefined && typeof params.stop !== 'string' && !(Array.isArray(params.stop) && params.stop.every((s) => typeof s === 'string'))) {
375
+ return error(400, 'stop must be a string or an array of strings', { param: 'stop' });
376
+ }
377
+ return null;
378
+ }
379
+
380
+ // response_format per the gateway docs: OpenAI json_schema format, legacy `json` format, or text.
381
+ function parseResponseFormat(raw: unknown): { value?: Record<string, unknown> } | { error: AiGatewayResponse } {
382
+ if (raw === undefined) return {};
383
+ if (!raw || typeof raw !== 'object' || Array.isArray(raw)) {
384
+ return { error: error(400, 'response_format must be an object', { param: 'response_format' }) };
385
+ }
386
+ const rf = raw as Record<string, unknown>;
387
+ // The gateway implements the OpenAI Chat Completions spec ('text'/'json_object'/'json_schema')
388
+ // plus its own documented legacy 'json' format.
389
+ if (rf.type !== 'text' && rf.type !== 'json_object' && rf.type !== 'json_schema' && rf.type !== 'json') {
390
+ return { error: error(400, "response_format.type must be one of 'text', 'json_object', 'json_schema', 'json'", { param: 'response_format' }) };
391
+ }
392
+ if (rf.type === 'json_schema') {
393
+ const js = rf.json_schema;
394
+ if (!js || typeof js !== 'object' || typeof (js as Record<string, unknown>).name !== 'string') {
395
+ return { error: error(400, 'response_format.json_schema requires a name', { param: 'response_format' }) };
396
+ }
397
+ }
398
+ return { value: rf };
399
+ }
400
+
401
+ const EFFORTS: Record<string, number> = { none: 0, minimal: 0.1, low: 0.2, medium: 0.5, high: 0.8, xhigh: 0.95 };
402
+
403
+ function parseReasoning(raw: unknown): { value?: ReasoningConfig } | { error: AiGatewayResponse } {
404
+ if (raw === undefined) return {};
405
+ if (!raw || typeof raw !== 'object' || Array.isArray(raw)) {
406
+ return { error: error(400, 'reasoning must be an object', { param: 'reasoning' }) };
407
+ }
408
+ const r = raw as Record<string, unknown>;
409
+ if (r.effort !== undefined && (typeof r.effort !== 'string' || !(r.effort in EFFORTS))) {
410
+ return { error: error(400, "reasoning.effort must be one of 'none', 'minimal', 'low', 'medium', 'high', 'xhigh'", { param: 'reasoning.effort' }) };
411
+ }
412
+ if (r.max_tokens !== undefined && (!Number.isInteger(r.max_tokens) || (r.max_tokens as number) < 0)) {
413
+ return { error: error(400, 'reasoning.max_tokens must be a non-negative integer', { param: 'reasoning.max_tokens' }) };
414
+ }
415
+ // Documented as mutually exclusive ("Advanced Configuration", reasoning parameters).
416
+ if (r.effort !== undefined && r.max_tokens !== undefined) {
417
+ return { error: error(400, 'reasoning.effort and reasoning.max_tokens are mutually exclusive', { param: 'reasoning' }) };
418
+ }
419
+ return {
420
+ value: {
421
+ ...(r.enabled !== undefined ? { enabled: r.enabled === true } : {}),
422
+ ...(r.max_tokens !== undefined ? { max_tokens: Number(r.max_tokens) } : {}),
423
+ ...(r.effort !== undefined ? { effort: String(r.effort) } : {}),
424
+ ...(r.exclude !== undefined ? { exclude: r.exclude === true } : {}),
425
+ },
426
+ };
427
+ }
428
+
429
+ function reasoningTokens(args: ChatArgs): number {
430
+ const r = args.reasoning;
431
+ if (!r) return 0;
432
+ if (r.effort === 'none') return 0;
433
+ if (r.max_tokens !== undefined) return r.max_tokens;
434
+ if (r.effort !== undefined) return Math.round((args.maxTokens ?? 1000) * (EFFORTS[r.effort] ?? 0));
435
+ return r.enabled ? 256 : 0;
436
+ }
437
+
438
+ function reasoningActive(args: ChatArgs): boolean {
439
+ const r = args.reasoning;
440
+ if (!r) return false;
441
+ if (r.effort === 'none') return false;
442
+ return r.enabled === true || r.effort !== undefined || r.max_tokens !== undefined;
443
+ }
444
+
445
+ // cache_control markers (manual prompt caching, Anthropic-style, documented in the gateway's
446
+ // "Advanced Configuration" → Prompt caching): marked prompt tokens report as cached.
447
+ function cachedPromptTokens(messages: ChatMessage[]): number {
448
+ let cached = 0;
449
+ for (const message of messages) {
450
+ const marked = message.cache_control !== undefined ||
451
+ (Array.isArray(message.content) && message.content.some((p) => p && typeof p === 'object' && 'cache_control' in (p as object)));
452
+ if (marked) cached += estimateTokens(`${message.role}: ${contentToText(message.content)}`);
453
+ }
454
+ return cached;
455
+ }
456
+
457
+ function validateChat(params: Record<string, unknown>): { args: ChatArgs } | { response: AiGatewayResponse } {
458
+ if (typeof params.model !== 'string' || !params.model) {
459
+ return { response: error(400, "Invalid request: missing required parameter 'model'", { param: 'model', code: 'missing_parameter' }) };
460
+ }
461
+ const messages = validateMessages(params.messages);
462
+ if ('response' in messages) return messages;
463
+ const sampling = validateSampling(params);
464
+ if (sampling) return { response: sampling };
465
+ const rawMax = params.max_completion_tokens ?? params.max_tokens;
466
+ if (rawMax !== undefined && (!Number.isInteger(rawMax) || (rawMax as number) < 1)) {
467
+ return { response: error(400, 'max_tokens must be an integer >= 1', { param: 'max_tokens' }) };
468
+ }
469
+ if (params.n !== undefined && (!Number.isInteger(params.n) || (params.n as number) < 1)) {
470
+ return { response: error(400, 'n must be an integer >= 1', { param: 'n' }) };
471
+ }
472
+ if (params.tools !== undefined && !Array.isArray(params.tools)) {
473
+ return { response: error(400, 'tools must be an array', { param: 'tools' }) };
474
+ }
475
+ if (params.tools !== undefined) {
476
+ for (const tool of params.tools as unknown[]) {
477
+ if (!normalizeTool(tool)) return { response: error(400, 'each tool must be a function tool with a name', { param: 'tools' }) };
478
+ }
479
+ }
480
+ if (params.tool_choice !== undefined) {
481
+ const tc = params.tool_choice;
482
+ const isString = tc === 'auto' || tc === 'none' || tc === 'required';
483
+ const isFn = !!tc && typeof tc === 'object' && (tc as Record<string, unknown>).type === 'function' &&
484
+ typeof ((tc as Record<string, unknown>).function as Record<string, unknown> | undefined)?.name === 'string';
485
+ if (!isString && !isFn) {
486
+ return { response: error(400, "tool_choice must be 'auto', 'none', 'required', or a named function", { param: 'tool_choice' }) };
487
+ }
488
+ }
489
+ const rf = parseResponseFormat(params.response_format);
490
+ if ('error' in rf) return { response: rf.error };
491
+ const reasoning = parseReasoning(params.reasoning);
492
+ if ('error' in reasoning) return { response: reasoning.error };
493
+ const gateway = parseGatewayOptions(params);
494
+ if ('error' in gateway) return { response: gateway.error };
495
+
496
+ const msgs = messages.messages;
497
+ return {
498
+ args: {
499
+ model: params.model,
500
+ messages: msgs,
501
+ stream: params.stream === true,
502
+ n: params.n === undefined ? 1 : Number(params.n),
503
+ ...(rawMax !== undefined ? { maxTokens: Number(rawMax) } : {}),
504
+ ...(params.tools !== undefined ? { tools: params.tools as unknown[] } : {}),
505
+ ...(params.tool_choice !== undefined ? { toolChoice: params.tool_choice } : {}),
506
+ ...(rf.value ? { responseFormat: rf.value } : {}),
507
+ ...(reasoning.value ? { reasoning: reasoning.value } : {}),
508
+ cachedTokens: Math.min(cachedPromptTokens(msgs), countPromptTokens(msgs)),
509
+ gateway: gateway.value,
510
+ },
511
+ };
512
+ }
513
+
514
+ function chooseTool(args: ChatArgs): { tool: ReturnType<typeof normalizeTool> } | { none: true } | { response: AiGatewayResponse } {
515
+ if (!args.tools || args.tools.length === 0 || args.toolChoice === 'none') return { none: true };
516
+ if (args.toolChoice && typeof args.toolChoice === 'object') {
517
+ const wanted = ((args.toolChoice as Record<string, unknown>).function as Record<string, unknown>).name as string;
518
+ const found = args.tools.map(normalizeTool).find((t) => t?.name === wanted);
519
+ if (!found) return { response: error(400, `tool_choice function '${wanted}' is not present in tools`, { param: 'tool_choice' }) };
520
+ return { tool: found };
521
+ }
522
+ return { tool: normalizeTool(args.tools[0]) };
523
+ }
524
+
525
+ function structuredJsonText(args: ChatArgs): string | null {
526
+ const rf = args.responseFormat;
527
+ if (!rf || rf.type === 'text') return null;
528
+ if (rf.type === 'json_schema') {
529
+ const js = rf.json_schema as Record<string, unknown>;
530
+ return JSON.stringify(synthesizeJsonSchemaValue(js.schema));
531
+ }
532
+ // 'json_object' (OpenAI spec) and the legacy { type: 'json', schema?, name?, description? } format
533
+ if (rf.schema !== undefined) return JSON.stringify(synthesizeJsonSchemaValue(rf.schema));
534
+ return JSON.stringify({ twin_stub: true, echo: lastUserText(args.messages) || 'empty prompt', model: args.model });
535
+ }
536
+
537
+ function reasoningDetails(modelId: string, text: string): Array<Record<string, unknown>> {
538
+ if (modelId.startsWith('anthropic/')) {
539
+ return [{ type: 'reasoning.text', text, signature: `twin-signature-${stableHash(text)}`, format: 'anthropic-claude-v1', index: 0 }];
540
+ }
541
+ if (modelId.startsWith('openai/')) {
542
+ return [
543
+ { type: 'reasoning.summary', summary: text, format: 'openai-responses-v1', index: 0 },
544
+ { type: 'reasoning.encrypted', data: `twin-encrypted-${stableHash(text)}`, format: 'openai-responses-v1', index: 1 },
545
+ ];
546
+ }
547
+ return [{ type: 'reasoning.text', text, format: 'unknown', index: 0 }];
548
+ }
549
+
550
+ type BuiltChoice = { message: Record<string, unknown>; finishReason: string; completionTokens: number; reasoningTokenCount: number };
551
+
552
+ function buildChoiceMessage(args: ChatArgs, resolvedModelId: string, scripted: ScriptedResult | null): BuiltChoice | { response: AiGatewayResponse } {
553
+ const message: Record<string, unknown> = { role: 'assistant' };
554
+ let finishReason = 'stop';
555
+ let completionTokens = 0;
556
+ let reasoningText: string | null = null;
557
+ let reasoningTokenCount = 0;
558
+
559
+ if (scripted) {
560
+ if (scripted.reasoning !== null) reasoningText = scripted.reasoning;
561
+ if (scripted.toolCalls.length > 0) {
562
+ message.content = scripted.text; // null unless the rule also scripts text
563
+ message.tool_calls = scripted.toolCalls.map((tc, i) => ({
564
+ id: tc.id ?? `call_twin_${stableHash(`${tc.name}:${i}`).slice(0, 8)}`,
565
+ type: 'function',
566
+ function: { name: tc.name, arguments: JSON.stringify(tc.arguments) },
567
+ }));
568
+ completionTokens = estimateTokens(JSON.stringify(message.tool_calls));
569
+ } else {
570
+ message.content = scripted.text ?? '';
571
+ completionTokens = estimateTokens(String(message.content));
572
+ }
573
+ finishReason = scripted.finishReason;
574
+ } else {
575
+ const chosen = chooseTool(args);
576
+ if ('response' in chosen) return chosen;
577
+ if ('tool' in chosen && chosen.tool) {
578
+ const call = buildToolCall(chosen.tool);
579
+ message.content = null;
580
+ message.tool_calls = [call];
581
+ finishReason = 'tool_calls';
582
+ completionTokens = estimateTokens(JSON.stringify(call));
583
+ } else {
584
+ let text = structuredJsonText(args) ?? stubAssistantText(args.messages, args.model);
585
+ if (args.maxTokens !== undefined && estimateTokens(text) > args.maxTokens) {
586
+ text = text.slice(0, args.maxTokens * 4);
587
+ finishReason = 'length';
588
+ }
589
+ message.content = text;
590
+ completionTokens = estimateTokens(text);
591
+ }
592
+ if (reasoningActive(args)) {
593
+ reasoningText = `[twin-stub:ai-gateway:reasoning] deterministic reasoning trace for ${args.model}`;
594
+ reasoningTokenCount = reasoningTokens(args);
595
+ }
596
+ }
597
+
598
+ if (scripted?.reasoning != null) reasoningTokenCount = estimateTokens(scripted.reasoning);
599
+ if (reasoningText !== null && args.reasoning?.exclude !== true) {
600
+ message.reasoning = reasoningText;
601
+ message.reasoning_details = reasoningDetails(resolvedModelId, reasoningText);
602
+ }
603
+ return { message, finishReason, completionTokens, reasoningTokenCount };
604
+ }
605
+
606
+ function formatUsd(n: number): string {
607
+ const s = n.toFixed(10).replace(/0+$/, '');
608
+ const [int, dec = ''] = s.split('.');
609
+ return `${int}.${dec.length >= 2 ? dec : (dec + '00').slice(0, 2)}`;
610
+ }
611
+
612
+ function generationCost(entry: AiGatewayCatalogEntry, promptTokens: number, completionTokens: number): number {
613
+ const input = Number(entry.model.pricing.input);
614
+ const output = Number(entry.model.pricing.output ?? '0');
615
+ return Number((promptTokens * input + completionTokens * output).toFixed(10));
616
+ }
617
+
618
+ // Generation ids are UNIQUE per generation (docs: `gen_<ulid>`): the seed includes a per-root
619
+ // monotonic sequence (the count of already-recorded generations), so identical back-to-back
620
+ // requests still mint distinct ids and the kernel's (type, id) resolution can never return a
621
+ // different request's record. Deterministic: same request at the same log position → same id.
622
+ function generationId(seed: unknown, seq: number): string {
623
+ const s = [seed, seq];
624
+ const hex = `${stableHash(s)}${stableHash([s, 'gen'])}${stableHash([s, 'id'])}${stableHash([s, 'x'])}`;
625
+ return `gen_${hex.toUpperCase().slice(0, 26)}`;
626
+ }
627
+
628
+ async function recordGeneration(
629
+ record: GenerationRecord,
630
+ root?: string,
631
+ occurredAt?: string,
632
+ ): Promise<void> {
633
+ await applyTwinWrite(SERVICE, {
634
+ operation: 'generation.record',
635
+ subjectType: 'generation',
636
+ subjectId: record.id,
637
+ fields: record as unknown as Record<string, unknown>,
638
+ occurredAt: occurredAt ?? FIXED_NOW,
639
+ actor: { kind: 'system' },
640
+ }, root);
641
+ }
642
+
643
+ async function chatCompletion(req: AiGatewayRequest, params: Record<string, unknown>): Promise<AiGatewayResponse> {
644
+ const validated = validateChat(params);
645
+ if ('response' in validated) return validated.response;
646
+ const args = validated.args;
647
+
648
+ const resolved = resolveRouting(args.model, args.gateway, req.occurredAt);
649
+ if ('response' in resolved) return resolved.response;
650
+ const { entry, routing, credentialType } = resolved;
651
+
652
+ let missTeach = '';
653
+ const scripted = req.scenarioEngine
654
+ ? (() => {
655
+ const decision = req.scenarioEngine!.next({ model: args.model, messages: args.messages, ...(args.tools ? { tools: args.tools } : {}) });
656
+ if (decision.kind === 'handler') return realizeAiGatewayRespond(decision.respond as AiGatewayScenarioRespond);
657
+ missTeach = `\n[twin-scenario miss — no handler matched. Author one in the world dir's handlers/ai-gateway.json (GET /twin explains; GET /twin/scenario lists handlers + misses). Features seen: ${JSON.stringify(decision.miss.features)}]`;
658
+ return null;
659
+ })()
660
+ : null;
661
+
662
+ const built = buildChoiceMessage(args, entry.model.id, scripted);
663
+ if ('response' in built) return built.response;
664
+ if (missTeach && typeof built.message.content === 'string') {
665
+ (built.message as { content: string }).content += missTeach;
666
+ }
667
+
668
+ const promptTokens = countPromptTokens(args.messages);
669
+ const perChoiceTokens = built.completionTokens + built.reasoningTokenCount;
670
+ const completionTokens = perChoiceTokens * args.n;
671
+ const marketCost = generationCost(entry, promptTokens, completionTokens);
672
+ // BYOK accounting per the docs: the provider bills the tokens, so the gateway's total_cost
673
+ // for the generation is 0.00 and the market price is reported as upstream_inference_cost.
674
+ const isByok = credentialType === 'byok';
675
+ const totalCost = isByok ? 0 : marketCost;
676
+ // ATOMICITY ASSUMPTION: no `await` may sit between this seq read and the recordGeneration()
677
+ // write below — that synchronous prefix is what makes concurrent identical requests mint
678
+ // unique ids (verified: 25 concurrent identical POSTs → 25 unique ids). An added await here
679
+ // would reintroduce the id-collision race.
680
+ const seq = generationRows(req.root).length;
681
+ const id = generationId({ args, promptTokens }, seq);
682
+
683
+ const usage: Record<string, unknown> = {
684
+ prompt_tokens: promptTokens,
685
+ completion_tokens: completionTokens,
686
+ total_tokens: promptTokens + completionTokens,
687
+ ...(args.cachedTokens > 0 ? { prompt_tokens_details: { cached_tokens: args.cachedTokens } } : {}),
688
+ ...(built.reasoningTokenCount > 0 ? { completion_tokens_details: { reasoning_tokens: built.reasoningTokenCount * args.n } } : {}),
689
+ };
690
+
691
+ const body: Record<string, unknown> = {
692
+ id,
693
+ object: 'chat.completion',
694
+ created: nowEpoch(req.occurredAt),
695
+ model: entry.model.id,
696
+ choices: Array.from({ length: args.n }, (_u, index) => ({ index, message: built.message, finish_reason: built.finishReason })),
697
+ usage,
698
+ providerMetadata: {
699
+ gateway: {
700
+ routing,
701
+ cost: formatUsd(totalCost),
702
+ marketCost: formatUsd(marketCost),
703
+ generationId: id,
704
+ },
705
+ },
706
+ };
707
+
708
+ await recordGeneration({
709
+ id,
710
+ total_cost: totalCost,
711
+ upstream_inference_cost: isByok ? marketCost : 0,
712
+ usage: totalCost,
713
+ created_at: nowIso(req.occurredAt),
714
+ model: entry.model.id,
715
+ is_byok: isByok,
716
+ provider_name: routing.finalProvider,
717
+ streamed: args.stream,
718
+ finish_reason: built.finishReason,
719
+ latency: 200,
720
+ generation_time: 1000 + completionTokens * 10,
721
+ tokens_prompt: promptTokens,
722
+ tokens_completion: completionTokens,
723
+ native_tokens_prompt: promptTokens,
724
+ native_tokens_completion: completionTokens,
725
+ native_tokens_reasoning: built.reasoningTokenCount * args.n,
726
+ native_tokens_cached: args.cachedTokens,
727
+ native_tokens_cache_creation: 0,
728
+ billable_web_search_calls: 0,
729
+ }, req.root, req.occurredAt);
730
+
731
+ if (args.stream && req.sseSink) streamChat(body, req.sseSink);
732
+ return { status: 200, body };
733
+ }
734
+
735
+ // SSE streaming: OpenAI chat.completion.chunk envelope; the generation id rides on every chunk
736
+ // starting with the FIRST (the gateway documents injecting the generation id into the first
737
+ // content chunk); reasoning streams via delta.reasoning (+ delta.reasoning_details); ends with
738
+ // a finish_reason + usage chunk and `data: [DONE]`. EVERY choice streams (OpenAI semantics:
739
+ // each chunk carries one element of `choices` with its own index), so streaming with n>1 emits
740
+ // the same n choices the unary body carries and bills — not just index 0. The final chunk (the
741
+ // last choice's finish) carries the n-scaled usage.
742
+ function streamChat(body: Record<string, unknown>, sink: (event: SseEvent) => void): void {
743
+ const base = { id: body.id, object: 'chat.completion.chunk', created: body.created, model: body.model };
744
+ const choices = body.choices as Array<Record<string, unknown>>;
745
+ choices.forEach((choice, choiceIndex) => {
746
+ const message = choice.message as Record<string, unknown>;
747
+ sink({ data: { ...base, choices: [{ index: choiceIndex, delta: { role: 'assistant' }, finish_reason: null }] } });
748
+ if (typeof message.reasoning === 'string') {
749
+ const reasoning = message.reasoning;
750
+ for (let i = 0; i < reasoning.length; i += 48) {
751
+ const slice = reasoning.slice(i, i + 48);
752
+ sink({ data: { ...base, choices: [{ index: choiceIndex, delta: { reasoning: slice, ...(i === 0 && message.reasoning_details ? { reasoning_details: message.reasoning_details } : {}) }, finish_reason: null }] } });
753
+ }
754
+ }
755
+ if (Array.isArray(message.tool_calls)) {
756
+ (message.tool_calls as Array<Record<string, unknown>>).forEach((call, index) => {
757
+ const fn = call.function as Record<string, unknown>;
758
+ sink({ data: { ...base, choices: [{ index: choiceIndex, delta: { tool_calls: [{ index, id: call.id, type: 'function', function: { name: fn.name, arguments: '' } }] }, finish_reason: null }] } });
759
+ sink({ data: { ...base, choices: [{ index: choiceIndex, delta: { tool_calls: [{ index, function: { arguments: fn.arguments } }] }, finish_reason: null }] } });
760
+ });
761
+ } else {
762
+ const text = String(message.content ?? '');
763
+ for (let i = 0; i < text.length; i += 24) {
764
+ sink({ data: { ...base, choices: [{ index: choiceIndex, delta: { content: text.slice(i, i + 24) }, finish_reason: null }] } });
765
+ }
766
+ }
767
+ const isLast = choiceIndex === choices.length - 1;
768
+ sink({ data: { ...base, choices: [{ index: choiceIndex, delta: {}, finish_reason: choice.finish_reason }], ...(isLast ? { usage: body.usage } : {}) } });
769
+ });
770
+ sink({ done: true });
771
+ }
772
+
773
+ // ── embeddings ────────────────────────────────────────────────────────────────────────────
774
+
775
+ // Matches the real default dimension of openai/text-embedding-3-small, so client code that
776
+ // sizes buffers/vector columns off the twin doesn't break against the vendor.
777
+ const DEFAULT_EMBED_DIM = 1536;
778
+
779
+ function embedVector(text: string, dim: number): number[] {
780
+ // Deterministic pseudo-embedding seeded from the text hash; L2-normalized.
781
+ const seed = parseInt(stableHash(text), 16) || 1;
782
+ let x = seed;
783
+ const raw = Array.from({ length: dim }, () => {
784
+ x = (Math.imul(x, 1103515245) + 12345) & 0x7fffffff;
785
+ return (x / 0x7fffffff) * 2 - 1;
786
+ });
787
+ const norm = Math.sqrt(raw.reduce((s, v) => s + v * v, 0)) || 1;
788
+ return raw.map((v) => Number((v / norm).toFixed(6)));
789
+ }
790
+
791
+ async function createEmbeddings(req: AiGatewayRequest, params: Record<string, unknown>): Promise<AiGatewayResponse> {
792
+ if (typeof params.model !== 'string' || !params.model) {
793
+ return error(400, "Invalid request: missing required parameter 'model'", { param: 'model', code: 'missing_parameter' });
794
+ }
795
+ const entry = findAiGatewayModel(params.model);
796
+ if (!entry) return modelNotFound(params.model);
797
+ if (entry.model.type !== 'embedding') {
798
+ return error(400, `The model '${params.model}' is not an embedding model.`, { param: 'model' });
799
+ }
800
+ if (params.input === undefined) return error(400, 'input is required', { param: 'input', code: 'missing_parameter' });
801
+ const inputs = Array.isArray(params.input) ? params.input : [params.input];
802
+ if (inputs.length === 0 || inputs.some((i) => typeof i !== 'string')) {
803
+ return error(400, 'input must be a non-empty string or array of strings', { param: 'input' });
804
+ }
805
+ let dim = DEFAULT_EMBED_DIM;
806
+ if (params.dimensions !== undefined) {
807
+ if (!Number.isInteger(params.dimensions) || (params.dimensions as number) < 1 || (params.dimensions as number) > 4096) {
808
+ return error(400, 'dimensions must be an integer between 1 and 4096', { param: 'dimensions' });
809
+ }
810
+ dim = params.dimensions as number;
811
+ }
812
+ const gateway = parseGatewayOptions(params);
813
+ if ('error' in gateway) return gateway.error;
814
+ const resolved = resolveRouting(entry.model.id, gateway.value, req.occurredAt);
815
+ if ('response' in resolved) return resolved.response;
816
+
817
+ const data = inputs.map((text, index) => ({ object: 'embedding', index, embedding: embedVector(`${text}:${dim}`, dim) }));
818
+ const promptTokens = inputs.reduce((s, t) => s + estimateTokens(t as string), 0);
819
+ const marketCost = Number((promptTokens * Number(entry.model.pricing.input)).toFixed(10));
820
+ // The real gateway BILLS embeddings — so the twin records a generation for every embeddings
821
+ // call through the same kernel fold as chat: credits (/v1/credits), the spend report
822
+ // (/v1/report), and generation lookup (/v1/generation?id) all move. Same BYOK rule as chat:
823
+ // BYOK requests record total_cost 0 with the market price as upstream_inference_cost.
824
+ const isByok = resolved.credentialType === 'byok';
825
+ const totalCost = isByok ? 0 : marketCost;
826
+ // Same ATOMICITY rule as chatCompletion: no `await` between this seq read and the
827
+ // recordGeneration() write, so concurrent identical requests still mint unique ids.
828
+ const seq = generationRows(req.root).length;
829
+ const id = generationId({ embeddings: { model: entry.model.id, inputs, dim }, promptTokens }, seq);
830
+
831
+ await recordGeneration({
832
+ id,
833
+ total_cost: totalCost,
834
+ upstream_inference_cost: isByok ? marketCost : 0,
835
+ usage: totalCost,
836
+ created_at: nowIso(req.occurredAt),
837
+ model: entry.model.id,
838
+ is_byok: isByok,
839
+ provider_name: resolved.routing.finalProvider,
840
+ streamed: false,
841
+ finish_reason: 'stop',
842
+ latency: 200,
843
+ generation_time: 100 + promptTokens,
844
+ tokens_prompt: promptTokens,
845
+ tokens_completion: 0,
846
+ native_tokens_prompt: promptTokens,
847
+ native_tokens_completion: 0,
848
+ native_tokens_reasoning: 0,
849
+ native_tokens_cached: 0,
850
+ native_tokens_cache_creation: 0,
851
+ billable_web_search_calls: 0,
852
+ }, req.root, req.occurredAt);
853
+
854
+ return {
855
+ status: 200,
856
+ body: {
857
+ object: 'list',
858
+ data,
859
+ model: entry.model.id,
860
+ usage: { prompt_tokens: promptTokens, total_tokens: promptTokens },
861
+ providerMetadata: { gateway: { routing: resolved.routing, cost: formatUsd(totalCost), marketCost: formatUsd(marketCost), generationId: id } },
862
+ },
863
+ };
864
+ }
865
+
866
+ // ── models / credits / generation / report ────────────────────────────────────────────────
867
+
868
+ function modelEndpoints(entry: AiGatewayCatalogEntry): AiGatewayResponse {
869
+ const m = entry.model;
870
+ return {
871
+ status: 200,
872
+ body: {
873
+ data: {
874
+ id: m.id,
875
+ name: m.name,
876
+ created: m.created,
877
+ released: m.released,
878
+ description: m.description,
879
+ architecture: {
880
+ tokenizer: null,
881
+ instruct_type: null,
882
+ modality: m.tags.includes('vision') ? 'text+image→text' : 'text→text',
883
+ input_modalities: m.tags.includes('vision') ? ['text', 'image'] : ['text'],
884
+ output_modalities: ['text'],
885
+ },
886
+ endpoints: entry.providers.map((provider) => ({
887
+ name: `${provider} | ${m.id}`,
888
+ model_name: m.name,
889
+ context_length: m.context_window,
890
+ pricing: {
891
+ prompt: m.pricing.input,
892
+ completion: m.pricing.output ?? '0',
893
+ ...(m.pricing.input_cache_read ? { input_cache_read: m.pricing.input_cache_read } : {}),
894
+ ...(m.pricing.input_cache_write ? { input_cache_write: m.pricing.input_cache_write } : {}),
895
+ },
896
+ provider_name: provider,
897
+ max_completion_tokens: m.max_tokens,
898
+ supported_parameters: ['max_tokens', 'temperature', 'tools', 'reasoning'],
899
+ status: 0,
900
+ uptime_last_15m: 100,
901
+ uptime_last_1h: 100,
902
+ uptime_last_1d: 100,
903
+ throughput_last_1h: { p50: providerMetric(provider, 'tps'), p95: providerMetric(provider, 'tps') + 5 },
904
+ latency_last_1h: { p50: providerMetric(provider, 'ttft'), p95: providerMetric(provider, 'ttft') + 200 },
905
+ supports_implicit_caching: false,
906
+ })),
907
+ },
908
+ },
909
+ };
910
+ }
911
+
912
+ function credits(root?: string): AiGatewayResponse {
913
+ const used = generationRows(root).reduce((sum, g) => sum + g.total_cost, 0);
914
+ return {
915
+ status: 200,
916
+ body: { balance: formatUsd(STARTING_CREDITS - used), total_used: formatUsd(used) },
917
+ };
918
+ }
919
+
920
+ function generationLookup(req: AiGatewayRequest, url: URL): AiGatewayResponse {
921
+ const id = url.searchParams.get('id');
922
+ if (!id) return error(400, "Invalid request: missing required parameter 'id'", { param: 'id', code: 'missing_parameter' });
923
+ const found = generationRows(req.root).find((g) => g.id === id);
924
+ if (!found) return error(404, `Generation '${id}' not found.`, { code: 'not_found' });
925
+ return { status: 200, body: { data: found } };
926
+ }
927
+
928
+ // Spend report (GET /v1/report), per the documented Custom Reporting response format: a
929
+ // `results` array whose rows carry the grouping field (day/model/provider/credential_type/…)
930
+ // plus the metric fields (total_cost/market_cost/surcharge_cost/gateway_cost, token counts,
931
+ // request_count). The twin records no per-request user/tag/api-key attribution yet (see the
932
+ // `ai-gateway.chat.reporting_tags` / `ai-gateway.report.attribution_groupings` todos), so
933
+ // attribution-keyed groupings and filters resolve to empty result sets rather than fabricating
934
+ // attribution that was never recorded.
935
+ const REPORT_GROUP_BY = new Set(['day', 'user', 'model', 'tag', 'provider', 'credential_type', 'zero_data_retention', 'api_key_name']);
936
+ const DATE_RE = /^\d{4}-\d{2}-\d{2}$/;
937
+
938
+ function spendReport(req: AiGatewayRequest, url: URL): AiGatewayResponse {
939
+ const q = url.searchParams;
940
+ const start = q.get('start_date');
941
+ const end = q.get('end_date');
942
+ if (!start || !DATE_RE.test(start)) return error(400, 'start_date is required in YYYY-MM-DD format', { param: 'start_date' });
943
+ if (!end || !DATE_RE.test(end)) return error(400, 'end_date is required in YYYY-MM-DD format', { param: 'end_date' });
944
+ const groupBy = q.get('group_by') ?? 'day';
945
+ if (!REPORT_GROUP_BY.has(groupBy)) {
946
+ return error(400, `group_by must be one of ${[...REPORT_GROUP_BY].join(', ')}`, { param: 'group_by' });
947
+ }
948
+ const datePart = q.get('date_part') ?? 'day';
949
+ if (datePart !== 'day' && datePart !== 'hour') {
950
+ return error(400, "date_part must be 'day' or 'hour'", { param: 'date_part' });
951
+ }
952
+ const credentialType = q.get('credential_type');
953
+ if (credentialType !== null && credentialType !== 'byok' && credentialType !== 'system') {
954
+ return error(400, "credential_type must be 'byok' or 'system'", { param: 'credential_type' });
955
+ }
956
+ const zdr = q.get('zero_data_retention');
957
+ if (zdr !== null && zdr !== 'true' && zdr !== 'false') {
958
+ return error(400, 'zero_data_retention must be true or false', { param: 'zero_data_retention' });
959
+ }
960
+ const tagsMatch = q.get('tags_match') ?? 'any';
961
+ if (tagsMatch !== 'any' && tagsMatch !== 'all') {
962
+ return error(400, "tags_match must be 'any' or 'all'", { param: 'tags_match' });
963
+ }
964
+
965
+ let rows = generationRows(req.root).filter((g) => {
966
+ const day = g.created_at.slice(0, 10);
967
+ return day >= start && day <= end;
968
+ });
969
+ const model = q.get('model');
970
+ if (model) rows = rows.filter((g) => g.model === model);
971
+ const provider = q.get('provider');
972
+ if (provider) rows = rows.filter((g) => g.provider_name === provider);
973
+ if (credentialType) rows = rows.filter((g) => (g.is_byok ? 'byok' : 'system') === credentialType);
974
+ // No generation ever requests ZDR / carries user or tag attribution in the twin yet:
975
+ if (zdr === 'true') rows = [];
976
+ if (q.get('user_id') || q.get('tags')) rows = [];
977
+
978
+ const keyOf = (g: GenerationRecord): string | null => {
979
+ // date_part=hour buckets are `YYYY-MM-DDTHH` under the `hour` grouping field (docs).
980
+ if (groupBy === 'day') return datePart === 'hour' ? g.created_at.slice(0, 13) : g.created_at.slice(0, 10);
981
+ if (groupBy === 'model') return g.model;
982
+ if (groupBy === 'provider') return g.provider_name;
983
+ if (groupBy === 'credential_type') return g.is_byok ? 'byok' : 'system';
984
+ if (groupBy === 'zero_data_retention') return 'false'; // no ZDR requests recorded
985
+ return null; // user/tag/api_key_name: no attribution recorded
986
+ };
987
+ type Row = {
988
+ total_cost: number; market_cost: number; surcharge_cost: number; gateway_cost: number;
989
+ input_tokens: number; output_tokens: number; cached_input_tokens: number;
990
+ cache_creation_input_tokens: number; reasoning_tokens: number; request_count: number;
991
+ };
992
+ const groups = new Map<string, Row>();
993
+ for (const g of rows) {
994
+ const key = keyOf(g);
995
+ if (key === null) continue;
996
+ const cur = groups.get(key) ?? {
997
+ total_cost: 0, market_cost: 0, surcharge_cost: 0, gateway_cost: 0,
998
+ input_tokens: 0, output_tokens: 0, cached_input_tokens: 0,
999
+ cache_creation_input_tokens: 0, reasoning_tokens: 0, request_count: 0,
1000
+ };
1001
+ cur.total_cost = Number((cur.total_cost + g.total_cost).toFixed(10));
1002
+ // market_cost includes both BYOK and non-BYOK cost; gateway_cost is the gateway's OWN cost
1003
+ // separate from the provider rate — 0 here ("no markup on tokens", matching the docs sample).
1004
+ cur.market_cost = Number((cur.market_cost + g.total_cost + g.upstream_inference_cost).toFixed(10));
1005
+ cur.input_tokens += g.tokens_prompt;
1006
+ cur.output_tokens += g.tokens_completion;
1007
+ cur.cached_input_tokens += g.native_tokens_cached;
1008
+ cur.cache_creation_input_tokens += g.native_tokens_cache_creation;
1009
+ cur.reasoning_tokens += g.native_tokens_reasoning;
1010
+ cur.request_count += 1;
1011
+ groups.set(key, cur);
1012
+ }
1013
+ const field = groupBy === 'day' ? (datePart === 'hour' ? 'hour' : 'day') : groupBy;
1014
+ const results: Array<Record<string, unknown>> = [...groups.entries()]
1015
+ .map(([key, value]): Record<string, unknown> => ({ [field]: key, ...value }))
1016
+ .sort((a, b) => (String(a[field]) < String(b[field]) ? -1 : 1));
1017
+ return { status: 200, body: { results } };
1018
+ }
1019
+
1020
+ // ── router ────────────────────────────────────────────────────────────────────────────────
1021
+
1022
+ export async function handleAiGatewayTwinRequest(req: AiGatewayRequest): Promise<AiGatewayResponse> {
1023
+ const method = req.method.toUpperCase();
1024
+ const url = new URL(req.path, 'http://twin.local');
1025
+ const path = url.pathname;
1026
+
1027
+ if (method === 'GET' && path === '/v1/models') {
1028
+ return { status: 200, body: { object: 'list', data: AI_GATEWAY_MODELS } };
1029
+ }
1030
+ if (method === 'GET' && path.startsWith('/v1/models/') && path.endsWith('/endpoints')) {
1031
+ const id = decodeURIComponent(path.slice('/v1/models/'.length, -'/endpoints'.length));
1032
+ const entry = findAiGatewayModel(id);
1033
+ return entry ? modelEndpoints(entry) : modelNotFound(id);
1034
+ }
1035
+ if (method === 'GET' && path.startsWith('/v1/models/')) {
1036
+ const id = decodeURIComponent(path.slice('/v1/models/'.length));
1037
+ const entry = findAiGatewayModel(id);
1038
+ return entry ? { status: 200, body: entry.model } : modelNotFound(id);
1039
+ }
1040
+ if (method === 'GET' && path === '/v1/credits') return credits(req.root);
1041
+ if (method === 'GET' && path === '/v1/generation') return generationLookup(req, url);
1042
+ if (method === 'GET' && path === '/v1/report') return spendReport(req, url);
1043
+
1044
+ if (method === 'POST' && path === '/v1/chat/completions') {
1045
+ if (req.readOnly) return readOnlyRejected();
1046
+ return chatCompletion(req, parseJson(req.body));
1047
+ }
1048
+ if (method === 'POST' && path === '/v1/embeddings') {
1049
+ if (req.readOnly) return readOnlyRejected();
1050
+ return createEmbeddings(req, parseJson(req.body));
1051
+ }
1052
+
1053
+ // ── AI SDK gateway protocol (/v3/ai) — what `createGateway` (the `ai` package /
1054
+ // @ai-sdk/gateway) actually speaks; see ai-gateway-v3.ts. Language-model calls run through
1055
+ // the SAME chatCompletion fold as /v1, so accounting is one surface across both protocols.
1056
+ if (method === 'GET' && path === '/v3/ai/config') return v3Config();
1057
+ if (method === 'POST' && path === '/v3/ai/language-model') {
1058
+ if (req.readOnly) return readOnlyRejected();
1059
+ return handleV3LanguageModel(req, chatCompletion);
1060
+ }
1061
+ if (path.startsWith('/v3/ai/')) {
1062
+ // embedding-model / image-model / video-model are real /v3/ai sub-surfaces the twin does
1063
+ // not model yet — honest 404s (open todos in the capability manifest), never fake success.
1064
+ return error(404, `AI SDK gateway protocol operation not modeled: ${method} ${path} (modeled: GET /v3/ai/config, POST /v3/ai/language-model)`, { code: 'not_found' });
1065
+ }
1066
+
1067
+ return error(404, `AI Gateway operation not modeled: ${method} ${path}`, { code: 'not_found' });
1068
+ }