@volter/twin-ai-gateway 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +202 -0
- package/README.md +176 -0
- package/dist/src/ai-gateway-capabilities.d.ts +4 -0
- package/dist/src/ai-gateway-capabilities.js +772 -0
- package/dist/src/ai-gateway-conformance.d.ts +11 -0
- package/dist/src/ai-gateway-conformance.js +72 -0
- package/dist/src/ai-gateway-connector.d.ts +50 -0
- package/dist/src/ai-gateway-connector.js +97 -0
- package/dist/src/ai-gateway-models.d.ts +27 -0
- package/dist/src/ai-gateway-models.js +65 -0
- package/dist/src/ai-gateway-perform-harness.d.ts +5 -0
- package/dist/src/ai-gateway-perform-harness.js +17 -0
- package/dist/src/ai-gateway-scenario.d.ts +36 -0
- package/dist/src/ai-gateway-scenario.js +125 -0
- package/dist/src/ai-gateway-server.d.ts +16 -0
- package/dist/src/ai-gateway-server.js +107 -0
- package/dist/src/ai-gateway-stub.d.ts +20 -0
- package/dist/src/ai-gateway-stub.js +124 -0
- package/dist/src/ai-gateway-twin.d.ts +2 -0
- package/dist/src/ai-gateway-twin.js +994 -0
- package/dist/src/ai-gateway-types.d.ts +94 -0
- package/dist/src/ai-gateway-types.js +1 -0
- package/dist/src/ai-gateway-v3.d.ts +5 -0
- package/dist/src/ai-gateway-v3.js +367 -0
- package/dist/src/cli.d.ts +2 -0
- package/dist/src/cli.js +24 -0
- package/dist/src/index.d.ts +12 -0
- package/dist/src/index.js +54 -0
- package/package.json +66 -0
- package/src/ai-gateway-capabilities.ts +894 -0
- package/src/ai-gateway-conformance.ts +77 -0
- package/src/ai-gateway-connector.ts +100 -0
- package/src/ai-gateway-models.ts +105 -0
- package/src/ai-gateway-perform-harness.ts +17 -0
- package/src/ai-gateway-scenario.ts +137 -0
- package/src/ai-gateway-server.ts +122 -0
- package/src/ai-gateway-stub.ts +115 -0
- package/src/ai-gateway-twin.ts +1068 -0
- package/src/ai-gateway-types.ts +96 -0
- package/src/ai-gateway-v3.ts +372 -0
- package/src/cli.ts +23 -0
- package/src/index.ts +67 -0
|
@@ -0,0 +1,994 @@
|
|
|
1
|
+
// Vercel AI Gateway twin — request handler. Routes the gateway's OpenAI-compatible surface
|
|
2
|
+
// (POST /v1/chat/completions incl. SSE streaming/tools/structured outputs/reasoning/provider
|
|
3
|
+
// routing, POST /v1/embeddings, GET /v1/models[...]) plus the gateway REST surface
|
|
4
|
+
// (GET /v1/credits, GET /v1/generation, GET /v1/report) AND the AI SDK gateway protocol
|
|
5
|
+
// (`@ai-sdk/gateway` / the `ai` package's `createGateway`: GET /v3/ai/config,
|
|
6
|
+
// POST /v3/ai/language-model — see ai-gateway-v3.ts) onto the shared @volter/world-core kernel.
|
|
7
|
+
// Model output is a deterministic, clearly-labeled stub; the protocol envelope — including
|
|
8
|
+
// providerMetadata.gateway routing/cost metadata — is vendor-faithful per Vercel's docs
|
|
9
|
+
// (docs/ai-gateway: "OpenAI Chat Completions API", "Advanced Configuration",
|
|
10
|
+
// "Provider Filtering, Ordering & Sorting", "REST API Reference") and, for /v3/ai, per the
|
|
11
|
+
// shipped @ai-sdk/gateway SDK itself.
|
|
12
|
+
import { applyTwinWrite, projectResources } from '@volter/world-core';
|
|
13
|
+
import { AI_GATEWAY_MODELS, findAiGatewayModel } from "./ai-gateway-models.js";
|
|
14
|
+
import { buildToolCall, contentToText, countPromptTokens, estimateTokens, lastUserText, normalizeTool, stableHash, stubAssistantText, synthesizeJsonSchemaValue, } from "./ai-gateway-stub.js";
|
|
15
|
+
import { realizeAiGatewayRespond } from "./ai-gateway-scenario.js";
|
|
16
|
+
import { handleV3LanguageModel, v3Config } from "./ai-gateway-v3.js";
|
|
17
|
+
const SERVICE = 'ai-gateway';
|
|
18
|
+
const FIXED_NOW = '1970-01-01T00:00:00.000Z';
|
|
19
|
+
const STARTING_CREDITS = 100;
|
|
20
|
+
function nowEpoch(occurredAt) {
|
|
21
|
+
return Math.floor(Date.parse(occurredAt ?? FIXED_NOW) / 1000);
|
|
22
|
+
}
|
|
23
|
+
function nowIso(occurredAt) {
|
|
24
|
+
return new Date(Date.parse(occurredAt ?? FIXED_NOW)).toISOString();
|
|
25
|
+
}
|
|
26
|
+
function parseJson(body) {
|
|
27
|
+
if (!body?.trim())
|
|
28
|
+
return {};
|
|
29
|
+
try {
|
|
30
|
+
const parsed = JSON.parse(body);
|
|
31
|
+
return parsed && typeof parsed === 'object' && !Array.isArray(parsed) ? parsed : {};
|
|
32
|
+
}
|
|
33
|
+
catch {
|
|
34
|
+
return {};
|
|
35
|
+
}
|
|
36
|
+
}
|
|
37
|
+
// Vendor error envelope: { error: { message, type, param?, code? } } — the documented AI
|
|
38
|
+
// Gateway error response format (OpenAI-compatible docs, "Error handling").
|
|
39
|
+
function error(status, message, opts = {}) {
|
|
40
|
+
return {
|
|
41
|
+
status,
|
|
42
|
+
body: {
|
|
43
|
+
error: {
|
|
44
|
+
message,
|
|
45
|
+
type: opts.type ?? 'invalid_request_error',
|
|
46
|
+
...(opts.param !== undefined ? { param: opts.param } : {}),
|
|
47
|
+
...(opts.code !== undefined ? { code: opts.code } : {}),
|
|
48
|
+
},
|
|
49
|
+
},
|
|
50
|
+
};
|
|
51
|
+
}
|
|
52
|
+
function modelNotFound(model) {
|
|
53
|
+
return error(404, `The model '${model}' does not exist or you do not have access to it.`, { code: 'model_not_found', param: 'model' });
|
|
54
|
+
}
|
|
55
|
+
function readOnlyRejected() {
|
|
56
|
+
return error(405, 'This AI Gateway twin was started read-only; local writes are disabled.', { code: 'read_only' });
|
|
57
|
+
}
|
|
58
|
+
function rowsOfType(type, root) {
|
|
59
|
+
return projectResources(SERVICE, root)
|
|
60
|
+
.filter((r) => r.type === type)
|
|
61
|
+
.map((r) => {
|
|
62
|
+
const { type: _type, updatedAt: _updatedAt, ...rest } = r;
|
|
63
|
+
return rest;
|
|
64
|
+
});
|
|
65
|
+
}
|
|
66
|
+
function generationRows(root) {
|
|
67
|
+
return rowsOfType('generation', root);
|
|
68
|
+
}
|
|
69
|
+
function parseStringArray(raw, name) {
|
|
70
|
+
if (raw === undefined)
|
|
71
|
+
return {};
|
|
72
|
+
if (!Array.isArray(raw) || raw.some((s) => typeof s !== 'string' || !s)) {
|
|
73
|
+
return { error: error(400, `${name} must be an array of non-empty strings`, { param: name }) };
|
|
74
|
+
}
|
|
75
|
+
return { value: raw };
|
|
76
|
+
}
|
|
77
|
+
const SORT_OPTIONS = new Set(['cost', 'ttft', 'tps']);
|
|
78
|
+
function parseGatewayOptions(params) {
|
|
79
|
+
const out = {};
|
|
80
|
+
const po = params.providerOptions;
|
|
81
|
+
if (po !== undefined && (!po || typeof po !== 'object' || Array.isArray(po))) {
|
|
82
|
+
return { error: error(400, 'providerOptions must be an object', { param: 'providerOptions' }) };
|
|
83
|
+
}
|
|
84
|
+
const gw = po ? po.gateway : undefined;
|
|
85
|
+
if (gw !== undefined && (!gw || typeof gw !== 'object' || Array.isArray(gw))) {
|
|
86
|
+
return { error: error(400, 'providerOptions.gateway must be an object', { param: 'providerOptions.gateway' }) };
|
|
87
|
+
}
|
|
88
|
+
const g = (gw ?? {});
|
|
89
|
+
for (const [key, name] of [['order', 'providerOptions.gateway.order'], ['only', 'providerOptions.gateway.only']]) {
|
|
90
|
+
const parsed = parseStringArray(g[key], name);
|
|
91
|
+
if ('error' in parsed)
|
|
92
|
+
return parsed;
|
|
93
|
+
if (parsed.value)
|
|
94
|
+
out[key] = parsed.value;
|
|
95
|
+
}
|
|
96
|
+
if (g.sort !== undefined) {
|
|
97
|
+
if (typeof g.sort !== 'string' || !SORT_OPTIONS.has(g.sort)) {
|
|
98
|
+
return { error: error(400, "providerOptions.gateway.sort must be one of 'cost', 'ttft', 'tps'", { param: 'providerOptions.gateway.sort' }) };
|
|
99
|
+
}
|
|
100
|
+
out.sort = g.sort;
|
|
101
|
+
}
|
|
102
|
+
// Model fallbacks: top-level `models` (Option 1) or providerOptions.gateway.models (Option 2).
|
|
103
|
+
const topModels = parseStringArray(params.models, 'models');
|
|
104
|
+
if ('error' in topModels)
|
|
105
|
+
return topModels;
|
|
106
|
+
const gwModels = parseStringArray(g.models, 'providerOptions.gateway.models');
|
|
107
|
+
if ('error' in gwModels)
|
|
108
|
+
return gwModels;
|
|
109
|
+
if (topModels.value && gwModels.value && JSON.stringify(topModels.value) !== JSON.stringify(gwModels.value)) {
|
|
110
|
+
return { error: error(400, 'models and providerOptions.gateway.models must resolve to the same value') };
|
|
111
|
+
}
|
|
112
|
+
const models = topModels.value ?? gwModels.value;
|
|
113
|
+
if (models)
|
|
114
|
+
out.models = models;
|
|
115
|
+
// Top-level `provider` shorthand — documented to support `sort`, equivalent to
|
|
116
|
+
// providerOptions.gateway.sort; conflicting values fail the request.
|
|
117
|
+
if (params.provider !== undefined) {
|
|
118
|
+
const p = params.provider;
|
|
119
|
+
if (!p || typeof p !== 'object' || Array.isArray(p)) {
|
|
120
|
+
return { error: error(400, 'provider must be an object', { param: 'provider' }) };
|
|
121
|
+
}
|
|
122
|
+
const sort = p.sort;
|
|
123
|
+
if (sort !== undefined) {
|
|
124
|
+
if (typeof sort !== 'string' || !SORT_OPTIONS.has(sort)) {
|
|
125
|
+
return { error: error(400, "provider.sort must be one of 'cost', 'ttft', 'tps'", { param: 'provider.sort' }) };
|
|
126
|
+
}
|
|
127
|
+
if (out.sort !== undefined && out.sort !== sort) {
|
|
128
|
+
return { error: error(400, 'provider.sort and providerOptions.gateway.sort must resolve to the same value') };
|
|
129
|
+
}
|
|
130
|
+
out.sort = sort;
|
|
131
|
+
}
|
|
132
|
+
}
|
|
133
|
+
if (g.byok !== undefined) {
|
|
134
|
+
const byok = g.byok;
|
|
135
|
+
if (!byok || typeof byok !== 'object' || Array.isArray(byok)) {
|
|
136
|
+
return { error: error(400, 'providerOptions.gateway.byok must be a record of provider slug to credential arrays', { param: 'providerOptions.gateway.byok' }) };
|
|
137
|
+
}
|
|
138
|
+
for (const [slug, creds] of Object.entries(byok)) {
|
|
139
|
+
if (!Array.isArray(creds) || creds.length === 0 || creds.some((c) => !c || typeof c !== 'object' || Array.isArray(c))) {
|
|
140
|
+
return { error: error(400, `providerOptions.gateway.byok.${slug} must be a non-empty array of credential objects`, { param: 'providerOptions.gateway.byok' }) };
|
|
141
|
+
}
|
|
142
|
+
}
|
|
143
|
+
out.byok = byok;
|
|
144
|
+
}
|
|
145
|
+
if (g.caching !== undefined) {
|
|
146
|
+
if (g.caching !== 'auto') {
|
|
147
|
+
return { error: error(400, "providerOptions.gateway.caching must be 'auto'", { param: 'providerOptions.gateway.caching' }) };
|
|
148
|
+
}
|
|
149
|
+
out.caching = 'auto';
|
|
150
|
+
}
|
|
151
|
+
return { value: out };
|
|
152
|
+
}
|
|
153
|
+
// Deterministic per-provider routing metric (twin simulation of the gateway's live cost/latency/
|
|
154
|
+
// throughput telemetry — stable across runs, seeded from the provider slug).
|
|
155
|
+
function providerMetric(provider, option) {
|
|
156
|
+
const seed = parseInt(stableHash(`${option}:${provider}`).slice(0, 6), 16);
|
|
157
|
+
if (option === 'cost')
|
|
158
|
+
return Number((0.001 + (seed % 500) / 100000).toFixed(6));
|
|
159
|
+
if (option === 'ttft')
|
|
160
|
+
return 200 + (seed % 800);
|
|
161
|
+
return 20 + (seed % 120); // tps
|
|
162
|
+
}
|
|
163
|
+
function resolveRouting(modelId, opts, occurredAt) {
|
|
164
|
+
const startMs = Date.parse(occurredAt ?? FIXED_NOW);
|
|
165
|
+
const modelAttempts = [];
|
|
166
|
+
// Model fallback chain: primary first, then the `models` fallbacks in order. In the twin the
|
|
167
|
+
// deterministic failure mode for a model attempt is "not in the catalog".
|
|
168
|
+
const chain = [modelId, ...(opts.models ?? [])];
|
|
169
|
+
let entry;
|
|
170
|
+
let resolvedModelId = modelId;
|
|
171
|
+
for (const candidate of chain) {
|
|
172
|
+
const found = findAiGatewayModel(candidate);
|
|
173
|
+
if (found) {
|
|
174
|
+
entry = found;
|
|
175
|
+
resolvedModelId = candidate;
|
|
176
|
+
break;
|
|
177
|
+
}
|
|
178
|
+
modelAttempts.push({ modelId: candidate, canonicalSlug: candidate, success: false, providerAttemptCount: 0, providerAttempts: [] });
|
|
179
|
+
}
|
|
180
|
+
if (!entry)
|
|
181
|
+
return { response: modelNotFound(modelId) };
|
|
182
|
+
let allowed = [...entry.providers];
|
|
183
|
+
if (opts.only) {
|
|
184
|
+
allowed = allowed.filter((p) => opts.only.includes(p));
|
|
185
|
+
if (allowed.length === 0) {
|
|
186
|
+
return {
|
|
187
|
+
response: error(400, `No allowed providers are available for model '${resolvedModelId}'. Allowed providers: ${opts.only.join(', ')}. Providers serving this model: ${entry.providers.join(', ')}.`, { code: 'no_available_providers' }),
|
|
188
|
+
};
|
|
189
|
+
}
|
|
190
|
+
}
|
|
191
|
+
let sortMetrics;
|
|
192
|
+
if (opts.sort) {
|
|
193
|
+
const metrics = {};
|
|
194
|
+
for (const p of allowed)
|
|
195
|
+
metrics[p] = providerMetric(p, opts.sort);
|
|
196
|
+
const dir = opts.sort === 'tps' ? -1 : 1; // tps: highest first; cost/ttft: lowest first
|
|
197
|
+
allowed = [...allowed].sort((a, b) => dir * ((metrics[a] ?? 0) - (metrics[b] ?? 0)) || a.localeCompare(b));
|
|
198
|
+
sortMetrics = metrics;
|
|
199
|
+
}
|
|
200
|
+
if (opts.order) {
|
|
201
|
+
// Documented combination semantics: `order` providers are promoted to the front, the rest
|
|
202
|
+
// keep the (possibly sorted) order — and `executionOrder` below reports the FINAL attempted
|
|
203
|
+
// order, promotion included.
|
|
204
|
+
const promoted = opts.order.filter((p) => allowed.includes(p));
|
|
205
|
+
allowed = [...promoted, ...allowed.filter((p) => !promoted.includes(p))];
|
|
206
|
+
}
|
|
207
|
+
const sortMeta = opts.sort && sortMetrics
|
|
208
|
+
? { option: opts.sort, executionOrder: [...allowed], metrics: sortMetrics, deprioritizedProviders: [] }
|
|
209
|
+
: undefined;
|
|
210
|
+
const resolvedProvider = allowed[0];
|
|
211
|
+
const credentialType = opts.byok && Object.prototype.hasOwnProperty.call(opts.byok, resolvedProvider) ? 'byok' : 'system';
|
|
212
|
+
const providerApiModelId = resolvedModelId.slice(resolvedModelId.indexOf('/') + 1);
|
|
213
|
+
const routing = {
|
|
214
|
+
originalModelId: modelId,
|
|
215
|
+
resolvedProvider,
|
|
216
|
+
resolvedProviderApiModelId: providerApiModelId,
|
|
217
|
+
fallbacksAvailable: allowed.slice(1),
|
|
218
|
+
planningReasoning: `${credentialType === 'byok' ? 'BYOK' : 'System'} credentials planned for: ${resolvedProvider}. Total execution order: ${allowed.map((p) => `${p}(${opts.byok && Object.prototype.hasOwnProperty.call(opts.byok, p) ? 'byok' : 'system'})`).join(', ')}`,
|
|
219
|
+
canonicalSlug: resolvedModelId,
|
|
220
|
+
finalProvider: resolvedProvider,
|
|
221
|
+
modelAttemptCount: modelAttempts.length + 1,
|
|
222
|
+
modelAttempts: [
|
|
223
|
+
...modelAttempts,
|
|
224
|
+
{
|
|
225
|
+
modelId: `${resolvedProvider}:${providerApiModelId}`,
|
|
226
|
+
canonicalSlug: resolvedModelId,
|
|
227
|
+
success: true,
|
|
228
|
+
providerAttemptCount: 1,
|
|
229
|
+
providerAttempts: [{
|
|
230
|
+
provider: resolvedProvider,
|
|
231
|
+
providerApiModelId,
|
|
232
|
+
credentialType,
|
|
233
|
+
success: true,
|
|
234
|
+
startTime: startMs,
|
|
235
|
+
endTime: startMs + 1000,
|
|
236
|
+
}],
|
|
237
|
+
},
|
|
238
|
+
],
|
|
239
|
+
totalProviderAttemptCount: 1,
|
|
240
|
+
...(sortMeta ? { sort: sortMeta } : {}),
|
|
241
|
+
};
|
|
242
|
+
return { entry, routing, credentialType };
|
|
243
|
+
}
|
|
244
|
+
const MESSAGE_ROLES = new Set(['system', 'developer', 'user', 'assistant', 'tool']);
|
|
245
|
+
const CONTENT_PART_TYPES = new Set(['text', 'image_url', 'file']);
|
|
246
|
+
function validateMessages(raw) {
|
|
247
|
+
if (!Array.isArray(raw) || raw.length === 0) {
|
|
248
|
+
return { response: error(400, 'messages must be a non-empty array', { param: 'messages', code: 'missing_parameter' }) };
|
|
249
|
+
}
|
|
250
|
+
for (const m of raw) {
|
|
251
|
+
if (!m || typeof m !== 'object' || typeof m.role !== 'string') {
|
|
252
|
+
return { response: error(400, 'each message must include a role', { param: 'messages' }) };
|
|
253
|
+
}
|
|
254
|
+
const message = m;
|
|
255
|
+
if (!MESSAGE_ROLES.has(message.role)) {
|
|
256
|
+
return { response: error(400, `unsupported message role: ${message.role}`, { param: 'messages' }) };
|
|
257
|
+
}
|
|
258
|
+
if (message.role === 'tool' && (typeof message.tool_call_id !== 'string' || !message.tool_call_id)) {
|
|
259
|
+
return { response: error(400, "messages with role 'tool' must include tool_call_id", { param: 'messages' }) };
|
|
260
|
+
}
|
|
261
|
+
const hasContent = typeof message.content === 'string'
|
|
262
|
+
? true
|
|
263
|
+
: Array.isArray(message.content) ? message.content.length > 0 : message.tool_calls !== undefined;
|
|
264
|
+
if (!hasContent) {
|
|
265
|
+
return { response: error(400, 'each message must include content or tool_calls', { param: 'messages' }) };
|
|
266
|
+
}
|
|
267
|
+
if (Array.isArray(message.content)) {
|
|
268
|
+
for (const part of message.content) {
|
|
269
|
+
const type = part?.type;
|
|
270
|
+
if (typeof type !== 'string' || !CONTENT_PART_TYPES.has(type)) {
|
|
271
|
+
return { response: error(400, `unsupported content part type: ${String(type)} (supported: text, image_url, file)`, { param: 'messages' }) };
|
|
272
|
+
}
|
|
273
|
+
if (type === 'image_url') {
|
|
274
|
+
const url = part.image_url?.url;
|
|
275
|
+
if (typeof url !== 'string' || !url)
|
|
276
|
+
return { response: error(400, 'image_url parts must include image_url.url', { param: 'messages' }) };
|
|
277
|
+
}
|
|
278
|
+
if (type === 'file') {
|
|
279
|
+
const file = part.file;
|
|
280
|
+
if (!file || typeof file !== 'object' || typeof file.data !== 'string') {
|
|
281
|
+
return { response: error(400, 'file parts must include file.data', { param: 'messages' }) };
|
|
282
|
+
}
|
|
283
|
+
}
|
|
284
|
+
}
|
|
285
|
+
}
|
|
286
|
+
}
|
|
287
|
+
return { messages: raw };
|
|
288
|
+
}
|
|
289
|
+
function validateSampling(params) {
|
|
290
|
+
const inRange = (v, min, max) => typeof v === 'number' && v >= min && v <= max;
|
|
291
|
+
if (params.temperature !== undefined && !inRange(params.temperature, 0, 2)) {
|
|
292
|
+
return error(400, 'temperature must be a number between 0 and 2', { param: 'temperature' });
|
|
293
|
+
}
|
|
294
|
+
if (params.top_p !== undefined && !inRange(params.top_p, 0, 1)) {
|
|
295
|
+
return error(400, 'top_p must be a number between 0 and 1', { param: 'top_p' });
|
|
296
|
+
}
|
|
297
|
+
if (params.frequency_penalty !== undefined && !inRange(params.frequency_penalty, -2, 2)) {
|
|
298
|
+
return error(400, 'frequency_penalty must be a number between -2 and 2', { param: 'frequency_penalty' });
|
|
299
|
+
}
|
|
300
|
+
if (params.presence_penalty !== undefined && !inRange(params.presence_penalty, -2, 2)) {
|
|
301
|
+
return error(400, 'presence_penalty must be a number between -2 and 2', { param: 'presence_penalty' });
|
|
302
|
+
}
|
|
303
|
+
if (params.stop !== undefined && typeof params.stop !== 'string' && !(Array.isArray(params.stop) && params.stop.every((s) => typeof s === 'string'))) {
|
|
304
|
+
return error(400, 'stop must be a string or an array of strings', { param: 'stop' });
|
|
305
|
+
}
|
|
306
|
+
return null;
|
|
307
|
+
}
|
|
308
|
+
// response_format per the gateway docs: OpenAI json_schema format, legacy `json` format, or text.
|
|
309
|
+
function parseResponseFormat(raw) {
|
|
310
|
+
if (raw === undefined)
|
|
311
|
+
return {};
|
|
312
|
+
if (!raw || typeof raw !== 'object' || Array.isArray(raw)) {
|
|
313
|
+
return { error: error(400, 'response_format must be an object', { param: 'response_format' }) };
|
|
314
|
+
}
|
|
315
|
+
const rf = raw;
|
|
316
|
+
// The gateway implements the OpenAI Chat Completions spec ('text'/'json_object'/'json_schema')
|
|
317
|
+
// plus its own documented legacy 'json' format.
|
|
318
|
+
if (rf.type !== 'text' && rf.type !== 'json_object' && rf.type !== 'json_schema' && rf.type !== 'json') {
|
|
319
|
+
return { error: error(400, "response_format.type must be one of 'text', 'json_object', 'json_schema', 'json'", { param: 'response_format' }) };
|
|
320
|
+
}
|
|
321
|
+
if (rf.type === 'json_schema') {
|
|
322
|
+
const js = rf.json_schema;
|
|
323
|
+
if (!js || typeof js !== 'object' || typeof js.name !== 'string') {
|
|
324
|
+
return { error: error(400, 'response_format.json_schema requires a name', { param: 'response_format' }) };
|
|
325
|
+
}
|
|
326
|
+
}
|
|
327
|
+
return { value: rf };
|
|
328
|
+
}
|
|
329
|
+
const EFFORTS = { none: 0, minimal: 0.1, low: 0.2, medium: 0.5, high: 0.8, xhigh: 0.95 };
|
|
330
|
+
function parseReasoning(raw) {
|
|
331
|
+
if (raw === undefined)
|
|
332
|
+
return {};
|
|
333
|
+
if (!raw || typeof raw !== 'object' || Array.isArray(raw)) {
|
|
334
|
+
return { error: error(400, 'reasoning must be an object', { param: 'reasoning' }) };
|
|
335
|
+
}
|
|
336
|
+
const r = raw;
|
|
337
|
+
if (r.effort !== undefined && (typeof r.effort !== 'string' || !(r.effort in EFFORTS))) {
|
|
338
|
+
return { error: error(400, "reasoning.effort must be one of 'none', 'minimal', 'low', 'medium', 'high', 'xhigh'", { param: 'reasoning.effort' }) };
|
|
339
|
+
}
|
|
340
|
+
if (r.max_tokens !== undefined && (!Number.isInteger(r.max_tokens) || r.max_tokens < 0)) {
|
|
341
|
+
return { error: error(400, 'reasoning.max_tokens must be a non-negative integer', { param: 'reasoning.max_tokens' }) };
|
|
342
|
+
}
|
|
343
|
+
// Documented as mutually exclusive ("Advanced Configuration", reasoning parameters).
|
|
344
|
+
if (r.effort !== undefined && r.max_tokens !== undefined) {
|
|
345
|
+
return { error: error(400, 'reasoning.effort and reasoning.max_tokens are mutually exclusive', { param: 'reasoning' }) };
|
|
346
|
+
}
|
|
347
|
+
return {
|
|
348
|
+
value: {
|
|
349
|
+
...(r.enabled !== undefined ? { enabled: r.enabled === true } : {}),
|
|
350
|
+
...(r.max_tokens !== undefined ? { max_tokens: Number(r.max_tokens) } : {}),
|
|
351
|
+
...(r.effort !== undefined ? { effort: String(r.effort) } : {}),
|
|
352
|
+
...(r.exclude !== undefined ? { exclude: r.exclude === true } : {}),
|
|
353
|
+
},
|
|
354
|
+
};
|
|
355
|
+
}
|
|
356
|
+
function reasoningTokens(args) {
|
|
357
|
+
const r = args.reasoning;
|
|
358
|
+
if (!r)
|
|
359
|
+
return 0;
|
|
360
|
+
if (r.effort === 'none')
|
|
361
|
+
return 0;
|
|
362
|
+
if (r.max_tokens !== undefined)
|
|
363
|
+
return r.max_tokens;
|
|
364
|
+
if (r.effort !== undefined)
|
|
365
|
+
return Math.round((args.maxTokens ?? 1000) * (EFFORTS[r.effort] ?? 0));
|
|
366
|
+
return r.enabled ? 256 : 0;
|
|
367
|
+
}
|
|
368
|
+
function reasoningActive(args) {
|
|
369
|
+
const r = args.reasoning;
|
|
370
|
+
if (!r)
|
|
371
|
+
return false;
|
|
372
|
+
if (r.effort === 'none')
|
|
373
|
+
return false;
|
|
374
|
+
return r.enabled === true || r.effort !== undefined || r.max_tokens !== undefined;
|
|
375
|
+
}
|
|
376
|
+
// cache_control markers (manual prompt caching, Anthropic-style, documented in the gateway's
|
|
377
|
+
// "Advanced Configuration" → Prompt caching): marked prompt tokens report as cached.
|
|
378
|
+
function cachedPromptTokens(messages) {
|
|
379
|
+
let cached = 0;
|
|
380
|
+
for (const message of messages) {
|
|
381
|
+
const marked = message.cache_control !== undefined ||
|
|
382
|
+
(Array.isArray(message.content) && message.content.some((p) => p && typeof p === 'object' && 'cache_control' in p));
|
|
383
|
+
if (marked)
|
|
384
|
+
cached += estimateTokens(`${message.role}: ${contentToText(message.content)}`);
|
|
385
|
+
}
|
|
386
|
+
return cached;
|
|
387
|
+
}
|
|
388
|
+
function validateChat(params) {
|
|
389
|
+
if (typeof params.model !== 'string' || !params.model) {
|
|
390
|
+
return { response: error(400, "Invalid request: missing required parameter 'model'", { param: 'model', code: 'missing_parameter' }) };
|
|
391
|
+
}
|
|
392
|
+
const messages = validateMessages(params.messages);
|
|
393
|
+
if ('response' in messages)
|
|
394
|
+
return messages;
|
|
395
|
+
const sampling = validateSampling(params);
|
|
396
|
+
if (sampling)
|
|
397
|
+
return { response: sampling };
|
|
398
|
+
const rawMax = params.max_completion_tokens ?? params.max_tokens;
|
|
399
|
+
if (rawMax !== undefined && (!Number.isInteger(rawMax) || rawMax < 1)) {
|
|
400
|
+
return { response: error(400, 'max_tokens must be an integer >= 1', { param: 'max_tokens' }) };
|
|
401
|
+
}
|
|
402
|
+
if (params.n !== undefined && (!Number.isInteger(params.n) || params.n < 1)) {
|
|
403
|
+
return { response: error(400, 'n must be an integer >= 1', { param: 'n' }) };
|
|
404
|
+
}
|
|
405
|
+
if (params.tools !== undefined && !Array.isArray(params.tools)) {
|
|
406
|
+
return { response: error(400, 'tools must be an array', { param: 'tools' }) };
|
|
407
|
+
}
|
|
408
|
+
if (params.tools !== undefined) {
|
|
409
|
+
for (const tool of params.tools) {
|
|
410
|
+
if (!normalizeTool(tool))
|
|
411
|
+
return { response: error(400, 'each tool must be a function tool with a name', { param: 'tools' }) };
|
|
412
|
+
}
|
|
413
|
+
}
|
|
414
|
+
if (params.tool_choice !== undefined) {
|
|
415
|
+
const tc = params.tool_choice;
|
|
416
|
+
const isString = tc === 'auto' || tc === 'none' || tc === 'required';
|
|
417
|
+
const isFn = !!tc && typeof tc === 'object' && tc.type === 'function' &&
|
|
418
|
+
typeof tc.function?.name === 'string';
|
|
419
|
+
if (!isString && !isFn) {
|
|
420
|
+
return { response: error(400, "tool_choice must be 'auto', 'none', 'required', or a named function", { param: 'tool_choice' }) };
|
|
421
|
+
}
|
|
422
|
+
}
|
|
423
|
+
const rf = parseResponseFormat(params.response_format);
|
|
424
|
+
if ('error' in rf)
|
|
425
|
+
return { response: rf.error };
|
|
426
|
+
const reasoning = parseReasoning(params.reasoning);
|
|
427
|
+
if ('error' in reasoning)
|
|
428
|
+
return { response: reasoning.error };
|
|
429
|
+
const gateway = parseGatewayOptions(params);
|
|
430
|
+
if ('error' in gateway)
|
|
431
|
+
return { response: gateway.error };
|
|
432
|
+
const msgs = messages.messages;
|
|
433
|
+
return {
|
|
434
|
+
args: {
|
|
435
|
+
model: params.model,
|
|
436
|
+
messages: msgs,
|
|
437
|
+
stream: params.stream === true,
|
|
438
|
+
n: params.n === undefined ? 1 : Number(params.n),
|
|
439
|
+
...(rawMax !== undefined ? { maxTokens: Number(rawMax) } : {}),
|
|
440
|
+
...(params.tools !== undefined ? { tools: params.tools } : {}),
|
|
441
|
+
...(params.tool_choice !== undefined ? { toolChoice: params.tool_choice } : {}),
|
|
442
|
+
...(rf.value ? { responseFormat: rf.value } : {}),
|
|
443
|
+
...(reasoning.value ? { reasoning: reasoning.value } : {}),
|
|
444
|
+
cachedTokens: Math.min(cachedPromptTokens(msgs), countPromptTokens(msgs)),
|
|
445
|
+
gateway: gateway.value,
|
|
446
|
+
},
|
|
447
|
+
};
|
|
448
|
+
}
|
|
449
|
+
function chooseTool(args) {
|
|
450
|
+
if (!args.tools || args.tools.length === 0 || args.toolChoice === 'none')
|
|
451
|
+
return { none: true };
|
|
452
|
+
if (args.toolChoice && typeof args.toolChoice === 'object') {
|
|
453
|
+
const wanted = args.toolChoice.function.name;
|
|
454
|
+
const found = args.tools.map(normalizeTool).find((t) => t?.name === wanted);
|
|
455
|
+
if (!found)
|
|
456
|
+
return { response: error(400, `tool_choice function '${wanted}' is not present in tools`, { param: 'tool_choice' }) };
|
|
457
|
+
return { tool: found };
|
|
458
|
+
}
|
|
459
|
+
return { tool: normalizeTool(args.tools[0]) };
|
|
460
|
+
}
|
|
461
|
+
function structuredJsonText(args) {
|
|
462
|
+
const rf = args.responseFormat;
|
|
463
|
+
if (!rf || rf.type === 'text')
|
|
464
|
+
return null;
|
|
465
|
+
if (rf.type === 'json_schema') {
|
|
466
|
+
const js = rf.json_schema;
|
|
467
|
+
return JSON.stringify(synthesizeJsonSchemaValue(js.schema));
|
|
468
|
+
}
|
|
469
|
+
// 'json_object' (OpenAI spec) and the legacy { type: 'json', schema?, name?, description? } format
|
|
470
|
+
if (rf.schema !== undefined)
|
|
471
|
+
return JSON.stringify(synthesizeJsonSchemaValue(rf.schema));
|
|
472
|
+
return JSON.stringify({ twin_stub: true, echo: lastUserText(args.messages) || 'empty prompt', model: args.model });
|
|
473
|
+
}
|
|
474
|
+
function reasoningDetails(modelId, text) {
|
|
475
|
+
if (modelId.startsWith('anthropic/')) {
|
|
476
|
+
return [{ type: 'reasoning.text', text, signature: `twin-signature-${stableHash(text)}`, format: 'anthropic-claude-v1', index: 0 }];
|
|
477
|
+
}
|
|
478
|
+
if (modelId.startsWith('openai/')) {
|
|
479
|
+
return [
|
|
480
|
+
{ type: 'reasoning.summary', summary: text, format: 'openai-responses-v1', index: 0 },
|
|
481
|
+
{ type: 'reasoning.encrypted', data: `twin-encrypted-${stableHash(text)}`, format: 'openai-responses-v1', index: 1 },
|
|
482
|
+
];
|
|
483
|
+
}
|
|
484
|
+
return [{ type: 'reasoning.text', text, format: 'unknown', index: 0 }];
|
|
485
|
+
}
|
|
486
|
+
function buildChoiceMessage(args, resolvedModelId, scripted) {
|
|
487
|
+
const message = { role: 'assistant' };
|
|
488
|
+
let finishReason = 'stop';
|
|
489
|
+
let completionTokens = 0;
|
|
490
|
+
let reasoningText = null;
|
|
491
|
+
let reasoningTokenCount = 0;
|
|
492
|
+
if (scripted) {
|
|
493
|
+
if (scripted.reasoning !== null)
|
|
494
|
+
reasoningText = scripted.reasoning;
|
|
495
|
+
if (scripted.toolCalls.length > 0) {
|
|
496
|
+
message.content = scripted.text; // null unless the rule also scripts text
|
|
497
|
+
message.tool_calls = scripted.toolCalls.map((tc, i) => ({
|
|
498
|
+
id: tc.id ?? `call_twin_${stableHash(`${tc.name}:${i}`).slice(0, 8)}`,
|
|
499
|
+
type: 'function',
|
|
500
|
+
function: { name: tc.name, arguments: JSON.stringify(tc.arguments) },
|
|
501
|
+
}));
|
|
502
|
+
completionTokens = estimateTokens(JSON.stringify(message.tool_calls));
|
|
503
|
+
}
|
|
504
|
+
else {
|
|
505
|
+
message.content = scripted.text ?? '';
|
|
506
|
+
completionTokens = estimateTokens(String(message.content));
|
|
507
|
+
}
|
|
508
|
+
finishReason = scripted.finishReason;
|
|
509
|
+
}
|
|
510
|
+
else {
|
|
511
|
+
const chosen = chooseTool(args);
|
|
512
|
+
if ('response' in chosen)
|
|
513
|
+
return chosen;
|
|
514
|
+
if ('tool' in chosen && chosen.tool) {
|
|
515
|
+
const call = buildToolCall(chosen.tool);
|
|
516
|
+
message.content = null;
|
|
517
|
+
message.tool_calls = [call];
|
|
518
|
+
finishReason = 'tool_calls';
|
|
519
|
+
completionTokens = estimateTokens(JSON.stringify(call));
|
|
520
|
+
}
|
|
521
|
+
else {
|
|
522
|
+
let text = structuredJsonText(args) ?? stubAssistantText(args.messages, args.model);
|
|
523
|
+
if (args.maxTokens !== undefined && estimateTokens(text) > args.maxTokens) {
|
|
524
|
+
text = text.slice(0, args.maxTokens * 4);
|
|
525
|
+
finishReason = 'length';
|
|
526
|
+
}
|
|
527
|
+
message.content = text;
|
|
528
|
+
completionTokens = estimateTokens(text);
|
|
529
|
+
}
|
|
530
|
+
if (reasoningActive(args)) {
|
|
531
|
+
reasoningText = `[twin-stub:ai-gateway:reasoning] deterministic reasoning trace for ${args.model}`;
|
|
532
|
+
reasoningTokenCount = reasoningTokens(args);
|
|
533
|
+
}
|
|
534
|
+
}
|
|
535
|
+
if (scripted?.reasoning != null)
|
|
536
|
+
reasoningTokenCount = estimateTokens(scripted.reasoning);
|
|
537
|
+
if (reasoningText !== null && args.reasoning?.exclude !== true) {
|
|
538
|
+
message.reasoning = reasoningText;
|
|
539
|
+
message.reasoning_details = reasoningDetails(resolvedModelId, reasoningText);
|
|
540
|
+
}
|
|
541
|
+
return { message, finishReason, completionTokens, reasoningTokenCount };
|
|
542
|
+
}
|
|
543
|
+
function formatUsd(n) {
|
|
544
|
+
const s = n.toFixed(10).replace(/0+$/, '');
|
|
545
|
+
const [int, dec = ''] = s.split('.');
|
|
546
|
+
return `${int}.${dec.length >= 2 ? dec : (dec + '00').slice(0, 2)}`;
|
|
547
|
+
}
|
|
548
|
+
function generationCost(entry, promptTokens, completionTokens) {
|
|
549
|
+
const input = Number(entry.model.pricing.input);
|
|
550
|
+
const output = Number(entry.model.pricing.output ?? '0');
|
|
551
|
+
return Number((promptTokens * input + completionTokens * output).toFixed(10));
|
|
552
|
+
}
|
|
553
|
+
// Generation ids are UNIQUE per generation (docs: `gen_<ulid>`): the seed includes a per-root
|
|
554
|
+
// monotonic sequence (the count of already-recorded generations), so identical back-to-back
|
|
555
|
+
// requests still mint distinct ids and the kernel's (type, id) resolution can never return a
|
|
556
|
+
// different request's record. Deterministic: same request at the same log position → same id.
|
|
557
|
+
function generationId(seed, seq) {
|
|
558
|
+
const s = [seed, seq];
|
|
559
|
+
const hex = `${stableHash(s)}${stableHash([s, 'gen'])}${stableHash([s, 'id'])}${stableHash([s, 'x'])}`;
|
|
560
|
+
return `gen_${hex.toUpperCase().slice(0, 26)}`;
|
|
561
|
+
}
|
|
562
|
+
async function recordGeneration(record, root, occurredAt) {
|
|
563
|
+
await applyTwinWrite(SERVICE, {
|
|
564
|
+
operation: 'generation.record',
|
|
565
|
+
subjectType: 'generation',
|
|
566
|
+
subjectId: record.id,
|
|
567
|
+
fields: record,
|
|
568
|
+
occurredAt: occurredAt ?? FIXED_NOW,
|
|
569
|
+
actor: { kind: 'system' },
|
|
570
|
+
}, root);
|
|
571
|
+
}
|
|
572
|
+
async function chatCompletion(req, params) {
|
|
573
|
+
const validated = validateChat(params);
|
|
574
|
+
if ('response' in validated)
|
|
575
|
+
return validated.response;
|
|
576
|
+
const args = validated.args;
|
|
577
|
+
const resolved = resolveRouting(args.model, args.gateway, req.occurredAt);
|
|
578
|
+
if ('response' in resolved)
|
|
579
|
+
return resolved.response;
|
|
580
|
+
const { entry, routing, credentialType } = resolved;
|
|
581
|
+
let missTeach = '';
|
|
582
|
+
const scripted = req.scenarioEngine
|
|
583
|
+
? (() => {
|
|
584
|
+
const decision = req.scenarioEngine.next({ model: args.model, messages: args.messages, ...(args.tools ? { tools: args.tools } : {}) });
|
|
585
|
+
if (decision.kind === 'handler')
|
|
586
|
+
return realizeAiGatewayRespond(decision.respond);
|
|
587
|
+
missTeach = `\n[twin-scenario miss — no handler matched. Author one in the world dir's handlers/ai-gateway.json (GET /twin explains; GET /twin/scenario lists handlers + misses). Features seen: ${JSON.stringify(decision.miss.features)}]`;
|
|
588
|
+
return null;
|
|
589
|
+
})()
|
|
590
|
+
: null;
|
|
591
|
+
const built = buildChoiceMessage(args, entry.model.id, scripted);
|
|
592
|
+
if ('response' in built)
|
|
593
|
+
return built.response;
|
|
594
|
+
if (missTeach && typeof built.message.content === 'string') {
|
|
595
|
+
built.message.content += missTeach;
|
|
596
|
+
}
|
|
597
|
+
const promptTokens = countPromptTokens(args.messages);
|
|
598
|
+
const perChoiceTokens = built.completionTokens + built.reasoningTokenCount;
|
|
599
|
+
const completionTokens = perChoiceTokens * args.n;
|
|
600
|
+
const marketCost = generationCost(entry, promptTokens, completionTokens);
|
|
601
|
+
// BYOK accounting per the docs: the provider bills the tokens, so the gateway's total_cost
|
|
602
|
+
// for the generation is 0.00 and the market price is reported as upstream_inference_cost.
|
|
603
|
+
const isByok = credentialType === 'byok';
|
|
604
|
+
const totalCost = isByok ? 0 : marketCost;
|
|
605
|
+
// ATOMICITY ASSUMPTION: no `await` may sit between this seq read and the recordGeneration()
|
|
606
|
+
// write below — that synchronous prefix is what makes concurrent identical requests mint
|
|
607
|
+
// unique ids (verified: 25 concurrent identical POSTs → 25 unique ids). An added await here
|
|
608
|
+
// would reintroduce the id-collision race.
|
|
609
|
+
const seq = generationRows(req.root).length;
|
|
610
|
+
const id = generationId({ args, promptTokens }, seq);
|
|
611
|
+
const usage = {
|
|
612
|
+
prompt_tokens: promptTokens,
|
|
613
|
+
completion_tokens: completionTokens,
|
|
614
|
+
total_tokens: promptTokens + completionTokens,
|
|
615
|
+
...(args.cachedTokens > 0 ? { prompt_tokens_details: { cached_tokens: args.cachedTokens } } : {}),
|
|
616
|
+
...(built.reasoningTokenCount > 0 ? { completion_tokens_details: { reasoning_tokens: built.reasoningTokenCount * args.n } } : {}),
|
|
617
|
+
};
|
|
618
|
+
const body = {
|
|
619
|
+
id,
|
|
620
|
+
object: 'chat.completion',
|
|
621
|
+
created: nowEpoch(req.occurredAt),
|
|
622
|
+
model: entry.model.id,
|
|
623
|
+
choices: Array.from({ length: args.n }, (_u, index) => ({ index, message: built.message, finish_reason: built.finishReason })),
|
|
624
|
+
usage,
|
|
625
|
+
providerMetadata: {
|
|
626
|
+
gateway: {
|
|
627
|
+
routing,
|
|
628
|
+
cost: formatUsd(totalCost),
|
|
629
|
+
marketCost: formatUsd(marketCost),
|
|
630
|
+
generationId: id,
|
|
631
|
+
},
|
|
632
|
+
},
|
|
633
|
+
};
|
|
634
|
+
await recordGeneration({
|
|
635
|
+
id,
|
|
636
|
+
total_cost: totalCost,
|
|
637
|
+
upstream_inference_cost: isByok ? marketCost : 0,
|
|
638
|
+
usage: totalCost,
|
|
639
|
+
created_at: nowIso(req.occurredAt),
|
|
640
|
+
model: entry.model.id,
|
|
641
|
+
is_byok: isByok,
|
|
642
|
+
provider_name: routing.finalProvider,
|
|
643
|
+
streamed: args.stream,
|
|
644
|
+
finish_reason: built.finishReason,
|
|
645
|
+
latency: 200,
|
|
646
|
+
generation_time: 1000 + completionTokens * 10,
|
|
647
|
+
tokens_prompt: promptTokens,
|
|
648
|
+
tokens_completion: completionTokens,
|
|
649
|
+
native_tokens_prompt: promptTokens,
|
|
650
|
+
native_tokens_completion: completionTokens,
|
|
651
|
+
native_tokens_reasoning: built.reasoningTokenCount * args.n,
|
|
652
|
+
native_tokens_cached: args.cachedTokens,
|
|
653
|
+
native_tokens_cache_creation: 0,
|
|
654
|
+
billable_web_search_calls: 0,
|
|
655
|
+
}, req.root, req.occurredAt);
|
|
656
|
+
if (args.stream && req.sseSink)
|
|
657
|
+
streamChat(body, req.sseSink);
|
|
658
|
+
return { status: 200, body };
|
|
659
|
+
}
|
|
660
|
+
// SSE streaming: OpenAI chat.completion.chunk envelope; the generation id rides on every chunk
|
|
661
|
+
// starting with the FIRST (the gateway documents injecting the generation id into the first
|
|
662
|
+
// content chunk); reasoning streams via delta.reasoning (+ delta.reasoning_details); ends with
|
|
663
|
+
// a finish_reason + usage chunk and `data: [DONE]`. EVERY choice streams (OpenAI semantics:
|
|
664
|
+
// each chunk carries one element of `choices` with its own index), so streaming with n>1 emits
|
|
665
|
+
// the same n choices the unary body carries and bills — not just index 0. The final chunk (the
|
|
666
|
+
// last choice's finish) carries the n-scaled usage.
|
|
667
|
+
function streamChat(body, sink) {
|
|
668
|
+
const base = { id: body.id, object: 'chat.completion.chunk', created: body.created, model: body.model };
|
|
669
|
+
const choices = body.choices;
|
|
670
|
+
choices.forEach((choice, choiceIndex) => {
|
|
671
|
+
const message = choice.message;
|
|
672
|
+
sink({ data: { ...base, choices: [{ index: choiceIndex, delta: { role: 'assistant' }, finish_reason: null }] } });
|
|
673
|
+
if (typeof message.reasoning === 'string') {
|
|
674
|
+
const reasoning = message.reasoning;
|
|
675
|
+
for (let i = 0; i < reasoning.length; i += 48) {
|
|
676
|
+
const slice = reasoning.slice(i, i + 48);
|
|
677
|
+
sink({ data: { ...base, choices: [{ index: choiceIndex, delta: { reasoning: slice, ...(i === 0 && message.reasoning_details ? { reasoning_details: message.reasoning_details } : {}) }, finish_reason: null }] } });
|
|
678
|
+
}
|
|
679
|
+
}
|
|
680
|
+
if (Array.isArray(message.tool_calls)) {
|
|
681
|
+
message.tool_calls.forEach((call, index) => {
|
|
682
|
+
const fn = call.function;
|
|
683
|
+
sink({ data: { ...base, choices: [{ index: choiceIndex, delta: { tool_calls: [{ index, id: call.id, type: 'function', function: { name: fn.name, arguments: '' } }] }, finish_reason: null }] } });
|
|
684
|
+
sink({ data: { ...base, choices: [{ index: choiceIndex, delta: { tool_calls: [{ index, function: { arguments: fn.arguments } }] }, finish_reason: null }] } });
|
|
685
|
+
});
|
|
686
|
+
}
|
|
687
|
+
else {
|
|
688
|
+
const text = String(message.content ?? '');
|
|
689
|
+
for (let i = 0; i < text.length; i += 24) {
|
|
690
|
+
sink({ data: { ...base, choices: [{ index: choiceIndex, delta: { content: text.slice(i, i + 24) }, finish_reason: null }] } });
|
|
691
|
+
}
|
|
692
|
+
}
|
|
693
|
+
const isLast = choiceIndex === choices.length - 1;
|
|
694
|
+
sink({ data: { ...base, choices: [{ index: choiceIndex, delta: {}, finish_reason: choice.finish_reason }], ...(isLast ? { usage: body.usage } : {}) } });
|
|
695
|
+
});
|
|
696
|
+
sink({ done: true });
|
|
697
|
+
}
|
|
698
|
+
// ── embeddings ────────────────────────────────────────────────────────────────────────────
|
|
699
|
+
// Matches the real default dimension of openai/text-embedding-3-small, so client code that
|
|
700
|
+
// sizes buffers/vector columns off the twin doesn't break against the vendor.
|
|
701
|
+
const DEFAULT_EMBED_DIM = 1536;
|
|
702
|
+
function embedVector(text, dim) {
|
|
703
|
+
// Deterministic pseudo-embedding seeded from the text hash; L2-normalized.
|
|
704
|
+
const seed = parseInt(stableHash(text), 16) || 1;
|
|
705
|
+
let x = seed;
|
|
706
|
+
const raw = Array.from({ length: dim }, () => {
|
|
707
|
+
x = (Math.imul(x, 1103515245) + 12345) & 0x7fffffff;
|
|
708
|
+
return (x / 0x7fffffff) * 2 - 1;
|
|
709
|
+
});
|
|
710
|
+
const norm = Math.sqrt(raw.reduce((s, v) => s + v * v, 0)) || 1;
|
|
711
|
+
return raw.map((v) => Number((v / norm).toFixed(6)));
|
|
712
|
+
}
|
|
713
|
+
async function createEmbeddings(req, params) {
|
|
714
|
+
if (typeof params.model !== 'string' || !params.model) {
|
|
715
|
+
return error(400, "Invalid request: missing required parameter 'model'", { param: 'model', code: 'missing_parameter' });
|
|
716
|
+
}
|
|
717
|
+
const entry = findAiGatewayModel(params.model);
|
|
718
|
+
if (!entry)
|
|
719
|
+
return modelNotFound(params.model);
|
|
720
|
+
if (entry.model.type !== 'embedding') {
|
|
721
|
+
return error(400, `The model '${params.model}' is not an embedding model.`, { param: 'model' });
|
|
722
|
+
}
|
|
723
|
+
if (params.input === undefined)
|
|
724
|
+
return error(400, 'input is required', { param: 'input', code: 'missing_parameter' });
|
|
725
|
+
const inputs = Array.isArray(params.input) ? params.input : [params.input];
|
|
726
|
+
if (inputs.length === 0 || inputs.some((i) => typeof i !== 'string')) {
|
|
727
|
+
return error(400, 'input must be a non-empty string or array of strings', { param: 'input' });
|
|
728
|
+
}
|
|
729
|
+
let dim = DEFAULT_EMBED_DIM;
|
|
730
|
+
if (params.dimensions !== undefined) {
|
|
731
|
+
if (!Number.isInteger(params.dimensions) || params.dimensions < 1 || params.dimensions > 4096) {
|
|
732
|
+
return error(400, 'dimensions must be an integer between 1 and 4096', { param: 'dimensions' });
|
|
733
|
+
}
|
|
734
|
+
dim = params.dimensions;
|
|
735
|
+
}
|
|
736
|
+
const gateway = parseGatewayOptions(params);
|
|
737
|
+
if ('error' in gateway)
|
|
738
|
+
return gateway.error;
|
|
739
|
+
const resolved = resolveRouting(entry.model.id, gateway.value, req.occurredAt);
|
|
740
|
+
if ('response' in resolved)
|
|
741
|
+
return resolved.response;
|
|
742
|
+
const data = inputs.map((text, index) => ({ object: 'embedding', index, embedding: embedVector(`${text}:${dim}`, dim) }));
|
|
743
|
+
const promptTokens = inputs.reduce((s, t) => s + estimateTokens(t), 0);
|
|
744
|
+
const marketCost = Number((promptTokens * Number(entry.model.pricing.input)).toFixed(10));
|
|
745
|
+
// The real gateway BILLS embeddings — so the twin records a generation for every embeddings
|
|
746
|
+
// call through the same kernel fold as chat: credits (/v1/credits), the spend report
|
|
747
|
+
// (/v1/report), and generation lookup (/v1/generation?id) all move. Same BYOK rule as chat:
|
|
748
|
+
// BYOK requests record total_cost 0 with the market price as upstream_inference_cost.
|
|
749
|
+
const isByok = resolved.credentialType === 'byok';
|
|
750
|
+
const totalCost = isByok ? 0 : marketCost;
|
|
751
|
+
// Same ATOMICITY rule as chatCompletion: no `await` between this seq read and the
|
|
752
|
+
// recordGeneration() write, so concurrent identical requests still mint unique ids.
|
|
753
|
+
const seq = generationRows(req.root).length;
|
|
754
|
+
const id = generationId({ embeddings: { model: entry.model.id, inputs, dim }, promptTokens }, seq);
|
|
755
|
+
await recordGeneration({
|
|
756
|
+
id,
|
|
757
|
+
total_cost: totalCost,
|
|
758
|
+
upstream_inference_cost: isByok ? marketCost : 0,
|
|
759
|
+
usage: totalCost,
|
|
760
|
+
created_at: nowIso(req.occurredAt),
|
|
761
|
+
model: entry.model.id,
|
|
762
|
+
is_byok: isByok,
|
|
763
|
+
provider_name: resolved.routing.finalProvider,
|
|
764
|
+
streamed: false,
|
|
765
|
+
finish_reason: 'stop',
|
|
766
|
+
latency: 200,
|
|
767
|
+
generation_time: 100 + promptTokens,
|
|
768
|
+
tokens_prompt: promptTokens,
|
|
769
|
+
tokens_completion: 0,
|
|
770
|
+
native_tokens_prompt: promptTokens,
|
|
771
|
+
native_tokens_completion: 0,
|
|
772
|
+
native_tokens_reasoning: 0,
|
|
773
|
+
native_tokens_cached: 0,
|
|
774
|
+
native_tokens_cache_creation: 0,
|
|
775
|
+
billable_web_search_calls: 0,
|
|
776
|
+
}, req.root, req.occurredAt);
|
|
777
|
+
return {
|
|
778
|
+
status: 200,
|
|
779
|
+
body: {
|
|
780
|
+
object: 'list',
|
|
781
|
+
data,
|
|
782
|
+
model: entry.model.id,
|
|
783
|
+
usage: { prompt_tokens: promptTokens, total_tokens: promptTokens },
|
|
784
|
+
providerMetadata: { gateway: { routing: resolved.routing, cost: formatUsd(totalCost), marketCost: formatUsd(marketCost), generationId: id } },
|
|
785
|
+
},
|
|
786
|
+
};
|
|
787
|
+
}
|
|
788
|
+
// ── models / credits / generation / report ────────────────────────────────────────────────
|
|
789
|
+
function modelEndpoints(entry) {
|
|
790
|
+
const m = entry.model;
|
|
791
|
+
return {
|
|
792
|
+
status: 200,
|
|
793
|
+
body: {
|
|
794
|
+
data: {
|
|
795
|
+
id: m.id,
|
|
796
|
+
name: m.name,
|
|
797
|
+
created: m.created,
|
|
798
|
+
released: m.released,
|
|
799
|
+
description: m.description,
|
|
800
|
+
architecture: {
|
|
801
|
+
tokenizer: null,
|
|
802
|
+
instruct_type: null,
|
|
803
|
+
modality: m.tags.includes('vision') ? 'text+image→text' : 'text→text',
|
|
804
|
+
input_modalities: m.tags.includes('vision') ? ['text', 'image'] : ['text'],
|
|
805
|
+
output_modalities: ['text'],
|
|
806
|
+
},
|
|
807
|
+
endpoints: entry.providers.map((provider) => ({
|
|
808
|
+
name: `${provider} | ${m.id}`,
|
|
809
|
+
model_name: m.name,
|
|
810
|
+
context_length: m.context_window,
|
|
811
|
+
pricing: {
|
|
812
|
+
prompt: m.pricing.input,
|
|
813
|
+
completion: m.pricing.output ?? '0',
|
|
814
|
+
...(m.pricing.input_cache_read ? { input_cache_read: m.pricing.input_cache_read } : {}),
|
|
815
|
+
...(m.pricing.input_cache_write ? { input_cache_write: m.pricing.input_cache_write } : {}),
|
|
816
|
+
},
|
|
817
|
+
provider_name: provider,
|
|
818
|
+
max_completion_tokens: m.max_tokens,
|
|
819
|
+
supported_parameters: ['max_tokens', 'temperature', 'tools', 'reasoning'],
|
|
820
|
+
status: 0,
|
|
821
|
+
uptime_last_15m: 100,
|
|
822
|
+
uptime_last_1h: 100,
|
|
823
|
+
uptime_last_1d: 100,
|
|
824
|
+
throughput_last_1h: { p50: providerMetric(provider, 'tps'), p95: providerMetric(provider, 'tps') + 5 },
|
|
825
|
+
latency_last_1h: { p50: providerMetric(provider, 'ttft'), p95: providerMetric(provider, 'ttft') + 200 },
|
|
826
|
+
supports_implicit_caching: false,
|
|
827
|
+
})),
|
|
828
|
+
},
|
|
829
|
+
},
|
|
830
|
+
};
|
|
831
|
+
}
|
|
832
|
+
function credits(root) {
|
|
833
|
+
const used = generationRows(root).reduce((sum, g) => sum + g.total_cost, 0);
|
|
834
|
+
return {
|
|
835
|
+
status: 200,
|
|
836
|
+
body: { balance: formatUsd(STARTING_CREDITS - used), total_used: formatUsd(used) },
|
|
837
|
+
};
|
|
838
|
+
}
|
|
839
|
+
function generationLookup(req, url) {
|
|
840
|
+
const id = url.searchParams.get('id');
|
|
841
|
+
if (!id)
|
|
842
|
+
return error(400, "Invalid request: missing required parameter 'id'", { param: 'id', code: 'missing_parameter' });
|
|
843
|
+
const found = generationRows(req.root).find((g) => g.id === id);
|
|
844
|
+
if (!found)
|
|
845
|
+
return error(404, `Generation '${id}' not found.`, { code: 'not_found' });
|
|
846
|
+
return { status: 200, body: { data: found } };
|
|
847
|
+
}
|
|
848
|
+
// Spend report (GET /v1/report), per the documented Custom Reporting response format: a
|
|
849
|
+
// `results` array whose rows carry the grouping field (day/model/provider/credential_type/…)
|
|
850
|
+
// plus the metric fields (total_cost/market_cost/surcharge_cost/gateway_cost, token counts,
|
|
851
|
+
// request_count). The twin records no per-request user/tag/api-key attribution yet (see the
|
|
852
|
+
// `ai-gateway.chat.reporting_tags` / `ai-gateway.report.attribution_groupings` todos), so
|
|
853
|
+
// attribution-keyed groupings and filters resolve to empty result sets rather than fabricating
|
|
854
|
+
// attribution that was never recorded.
|
|
855
|
+
const REPORT_GROUP_BY = new Set(['day', 'user', 'model', 'tag', 'provider', 'credential_type', 'zero_data_retention', 'api_key_name']);
|
|
856
|
+
const DATE_RE = /^\d{4}-\d{2}-\d{2}$/;
|
|
857
|
+
function spendReport(req, url) {
|
|
858
|
+
const q = url.searchParams;
|
|
859
|
+
const start = q.get('start_date');
|
|
860
|
+
const end = q.get('end_date');
|
|
861
|
+
if (!start || !DATE_RE.test(start))
|
|
862
|
+
return error(400, 'start_date is required in YYYY-MM-DD format', { param: 'start_date' });
|
|
863
|
+
if (!end || !DATE_RE.test(end))
|
|
864
|
+
return error(400, 'end_date is required in YYYY-MM-DD format', { param: 'end_date' });
|
|
865
|
+
const groupBy = q.get('group_by') ?? 'day';
|
|
866
|
+
if (!REPORT_GROUP_BY.has(groupBy)) {
|
|
867
|
+
return error(400, `group_by must be one of ${[...REPORT_GROUP_BY].join(', ')}`, { param: 'group_by' });
|
|
868
|
+
}
|
|
869
|
+
const datePart = q.get('date_part') ?? 'day';
|
|
870
|
+
if (datePart !== 'day' && datePart !== 'hour') {
|
|
871
|
+
return error(400, "date_part must be 'day' or 'hour'", { param: 'date_part' });
|
|
872
|
+
}
|
|
873
|
+
const credentialType = q.get('credential_type');
|
|
874
|
+
if (credentialType !== null && credentialType !== 'byok' && credentialType !== 'system') {
|
|
875
|
+
return error(400, "credential_type must be 'byok' or 'system'", { param: 'credential_type' });
|
|
876
|
+
}
|
|
877
|
+
const zdr = q.get('zero_data_retention');
|
|
878
|
+
if (zdr !== null && zdr !== 'true' && zdr !== 'false') {
|
|
879
|
+
return error(400, 'zero_data_retention must be true or false', { param: 'zero_data_retention' });
|
|
880
|
+
}
|
|
881
|
+
const tagsMatch = q.get('tags_match') ?? 'any';
|
|
882
|
+
if (tagsMatch !== 'any' && tagsMatch !== 'all') {
|
|
883
|
+
return error(400, "tags_match must be 'any' or 'all'", { param: 'tags_match' });
|
|
884
|
+
}
|
|
885
|
+
let rows = generationRows(req.root).filter((g) => {
|
|
886
|
+
const day = g.created_at.slice(0, 10);
|
|
887
|
+
return day >= start && day <= end;
|
|
888
|
+
});
|
|
889
|
+
const model = q.get('model');
|
|
890
|
+
if (model)
|
|
891
|
+
rows = rows.filter((g) => g.model === model);
|
|
892
|
+
const provider = q.get('provider');
|
|
893
|
+
if (provider)
|
|
894
|
+
rows = rows.filter((g) => g.provider_name === provider);
|
|
895
|
+
if (credentialType)
|
|
896
|
+
rows = rows.filter((g) => (g.is_byok ? 'byok' : 'system') === credentialType);
|
|
897
|
+
// No generation ever requests ZDR / carries user or tag attribution in the twin yet:
|
|
898
|
+
if (zdr === 'true')
|
|
899
|
+
rows = [];
|
|
900
|
+
if (q.get('user_id') || q.get('tags'))
|
|
901
|
+
rows = [];
|
|
902
|
+
const keyOf = (g) => {
|
|
903
|
+
// date_part=hour buckets are `YYYY-MM-DDTHH` under the `hour` grouping field (docs).
|
|
904
|
+
if (groupBy === 'day')
|
|
905
|
+
return datePart === 'hour' ? g.created_at.slice(0, 13) : g.created_at.slice(0, 10);
|
|
906
|
+
if (groupBy === 'model')
|
|
907
|
+
return g.model;
|
|
908
|
+
if (groupBy === 'provider')
|
|
909
|
+
return g.provider_name;
|
|
910
|
+
if (groupBy === 'credential_type')
|
|
911
|
+
return g.is_byok ? 'byok' : 'system';
|
|
912
|
+
if (groupBy === 'zero_data_retention')
|
|
913
|
+
return 'false'; // no ZDR requests recorded
|
|
914
|
+
return null; // user/tag/api_key_name: no attribution recorded
|
|
915
|
+
};
|
|
916
|
+
const groups = new Map();
|
|
917
|
+
for (const g of rows) {
|
|
918
|
+
const key = keyOf(g);
|
|
919
|
+
if (key === null)
|
|
920
|
+
continue;
|
|
921
|
+
const cur = groups.get(key) ?? {
|
|
922
|
+
total_cost: 0, market_cost: 0, surcharge_cost: 0, gateway_cost: 0,
|
|
923
|
+
input_tokens: 0, output_tokens: 0, cached_input_tokens: 0,
|
|
924
|
+
cache_creation_input_tokens: 0, reasoning_tokens: 0, request_count: 0,
|
|
925
|
+
};
|
|
926
|
+
cur.total_cost = Number((cur.total_cost + g.total_cost).toFixed(10));
|
|
927
|
+
// market_cost includes both BYOK and non-BYOK cost; gateway_cost is the gateway's OWN cost
|
|
928
|
+
// separate from the provider rate — 0 here ("no markup on tokens", matching the docs sample).
|
|
929
|
+
cur.market_cost = Number((cur.market_cost + g.total_cost + g.upstream_inference_cost).toFixed(10));
|
|
930
|
+
cur.input_tokens += g.tokens_prompt;
|
|
931
|
+
cur.output_tokens += g.tokens_completion;
|
|
932
|
+
cur.cached_input_tokens += g.native_tokens_cached;
|
|
933
|
+
cur.cache_creation_input_tokens += g.native_tokens_cache_creation;
|
|
934
|
+
cur.reasoning_tokens += g.native_tokens_reasoning;
|
|
935
|
+
cur.request_count += 1;
|
|
936
|
+
groups.set(key, cur);
|
|
937
|
+
}
|
|
938
|
+
const field = groupBy === 'day' ? (datePart === 'hour' ? 'hour' : 'day') : groupBy;
|
|
939
|
+
const results = [...groups.entries()]
|
|
940
|
+
.map(([key, value]) => ({ [field]: key, ...value }))
|
|
941
|
+
.sort((a, b) => (String(a[field]) < String(b[field]) ? -1 : 1));
|
|
942
|
+
return { status: 200, body: { results } };
|
|
943
|
+
}
|
|
944
|
+
// ── router ────────────────────────────────────────────────────────────────────────────────
|
|
945
|
+
export async function handleAiGatewayTwinRequest(req) {
|
|
946
|
+
const method = req.method.toUpperCase();
|
|
947
|
+
const url = new URL(req.path, 'http://twin.local');
|
|
948
|
+
const path = url.pathname;
|
|
949
|
+
if (method === 'GET' && path === '/v1/models') {
|
|
950
|
+
return { status: 200, body: { object: 'list', data: AI_GATEWAY_MODELS } };
|
|
951
|
+
}
|
|
952
|
+
if (method === 'GET' && path.startsWith('/v1/models/') && path.endsWith('/endpoints')) {
|
|
953
|
+
const id = decodeURIComponent(path.slice('/v1/models/'.length, -'/endpoints'.length));
|
|
954
|
+
const entry = findAiGatewayModel(id);
|
|
955
|
+
return entry ? modelEndpoints(entry) : modelNotFound(id);
|
|
956
|
+
}
|
|
957
|
+
if (method === 'GET' && path.startsWith('/v1/models/')) {
|
|
958
|
+
const id = decodeURIComponent(path.slice('/v1/models/'.length));
|
|
959
|
+
const entry = findAiGatewayModel(id);
|
|
960
|
+
return entry ? { status: 200, body: entry.model } : modelNotFound(id);
|
|
961
|
+
}
|
|
962
|
+
if (method === 'GET' && path === '/v1/credits')
|
|
963
|
+
return credits(req.root);
|
|
964
|
+
if (method === 'GET' && path === '/v1/generation')
|
|
965
|
+
return generationLookup(req, url);
|
|
966
|
+
if (method === 'GET' && path === '/v1/report')
|
|
967
|
+
return spendReport(req, url);
|
|
968
|
+
if (method === 'POST' && path === '/v1/chat/completions') {
|
|
969
|
+
if (req.readOnly)
|
|
970
|
+
return readOnlyRejected();
|
|
971
|
+
return chatCompletion(req, parseJson(req.body));
|
|
972
|
+
}
|
|
973
|
+
if (method === 'POST' && path === '/v1/embeddings') {
|
|
974
|
+
if (req.readOnly)
|
|
975
|
+
return readOnlyRejected();
|
|
976
|
+
return createEmbeddings(req, parseJson(req.body));
|
|
977
|
+
}
|
|
978
|
+
// ── AI SDK gateway protocol (/v3/ai) — what `createGateway` (the `ai` package /
|
|
979
|
+
// @ai-sdk/gateway) actually speaks; see ai-gateway-v3.ts. Language-model calls run through
|
|
980
|
+
// the SAME chatCompletion fold as /v1, so accounting is one surface across both protocols.
|
|
981
|
+
if (method === 'GET' && path === '/v3/ai/config')
|
|
982
|
+
return v3Config();
|
|
983
|
+
if (method === 'POST' && path === '/v3/ai/language-model') {
|
|
984
|
+
if (req.readOnly)
|
|
985
|
+
return readOnlyRejected();
|
|
986
|
+
return handleV3LanguageModel(req, chatCompletion);
|
|
987
|
+
}
|
|
988
|
+
if (path.startsWith('/v3/ai/')) {
|
|
989
|
+
// embedding-model / image-model / video-model are real /v3/ai sub-surfaces the twin does
|
|
990
|
+
// not model yet — honest 404s (open todos in the capability manifest), never fake success.
|
|
991
|
+
return error(404, `AI SDK gateway protocol operation not modeled: ${method} ${path} (modeled: GET /v3/ai/config, POST /v3/ai/language-model)`, { code: 'not_found' });
|
|
992
|
+
}
|
|
993
|
+
return error(404, `AI Gateway operation not modeled: ${method} ${path}`, { code: 'not_found' });
|
|
994
|
+
}
|