@volter/twin-ai-gateway 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (42) hide show
  1. package/LICENSE +202 -0
  2. package/README.md +176 -0
  3. package/dist/src/ai-gateway-capabilities.d.ts +4 -0
  4. package/dist/src/ai-gateway-capabilities.js +772 -0
  5. package/dist/src/ai-gateway-conformance.d.ts +11 -0
  6. package/dist/src/ai-gateway-conformance.js +72 -0
  7. package/dist/src/ai-gateway-connector.d.ts +50 -0
  8. package/dist/src/ai-gateway-connector.js +97 -0
  9. package/dist/src/ai-gateway-models.d.ts +27 -0
  10. package/dist/src/ai-gateway-models.js +65 -0
  11. package/dist/src/ai-gateway-perform-harness.d.ts +5 -0
  12. package/dist/src/ai-gateway-perform-harness.js +17 -0
  13. package/dist/src/ai-gateway-scenario.d.ts +36 -0
  14. package/dist/src/ai-gateway-scenario.js +125 -0
  15. package/dist/src/ai-gateway-server.d.ts +16 -0
  16. package/dist/src/ai-gateway-server.js +107 -0
  17. package/dist/src/ai-gateway-stub.d.ts +20 -0
  18. package/dist/src/ai-gateway-stub.js +124 -0
  19. package/dist/src/ai-gateway-twin.d.ts +2 -0
  20. package/dist/src/ai-gateway-twin.js +994 -0
  21. package/dist/src/ai-gateway-types.d.ts +94 -0
  22. package/dist/src/ai-gateway-types.js +1 -0
  23. package/dist/src/ai-gateway-v3.d.ts +5 -0
  24. package/dist/src/ai-gateway-v3.js +367 -0
  25. package/dist/src/cli.d.ts +2 -0
  26. package/dist/src/cli.js +24 -0
  27. package/dist/src/index.d.ts +12 -0
  28. package/dist/src/index.js +54 -0
  29. package/package.json +66 -0
  30. package/src/ai-gateway-capabilities.ts +894 -0
  31. package/src/ai-gateway-conformance.ts +77 -0
  32. package/src/ai-gateway-connector.ts +100 -0
  33. package/src/ai-gateway-models.ts +105 -0
  34. package/src/ai-gateway-perform-harness.ts +17 -0
  35. package/src/ai-gateway-scenario.ts +137 -0
  36. package/src/ai-gateway-server.ts +122 -0
  37. package/src/ai-gateway-stub.ts +115 -0
  38. package/src/ai-gateway-twin.ts +1068 -0
  39. package/src/ai-gateway-types.ts +96 -0
  40. package/src/ai-gateway-v3.ts +372 -0
  41. package/src/cli.ts +23 -0
  42. package/src/index.ts +67 -0
@@ -0,0 +1,94 @@
1
+ export type SseEvent = {
2
+ data?: Record<string, unknown>;
3
+ done?: boolean;
4
+ };
5
+ export type SseSink = (event: SseEvent) => void;
6
+ import type { AiGatewayScenarioEngine } from './ai-gateway-scenario.js';
7
+ export type AiGatewayRequest = {
8
+ method: string;
9
+ path: string;
10
+ body?: string;
11
+ /** Lower-cased request headers. The AI SDK gateway protocol (/v3/ai) carries its call
12
+ * configuration in headers (`ai-language-model-id`, `ai-language-model-streaming`,
13
+ * `ai-language-model-specification-version`); the /v1 surface ignores them. */
14
+ headers?: Record<string, string>;
15
+ root?: string;
16
+ readOnly?: boolean;
17
+ occurredAt?: string;
18
+ sseSink?: SseSink;
19
+ /** Optional scenario-scripting session (ai-gateway-scenario.ts): when set, POST
20
+ * /v1/chat/completions first consults the scenario's ordered rules and serves a scripted
21
+ * completion on a match (protocol envelope stays vendor-faithful). Test scaffolding — NOT
22
+ * part of the vendor surface and excluded from the capability manifest. */
23
+ scenarioEngine?: AiGatewayScenarioEngine;
24
+ };
25
+ export type AiGatewayResponse = {
26
+ status: number;
27
+ body: unknown;
28
+ };
29
+ export type ChatMessage = {
30
+ role: string;
31
+ content?: unknown;
32
+ tool_calls?: unknown;
33
+ tool_call_id?: unknown;
34
+ cache_control?: unknown;
35
+ };
36
+ /** One provider attempt inside routing metadata (providerMetadata.gateway.routing). */
37
+ export type ProviderAttempt = {
38
+ provider: string;
39
+ providerApiModelId: string;
40
+ credentialType: 'system' | 'byok';
41
+ success: boolean;
42
+ error?: string;
43
+ startTime: number;
44
+ endTime: number;
45
+ };
46
+ export type ModelAttempt = {
47
+ modelId: string;
48
+ canonicalSlug: string;
49
+ success: boolean;
50
+ providerAttemptCount: number;
51
+ providerAttempts: ProviderAttempt[];
52
+ };
53
+ export type GatewayRouting = {
54
+ originalModelId: string;
55
+ resolvedProvider: string;
56
+ resolvedProviderApiModelId: string;
57
+ fallbacksAvailable: string[];
58
+ planningReasoning: string;
59
+ canonicalSlug: string;
60
+ finalProvider: string;
61
+ modelAttemptCount: number;
62
+ modelAttempts: ModelAttempt[];
63
+ totalProviderAttemptCount: number;
64
+ sort?: {
65
+ option: string;
66
+ executionOrder: string[];
67
+ metrics: Record<string, number | null>;
68
+ deprioritizedProviders: string[];
69
+ };
70
+ };
71
+ /** The locally recorded generation row — the shape GET /v1/generation serves (REST API docs,
72
+ * "Look up a generation"), folded into the kernel projection on every chat completion. */
73
+ export type GenerationRecord = {
74
+ id: string;
75
+ total_cost: number;
76
+ upstream_inference_cost: number;
77
+ usage: number;
78
+ created_at: string;
79
+ model: string;
80
+ is_byok: boolean;
81
+ provider_name: string;
82
+ streamed: boolean;
83
+ finish_reason: string;
84
+ latency: number;
85
+ generation_time: number;
86
+ tokens_prompt: number;
87
+ tokens_completion: number;
88
+ native_tokens_prompt: number;
89
+ native_tokens_completion: number;
90
+ native_tokens_reasoning: number;
91
+ native_tokens_cached: number;
92
+ native_tokens_cache_creation: number;
93
+ billable_web_search_calls: number;
94
+ };
@@ -0,0 +1 @@
1
+ export {};
@@ -0,0 +1,5 @@
1
+ import type { AiGatewayRequest, AiGatewayResponse } from './ai-gateway-types.js';
2
+ type ChatFn = (req: AiGatewayRequest, params: Record<string, unknown>) => Promise<AiGatewayResponse>;
3
+ export declare function v3Config(): AiGatewayResponse;
4
+ export declare function handleV3LanguageModel(req: AiGatewayRequest, chat: ChatFn): Promise<AiGatewayResponse>;
5
+ export {};
@@ -0,0 +1,367 @@
1
+ // AI SDK gateway protocol (/v3/ai) — the surface `@ai-sdk/gateway` (and therefore the `ai`
2
+ // package's `createGateway`, the AI SDK's default provider) actually speaks. This is a DIFFERENT
3
+ // wire protocol from the OpenAI-compatible /v1 surface: the SDK's GatewayLanguageModel POSTs the
4
+ // raw LanguageModelV3 call options to `{baseURL}/language-model` with the model id carried in the
5
+ // `ai-language-model-id` header (plus `ai-language-model-specification-version: 3` and
6
+ // `ai-language-model-streaming`), and expects a LanguageModelV3 generate result (content parts /
7
+ // finishReason {unified,raw} / nested usage) back — or an SSE stream of LanguageModelV3 stream
8
+ // parts. `GET {baseURL}/config` serves the model/provider config document `getAvailableModels()`
9
+ // consumes. Protocol derived from the shipped SDK itself (@ai-sdk/gateway dist:
10
+ // gateway-language-model.ts getUrl()/getModelConfigHeaders(), gateway-fetch-metadata.ts
11
+ // gatewayAvailableModelsResponseSchema, and @ai-sdk/provider's LanguageModelV3 types) — there is
12
+ // no separately published wire spec.
13
+ //
14
+ // Design rule: this module is a TRANSLATION layer only. A /v3/ai language-model call is mapped
15
+ // onto the twin's existing deterministic chat machinery (`chatCompletion`, injected as `chat` to
16
+ // avoid an import cycle) so accounting/generation records flow through the SAME kernel fold as
17
+ // /v1 chats — one generation log, one credits/report surface, whichever protocol the consumer
18
+ // speaks.
19
+ import { AI_GATEWAY_CATALOG } from "./ai-gateway-models.js";
20
+ const FIXED_NOW = '1970-01-01T00:00:00.000Z';
21
+ function nowIso(occurredAt) {
22
+ return new Date(Date.parse(occurredAt ?? FIXED_NOW)).toISOString();
23
+ }
24
+ // Same vendor error envelope as /v1 — the SDK's createGatewayErrorFromResponse parses
25
+ // { error: { message, type, param, code } } (param may be an OBJECT for model_not_found).
26
+ function v3Error(status, message, opts = {}) {
27
+ return {
28
+ status,
29
+ body: {
30
+ error: {
31
+ message,
32
+ type: opts.type ?? 'invalid_request_error',
33
+ ...(opts.param !== undefined ? { param: opts.param } : {}),
34
+ ...(opts.code !== undefined ? { code: opts.code } : {}),
35
+ },
36
+ },
37
+ };
38
+ }
39
+ // ── GET /v3/ai/config ─────────────────────────────────────────────────────────────────────
40
+ // The config document `provider.getAvailableModels()` fetches. Shape per the SDK's
41
+ // gatewayAvailableModelsResponseSchema: models[] with id/name/description, decimal-string
42
+ // pricing, a v3 specification block, and modelType.
43
+ export function v3Config() {
44
+ return {
45
+ status: 200,
46
+ body: {
47
+ models: AI_GATEWAY_CATALOG.map(({ model }) => ({
48
+ id: model.id,
49
+ name: model.name,
50
+ description: model.description,
51
+ pricing: {
52
+ input: model.pricing.input,
53
+ output: model.pricing.output ?? '0',
54
+ ...(model.pricing.input_cache_read ? { input_cache_read: model.pricing.input_cache_read } : {}),
55
+ ...(model.pricing.input_cache_write ? { input_cache_write: model.pricing.input_cache_write } : {}),
56
+ },
57
+ specification: {
58
+ specificationVersion: 'v3',
59
+ provider: model.owned_by,
60
+ modelId: model.id,
61
+ },
62
+ modelType: model.type,
63
+ })),
64
+ },
65
+ };
66
+ }
67
+ // ── LanguageModelV3 prompt → OpenAI-compatible messages ──────────────────────────────────
68
+ function toolResultOutputText(output) {
69
+ switch (output.type) {
70
+ case 'text':
71
+ case 'error-text':
72
+ return String(output.value ?? '');
73
+ case 'json':
74
+ case 'error-json':
75
+ return JSON.stringify(output.value ?? null);
76
+ case 'execution-denied':
77
+ return output.reason ? `execution denied: ${String(output.reason)}` : 'execution denied';
78
+ case 'content':
79
+ return (Array.isArray(output.value) ? output.value : [])
80
+ .map((p) => (p?.type === 'text' ? String(p.text ?? '') : ''))
81
+ .filter(Boolean)
82
+ .join(' ');
83
+ default:
84
+ return JSON.stringify(output);
85
+ }
86
+ }
87
+ function filePartToOpenAi(part) {
88
+ const mediaType = typeof part.mediaType === 'string' ? part.mediaType : 'application/octet-stream';
89
+ const data = part.data;
90
+ // Over the wire the SDK sends file data as a string: base64, or a URL (the client's
91
+ // maybeEncodeFileParts converts Uint8Array → data: URL; URL objects JSON-serialize to strings).
92
+ if (typeof data !== 'string' || !data)
93
+ return { error: 'file parts must carry string data (base64 or URL) over the wire' };
94
+ const url = data.startsWith('http://') || data.startsWith('https://') || data.startsWith('data:')
95
+ ? data
96
+ : `data:${mediaType};base64,${data}`;
97
+ if (mediaType.startsWith('image/'))
98
+ return { part: { type: 'image_url', image_url: { url } } };
99
+ return { part: { type: 'file', file: { data: url, media_type: mediaType, ...(typeof part.filename === 'string' ? { filename: part.filename } : {}) } } };
100
+ }
101
+ function translatePrompt(prompt) {
102
+ if (!Array.isArray(prompt) || prompt.length === 0) {
103
+ return { response: v3Error(400, 'prompt must be a non-empty array of LanguageModelV3 messages', { param: 'prompt' }) };
104
+ }
105
+ const messages = [];
106
+ for (const raw of prompt) {
107
+ if (!raw || typeof raw !== 'object')
108
+ return { response: v3Error(400, 'each prompt message must be an object', { param: 'prompt' }) };
109
+ const message = raw;
110
+ const role = message.role;
111
+ if (role === 'system') {
112
+ messages.push({ role: 'system', content: String(message.content ?? '') });
113
+ continue;
114
+ }
115
+ if (role === 'user') {
116
+ const parts = Array.isArray(message.content) ? message.content : [];
117
+ const content = [];
118
+ for (const part of parts) {
119
+ if (part?.type === 'text')
120
+ content.push({ type: 'text', text: String(part.text ?? '') });
121
+ else if (part?.type === 'file') {
122
+ const mapped = filePartToOpenAi(part);
123
+ if ('error' in mapped)
124
+ return { response: v3Error(400, mapped.error, { param: 'prompt' }) };
125
+ content.push(mapped.part);
126
+ }
127
+ else
128
+ return { response: v3Error(400, `unsupported user content part type: ${String(part?.type)}`, { param: 'prompt' }) };
129
+ }
130
+ messages.push({ role: 'user', content });
131
+ continue;
132
+ }
133
+ if (role === 'assistant') {
134
+ const parts = Array.isArray(message.content) ? message.content : [];
135
+ const text = parts.filter((p) => p?.type === 'text').map((p) => String(p.text ?? '')).join('');
136
+ const toolCalls = parts.filter((p) => p?.type === 'tool-call').map((p) => ({
137
+ id: String(p.toolCallId ?? ''),
138
+ type: 'function',
139
+ function: {
140
+ name: String(p.toolName ?? ''),
141
+ arguments: typeof p.input === 'string' ? p.input : JSON.stringify(p.input ?? {}),
142
+ },
143
+ }));
144
+ // Reasoning parts in the history contribute nothing the stub needs; they are dropped
145
+ // (the OpenAI-compat machinery has no assistant-reasoning input field either).
146
+ messages.push({
147
+ role: 'assistant',
148
+ content: text || (toolCalls.length > 0 ? null : text),
149
+ ...(toolCalls.length > 0 ? { tool_calls: toolCalls } : {}),
150
+ });
151
+ continue;
152
+ }
153
+ if (role === 'tool') {
154
+ const parts = Array.isArray(message.content) ? message.content : [];
155
+ for (const part of parts) {
156
+ if (part?.type !== 'tool-result')
157
+ continue; // tool-approval-response: provider-executed tools are not modeled (open todo)
158
+ messages.push({
159
+ role: 'tool',
160
+ tool_call_id: String(part.toolCallId ?? ''),
161
+ content: toolResultOutputText((part.output ?? {})),
162
+ });
163
+ }
164
+ continue;
165
+ }
166
+ return { response: v3Error(400, `unsupported prompt message role: ${String(role)}`, { param: 'prompt' }) };
167
+ }
168
+ return { messages };
169
+ }
170
+ // ── LanguageModelV3 call options → OpenAI-compatible chat params ─────────────────────────
171
+ function translateCallOptions(modelId, streaming, options) {
172
+ const translated = translatePrompt(options.prompt);
173
+ if ('response' in translated)
174
+ return translated;
175
+ const params = { model: modelId, messages: translated.messages, stream: streaming };
176
+ if (options.maxOutputTokens !== undefined)
177
+ params.max_tokens = options.maxOutputTokens;
178
+ if (options.temperature !== undefined)
179
+ params.temperature = options.temperature;
180
+ if (options.topP !== undefined)
181
+ params.top_p = options.topP;
182
+ if (options.frequencyPenalty !== undefined)
183
+ params.frequency_penalty = options.frequencyPenalty;
184
+ if (options.presencePenalty !== undefined)
185
+ params.presence_penalty = options.presencePenalty;
186
+ if (options.stopSequences !== undefined)
187
+ params.stop = options.stopSequences;
188
+ if (options.tools !== undefined) {
189
+ if (!Array.isArray(options.tools))
190
+ return { response: v3Error(400, 'tools must be an array', { param: 'tools' }) };
191
+ const tools = [];
192
+ for (const raw of options.tools) {
193
+ if (raw?.type === 'function') {
194
+ tools.push({
195
+ type: 'function',
196
+ function: {
197
+ name: String(raw.name ?? ''),
198
+ ...(raw.description !== undefined ? { description: raw.description } : {}),
199
+ parameters: raw.inputSchema ?? {},
200
+ },
201
+ });
202
+ }
203
+ else {
204
+ // Provider-executed tools (gateway.parallel_search / gateway.perplexity_search) are not
205
+ // modeled yet — fail loudly rather than silently dropping the tool (open todo:
206
+ // ai-gateway.v3.provider_tools).
207
+ return { response: v3Error(400, `provider-executed tools are not modeled by the twin yet: ${String(raw?.id ?? raw?.name)}`, { param: 'tools', code: 'not_found' }) };
208
+ }
209
+ }
210
+ params.tools = tools;
211
+ }
212
+ const tc = options.toolChoice;
213
+ if (tc !== undefined) {
214
+ if (tc?.type === 'auto' || tc?.type === 'none' || tc?.type === 'required')
215
+ params.tool_choice = tc.type;
216
+ else if (tc?.type === 'tool' && typeof tc.toolName === 'string')
217
+ params.tool_choice = { type: 'function', function: { name: tc.toolName } };
218
+ else
219
+ return { response: v3Error(400, 'toolChoice must be auto/none/required or { type: "tool", toolName }', { param: 'toolChoice' }) };
220
+ }
221
+ const rf = options.responseFormat;
222
+ if (rf !== undefined && rf?.type === 'json') {
223
+ // V3's { type:'json', schema? } maps onto the gateway's legacy json response_format, which
224
+ // the chat machinery already synthesizes schema-conformant output for.
225
+ params.response_format = { type: 'json', ...(rf.schema !== undefined ? { schema: rf.schema } : {}) };
226
+ }
227
+ // providerOptions.gateway (order/only/sort/models/byok/caching) flows through verbatim — the
228
+ // chat machinery already parses exactly this shape.
229
+ if (options.providerOptions !== undefined)
230
+ params.providerOptions = options.providerOptions;
231
+ return { params };
232
+ }
233
+ // ── OpenAI-compatible chat body → LanguageModelV3 generate result ────────────────────────
234
+ function mapFinishReason(raw) {
235
+ const unified = raw === 'tool_calls' ? 'tool-calls' : raw === 'stop' || raw === 'length' ? raw : 'other';
236
+ return { unified, raw };
237
+ }
238
+ function mapUsage(usage) {
239
+ const prompt = Number(usage.prompt_tokens ?? 0);
240
+ const completion = Number(usage.completion_tokens ?? 0);
241
+ const cached = Number(usage.prompt_tokens_details?.cached_tokens ?? 0);
242
+ const reasoning = Number(usage.completion_tokens_details?.reasoning_tokens ?? 0);
243
+ return {
244
+ inputTokens: { total: prompt, noCache: prompt - cached, cacheRead: cached, cacheWrite: 0 },
245
+ outputTokens: { total: completion, text: completion - reasoning, reasoning },
246
+ raw: usage,
247
+ };
248
+ }
249
+ function contentParts(message) {
250
+ const parts = [];
251
+ if (typeof message.reasoning === 'string')
252
+ parts.push({ type: 'reasoning', text: message.reasoning });
253
+ if (Array.isArray(message.tool_calls)) {
254
+ for (const call of message.tool_calls) {
255
+ const fn = (call.function ?? {});
256
+ parts.push({ type: 'tool-call', toolCallId: String(call.id ?? ''), toolName: String(fn.name ?? ''), input: String(fn.arguments ?? '{}') });
257
+ }
258
+ }
259
+ else {
260
+ parts.push({ type: 'text', text: String(message.content ?? '') });
261
+ }
262
+ return parts;
263
+ }
264
+ function v3GenerateBody(chatBody, occurredAt) {
265
+ const choice = chatBody.choices[0];
266
+ const message = choice.message;
267
+ return {
268
+ content: contentParts(message),
269
+ finishReason: mapFinishReason(String(choice.finish_reason)),
270
+ usage: mapUsage((chatBody.usage ?? {})),
271
+ ...(chatBody.providerMetadata !== undefined ? { providerMetadata: chatBody.providerMetadata } : {}),
272
+ response: { id: chatBody.id, modelId: chatBody.model, timestamp: nowIso(occurredAt) },
273
+ warnings: [],
274
+ };
275
+ }
276
+ // SSE stream of LanguageModelV3 stream parts: response-metadata first (the generation id rides
277
+ // on it), reasoning-start/-delta/-end, then text-start/-delta/-end OR
278
+ // tool-input-start/-delta/-end + a tool-call part, a finish part carrying finishReason + usage
279
+ // (+ providerMetadata.gateway), and `data: [DONE]`.
280
+ function streamV3(chatBody, sink, occurredAt) {
281
+ const choice = chatBody.choices[0];
282
+ const message = choice.message;
283
+ sink({ data: { type: 'response-metadata', id: chatBody.id, modelId: chatBody.model, timestamp: nowIso(occurredAt) } });
284
+ if (typeof message.reasoning === 'string') {
285
+ const reasoning = message.reasoning;
286
+ sink({ data: { type: 'reasoning-start', id: 'reasoning-0' } });
287
+ for (let i = 0; i < reasoning.length; i += 48) {
288
+ sink({ data: { type: 'reasoning-delta', id: 'reasoning-0', delta: reasoning.slice(i, i + 48) } });
289
+ }
290
+ sink({ data: { type: 'reasoning-end', id: 'reasoning-0' } });
291
+ }
292
+ if (Array.isArray(message.tool_calls)) {
293
+ for (const call of message.tool_calls) {
294
+ const fn = (call.function ?? {});
295
+ const id = String(call.id ?? '');
296
+ const args = String(fn.arguments ?? '{}');
297
+ sink({ data: { type: 'tool-input-start', id, toolName: String(fn.name ?? '') } });
298
+ sink({ data: { type: 'tool-input-delta', id, delta: args } });
299
+ sink({ data: { type: 'tool-input-end', id } });
300
+ sink({ data: { type: 'tool-call', toolCallId: id, toolName: String(fn.name ?? ''), input: args } });
301
+ }
302
+ }
303
+ else {
304
+ const text = String(message.content ?? '');
305
+ sink({ data: { type: 'text-start', id: 'text-0' } });
306
+ for (let i = 0; i < text.length; i += 24) {
307
+ sink({ data: { type: 'text-delta', id: 'text-0', delta: text.slice(i, i + 24) } });
308
+ }
309
+ sink({ data: { type: 'text-end', id: 'text-0' } });
310
+ }
311
+ sink({
312
+ data: {
313
+ type: 'finish',
314
+ finishReason: mapFinishReason(String(choice.finish_reason)),
315
+ usage: mapUsage((chatBody.usage ?? {})),
316
+ ...(chatBody.providerMetadata !== undefined ? { providerMetadata: chatBody.providerMetadata } : {}),
317
+ },
318
+ });
319
+ sink({ done: true });
320
+ }
321
+ // ── POST /v3/ai/language-model ────────────────────────────────────────────────────────────
322
+ export async function handleV3LanguageModel(req, chat) {
323
+ const headers = req.headers ?? {};
324
+ const modelId = headers['ai-language-model-id'];
325
+ if (!modelId) {
326
+ return v3Error(400, "Invalid request: missing required header 'ai-language-model-id'", { param: 'ai-language-model-id', code: 'missing_parameter' });
327
+ }
328
+ const specVersion = headers['ai-language-model-specification-version'];
329
+ if (specVersion !== undefined && specVersion !== '3') {
330
+ return v3Error(400, `Unsupported ai-language-model-specification-version: ${specVersion} (the twin models specification version 3)`, { param: 'ai-language-model-specification-version' });
331
+ }
332
+ const streaming = headers['ai-language-model-streaming'] === 'true';
333
+ let options = {};
334
+ if (req.body?.trim()) {
335
+ try {
336
+ const parsed = JSON.parse(req.body);
337
+ if (parsed && typeof parsed === 'object' && !Array.isArray(parsed))
338
+ options = parsed;
339
+ }
340
+ catch {
341
+ return v3Error(400, 'request body must be a JSON object of LanguageModelV3 call options');
342
+ }
343
+ }
344
+ const translated = translateCallOptions(modelId, streaming, options);
345
+ if ('response' in translated)
346
+ return translated.response;
347
+ // Run the SAME chat machinery/kernel fold as /v1 (validation, routing, scenario scripting,
348
+ // generation record, credits) — but WITHOUT the OpenAI SSE sink: the v3 stream shape is
349
+ // emitted below from the unary body.
350
+ const { sseSink: _sseSink, ...chatReq } = req;
351
+ const result = await chat(chatReq, translated.params);
352
+ if (result.status !== 200) {
353
+ const errorBody = result.body;
354
+ const err = (errorBody?.error ?? {});
355
+ // The /v1 surface reports unknown models OpenAI-style (type invalid_request_error, code
356
+ // model_not_found); the AI SDK protocol's typed errors key on error.type, with the model id
357
+ // in an OBJECT param — remap so the SDK surfaces GatewayModelNotFoundError.
358
+ if (result.status === 404 && err.code === 'model_not_found') {
359
+ return v3Error(404, String(err.message ?? 'Model not found'), { type: 'model_not_found', param: { modelId } });
360
+ }
361
+ return result; // same { error: { message, type, param, code } } envelope the SDK parses
362
+ }
363
+ const chatBody = result.body;
364
+ if (streaming && req.sseSink)
365
+ streamV3(chatBody, req.sseSink, req.occurredAt);
366
+ return { status: 200, body: v3GenerateBody(chatBody, req.occurredAt) };
367
+ }
@@ -0,0 +1,2 @@
1
+ #!/usr/bin/env node
2
+ export {};
@@ -0,0 +1,24 @@
1
+ #!/usr/bin/env node
2
+ import { keepProcessAlive } from '@volter/world-core/lifecycle';
3
+ import { hasFlag, optionValue } from '@volter/world-core/args';
4
+ import { createAiGatewayTwinServer } from "./ai-gateway-server.js";
5
+ const [cmd, ...rest] = process.argv.slice(2);
6
+ const port = Number(optionValue(rest, '--port', '0')) || undefined;
7
+ const root = optionValue(rest, '--root') || undefined;
8
+ const readOnly = hasFlag(rest, '--read-only');
9
+ const scenario = optionValue(rest, '--scenario') || undefined; // scripted completions (JSON scenario file)
10
+ if (cmd === 'serve') {
11
+ const server = await createAiGatewayTwinServer({ readOnly, ...(root ? { root } : {}), ...(port ? { port } : {}), ...(scenario ? { scenarioPath: scenario } : {}) });
12
+ process.stdout.write(`ai-gateway twin (deterministic stub output)${readOnly ? ' [read-only]' : ''}${scenario ? ` [scenario: ${scenario}]` : ''} at http://127.0.0.1:${server.port}\n`);
13
+ await keepProcessAlive();
14
+ }
15
+ else if (cmd === 'conformance') {
16
+ const { checkAiGatewayConformance } = await import("./ai-gateway-conformance.js"); // lazy — dev-only
17
+ const report = await checkAiGatewayConformance({ ...(root ? { root } : {}) });
18
+ process.stdout.write(`${JSON.stringify(report, null, 2)}\n`);
19
+ if (!report.ok)
20
+ process.exitCode = 1;
21
+ }
22
+ else {
23
+ process.stdout.write('Usage: world-ai-gateway serve|conformance [--port N] [--root DIR] [--read-only] [--scenario FILE]\n');
24
+ }
@@ -0,0 +1,12 @@
1
+ export { handleAiGatewayTwinRequest } from './ai-gateway-twin.js';
2
+ export type { AiGatewayRequest, AiGatewayResponse, SseEvent, SseSink, GenerationRecord, GatewayRouting, ModelAttempt, ProviderAttempt } from './ai-gateway-types.js';
3
+ export { createAiGatewayTwinFetch, createAiGatewayTwinServer, type AiGatewayTwinFetchOptions } from './ai-gateway-server.js';
4
+ export { AI_GATEWAY_CATALOG, AI_GATEWAY_MODELS, findAiGatewayModel } from './ai-gateway-models.js';
5
+ export type { AiGatewayModel, AiGatewayCatalogEntry } from './ai-gateway-models.js';
6
+ export { contentToText, countPromptTokens, estimateTokens, lastUserText, stableHash, stubAssistantText, synthesizeJsonSchemaValue } from './ai-gateway-stub.js';
7
+ export { aiGatewayScenarioAdapter, createAiGatewayScenarioEngine, loadAiGatewayScenarioDocument, realizeAiGatewayRespond } from './ai-gateway-scenario.js';
8
+ export type { AiGatewayScenarioEngine, AiGatewayScenarioRequest, AiGatewayScenarioRespond, ScenarioToolCall, ScriptedResult } from './ai-gateway-scenario.js';
9
+ export { aiGatewayRequestForAction, pullAiGatewayState, pushAiGatewayAction, syncAiGatewayFromReal, } from './ai-gateway-connector.js';
10
+ export type { AiGatewayExecute } from './ai-gateway-connector.js';
11
+ import { type TwinPack } from '@volter/world-core';
12
+ export declare const pack: TwinPack;
@@ -0,0 +1,54 @@
1
+ export { handleAiGatewayTwinRequest } from "./ai-gateway-twin.js";
2
+ export { createAiGatewayTwinFetch, createAiGatewayTwinServer } from "./ai-gateway-server.js";
3
+ export { AI_GATEWAY_CATALOG, AI_GATEWAY_MODELS, findAiGatewayModel } from "./ai-gateway-models.js";
4
+ export { contentToText, countPromptTokens, estimateTokens, lastUserText, stableHash, stubAssistantText, synthesizeJsonSchemaValue } from "./ai-gateway-stub.js";
5
+ export { aiGatewayScenarioAdapter, createAiGatewayScenarioEngine, loadAiGatewayScenarioDocument, realizeAiGatewayRespond } from "./ai-gateway-scenario.js";
6
+ export { aiGatewayRequestForAction, pullAiGatewayState, pushAiGatewayAction, syncAiGatewayFromReal, } from "./ai-gateway-connector.js";
7
+ import { registerPack } from '@volter/world-core';
8
+ import { performAiGatewayAction, syncAiGatewayFromRemote } from "./ai-gateway-connector.js";
9
+ export const pack = {
10
+ // PROTOCOL 2 (docs/contributing/architecture.md#protocol-2-the-pack-is-a-plugin): the pack is a plugin — its wire, its tree, and its half of the real
11
+ // state system. Moved 2026-09-08. The gateway has NO write API: inference is the only "write", and
12
+ // replaying one would spend real money — so a recorded generation is RECONCILED against the gateway
13
+ // rather than written, and everything else settles with that reason.
14
+ protocol: '2',
15
+ refresh: { every: '15m', onDemand: { atMost: '60s' } },
16
+ stateSystem: { perform: performAiGatewayAction, refresh: syncAiGatewayFromRemote },
17
+ // the round trip: a chat completion — the twin records the generation it answered, which is the row a
18
+ // caller makes here, and a fresh one each time
19
+ roundTrip: { method: 'POST', path: '/v1/chat/completions', body: { model: 'openai/gpt-5', messages: [{ role: 'user', content: 'round trip' }] }, headers: { authorization: 'Bearer round-trip' } },
20
+ parityOrigin: 'http://twin',
21
+ vendor: 'ai-gateway',
22
+ transport: 'rest',
23
+ archetype: 'generative',
24
+ bin: 'world-ai-gateway',
25
+ // The subject types the twin SERVES from its own state — the R2 resource-level claim.
26
+ // `model` and `credits` are deliberately NOT here: both are connector-side projections
27
+ // that only `pullAiGatewayState` / `syncAiGatewayFromReal` fold in from a real gateway
28
+ // through an injected client, and no route reads either back — GET /v1/models serves the
29
+ // static AI_GATEWAY_MODELS catalog and GET /v1/credits a balance DERIVED from the
30
+ // generation rows, neither from a stored subject.
31
+ resources: ['generation'],
32
+ specSource: 'Vercel AI Gateway docs (OpenAI-compatible Chat Completions/Embeddings + REST API Reference: models/credits/generation/report) + the shipped @ai-sdk/gateway SDK for the /v3/ai AI SDK gateway protocol (config + language-model); model output is a deterministic stub',
33
+ description: 'Vercel AI Gateway twin — OpenAI-compatible chat completions/streaming with faithful provider-routing metadata, the AI SDK gateway protocol (/v3/ai: config + language-model for `createGateway`), models catalog with per-provider endpoints, billed embeddings, credits/generation/report surfaces; generative output is a labeled deterministic stub.',
34
+ // The gateway serves TWO API prefixes on one host: '/v1/' (OpenAI-compat + REST reference)
35
+ // and '/v3/ai' (the AI SDK gateway protocol `createGateway` speaks). `apiPathPrefix` is a
36
+ // single startsWith prefix (control-plane VendorRoute), so '/v' is the narrowest value that
37
+ // routes BOTH surfaces to the twin.
38
+ browserRouting: { apiPathPrefix: '/v', loaderHost: 'https://ai-gateway.vercel.sh' },
39
+ // INTERCEPTION RULING — hostsNone, the pack's own home for it: SDK is base-URL-configured
40
+ // (@ai-sdk/gateway baseURL / AI_GATEWAY_BASE_URL) — worlds wire the twin through explicit
41
+ // endpoint config; no injector entry grounded yet.
42
+ hostsNone: "SDK is base-URL-configured (@ai-sdk/gateway baseURL / AI_GATEWAY_BASE_URL) — worlds wire the twin through explicit endpoint config; no injector entry grounded yet",
43
+ // Adoption, moved off the central maps unchanged (descriptor-first back-migration, adding-a-twin.md §3,
44
+ // 2026-08-31).
45
+ adoption: {
46
+ // No Python distribution: Vercel's AI Gateway is addressed through the TypeScript AI SDK
47
+ // (`@ai-sdk/gateway`); a Python caller points an OpenAI-compatible client at the gateway base
48
+ // URL, and that distribution (`openai`) belongs to the openai pack.
49
+ pypi: [],
50
+ sdks: ['@ai-sdk/gateway'], envStems: ['AIGATEWAY'],
51
+ },
52
+ };
53
+ // registered at import: the kernel learns the pack's state system (protocol 2)
54
+ registerPack(pack);
package/package.json ADDED
@@ -0,0 +1,66 @@
1
+ {
2
+ "name": "@volter/twin-ai-gateway",
3
+ "version": "0.1.0",
4
+ "description": "Local Vercel AI Gateway twin — a faithful, stateful local AI Gateway (ai-gateway.vercel.sh) your SDK talks to unmodified, over both wire protocols: the AI SDK gateway protocol at /v3/ai (createGateway / @ai-sdk/gateway) and the OpenAI-compatible /v1 surface. Chat completions/streaming/tool calls return deterministic labeled stubs with faithful provider-routing metadata; models, credits, generation lookup, and spend reporting are local. Built on @volter/world-core.",
5
+ "keywords": [
6
+ "twin",
7
+ "local",
8
+ "mock",
9
+ "mirror",
10
+ "simulator",
11
+ "sdk",
12
+ "api",
13
+ "vercel",
14
+ "ai-gateway",
15
+ "llm"
16
+ ],
17
+ "author": "Volter (https://github.com/volter-ai)",
18
+ "license": "Apache-2.0",
19
+ "files": [
20
+ "src",
21
+ "test-fixtures",
22
+ "README.md",
23
+ "LICENSE",
24
+ "!**/*.test.ts",
25
+ "!**/*.test.tsx",
26
+ "dist"
27
+ ],
28
+ "repository": {
29
+ "type": "git",
30
+ "url": "git+https://github.com/volter-ai/twin.git",
31
+ "directory": "packages/twin/ai-gateway"
32
+ },
33
+ "homepage": "https://github.com/volter-ai/twin/tree/main/packages/twin/ai-gateway#readme",
34
+ "type": "module",
35
+ "exports": {
36
+ ".": {
37
+ "types": "./dist/src/index.d.ts",
38
+ "default": "./dist/src/index.js"
39
+ }
40
+ },
41
+ "bin": {
42
+ "world-ai-gateway": "dist/src/cli.js"
43
+ },
44
+ "scripts": {
45
+ "test": "bun test src/*.test.ts",
46
+ "typecheck": "tsc --noEmit",
47
+ "build": "node ../../../scripts/publish/build.mjs",
48
+ "prepack": "node ../../../scripts/publish/prepare-publish.mjs prepack",
49
+ "postpack": "node ../../../scripts/publish/prepare-publish.mjs postpack"
50
+ },
51
+ "peerDependencies": {
52
+ "@volter/world-core": "2.0.0"
53
+ },
54
+ "devDependencies": {
55
+ "@volter/world-core": "2.0.0",
56
+ "@volter/world-tooling": "0.1.0",
57
+ "@ai-sdk/openai": "^3.0.29",
58
+ "ai": "^6.0.86",
59
+ "@types/bun": "^1.2.20",
60
+ "@types/node": "^24.0.0",
61
+ "typescript": "^5.9.0"
62
+ },
63
+ "engines": {
64
+ "node": ">=22.3"
65
+ }
66
+ }