@volter/twin-ai-gateway 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +202 -0
- package/README.md +176 -0
- package/dist/src/ai-gateway-capabilities.d.ts +4 -0
- package/dist/src/ai-gateway-capabilities.js +772 -0
- package/dist/src/ai-gateway-conformance.d.ts +11 -0
- package/dist/src/ai-gateway-conformance.js +72 -0
- package/dist/src/ai-gateway-connector.d.ts +50 -0
- package/dist/src/ai-gateway-connector.js +97 -0
- package/dist/src/ai-gateway-models.d.ts +27 -0
- package/dist/src/ai-gateway-models.js +65 -0
- package/dist/src/ai-gateway-perform-harness.d.ts +5 -0
- package/dist/src/ai-gateway-perform-harness.js +17 -0
- package/dist/src/ai-gateway-scenario.d.ts +36 -0
- package/dist/src/ai-gateway-scenario.js +125 -0
- package/dist/src/ai-gateway-server.d.ts +16 -0
- package/dist/src/ai-gateway-server.js +107 -0
- package/dist/src/ai-gateway-stub.d.ts +20 -0
- package/dist/src/ai-gateway-stub.js +124 -0
- package/dist/src/ai-gateway-twin.d.ts +2 -0
- package/dist/src/ai-gateway-twin.js +994 -0
- package/dist/src/ai-gateway-types.d.ts +94 -0
- package/dist/src/ai-gateway-types.js +1 -0
- package/dist/src/ai-gateway-v3.d.ts +5 -0
- package/dist/src/ai-gateway-v3.js +367 -0
- package/dist/src/cli.d.ts +2 -0
- package/dist/src/cli.js +24 -0
- package/dist/src/index.d.ts +12 -0
- package/dist/src/index.js +54 -0
- package/package.json +66 -0
- package/src/ai-gateway-capabilities.ts +894 -0
- package/src/ai-gateway-conformance.ts +77 -0
- package/src/ai-gateway-connector.ts +100 -0
- package/src/ai-gateway-models.ts +105 -0
- package/src/ai-gateway-perform-harness.ts +17 -0
- package/src/ai-gateway-scenario.ts +137 -0
- package/src/ai-gateway-server.ts +122 -0
- package/src/ai-gateway-stub.ts +115 -0
- package/src/ai-gateway-twin.ts +1068 -0
- package/src/ai-gateway-types.ts +96 -0
- package/src/ai-gateway-v3.ts +372 -0
- package/src/cli.ts +23 -0
- package/src/index.ts +67 -0
|
@@ -0,0 +1,96 @@
|
|
|
1
|
+
export type SseEvent = { data?: Record<string, unknown>; done?: boolean };
|
|
2
|
+
export type SseSink = (event: SseEvent) => void;
|
|
3
|
+
|
|
4
|
+
import type { AiGatewayScenarioEngine } from './ai-gateway-scenario.ts';
|
|
5
|
+
|
|
6
|
+
export type AiGatewayRequest = {
|
|
7
|
+
method: string;
|
|
8
|
+
path: string;
|
|
9
|
+
body?: string;
|
|
10
|
+
/** Lower-cased request headers. The AI SDK gateway protocol (/v3/ai) carries its call
|
|
11
|
+
* configuration in headers (`ai-language-model-id`, `ai-language-model-streaming`,
|
|
12
|
+
* `ai-language-model-specification-version`); the /v1 surface ignores them. */
|
|
13
|
+
headers?: Record<string, string>;
|
|
14
|
+
root?: string;
|
|
15
|
+
readOnly?: boolean;
|
|
16
|
+
occurredAt?: string;
|
|
17
|
+
sseSink?: SseSink;
|
|
18
|
+
/** Optional scenario-scripting session (ai-gateway-scenario.ts): when set, POST
|
|
19
|
+
* /v1/chat/completions first consults the scenario's ordered rules and serves a scripted
|
|
20
|
+
* completion on a match (protocol envelope stays vendor-faithful). Test scaffolding — NOT
|
|
21
|
+
* part of the vendor surface and excluded from the capability manifest. */
|
|
22
|
+
scenarioEngine?: AiGatewayScenarioEngine;
|
|
23
|
+
};
|
|
24
|
+
|
|
25
|
+
export type AiGatewayResponse = { status: number; body: unknown };
|
|
26
|
+
|
|
27
|
+
export type ChatMessage = {
|
|
28
|
+
role: string;
|
|
29
|
+
content?: unknown;
|
|
30
|
+
tool_calls?: unknown;
|
|
31
|
+
tool_call_id?: unknown;
|
|
32
|
+
cache_control?: unknown;
|
|
33
|
+
};
|
|
34
|
+
|
|
35
|
+
/** One provider attempt inside routing metadata (providerMetadata.gateway.routing). */
|
|
36
|
+
export type ProviderAttempt = {
|
|
37
|
+
provider: string;
|
|
38
|
+
providerApiModelId: string;
|
|
39
|
+
credentialType: 'system' | 'byok';
|
|
40
|
+
success: boolean;
|
|
41
|
+
error?: string;
|
|
42
|
+
startTime: number;
|
|
43
|
+
endTime: number;
|
|
44
|
+
};
|
|
45
|
+
|
|
46
|
+
export type ModelAttempt = {
|
|
47
|
+
modelId: string;
|
|
48
|
+
canonicalSlug: string;
|
|
49
|
+
success: boolean;
|
|
50
|
+
providerAttemptCount: number;
|
|
51
|
+
providerAttempts: ProviderAttempt[];
|
|
52
|
+
};
|
|
53
|
+
|
|
54
|
+
export type GatewayRouting = {
|
|
55
|
+
originalModelId: string;
|
|
56
|
+
resolvedProvider: string;
|
|
57
|
+
resolvedProviderApiModelId: string;
|
|
58
|
+
fallbacksAvailable: string[];
|
|
59
|
+
planningReasoning: string;
|
|
60
|
+
canonicalSlug: string;
|
|
61
|
+
finalProvider: string;
|
|
62
|
+
modelAttemptCount: number;
|
|
63
|
+
modelAttempts: ModelAttempt[];
|
|
64
|
+
totalProviderAttemptCount: number;
|
|
65
|
+
sort?: {
|
|
66
|
+
option: string;
|
|
67
|
+
executionOrder: string[];
|
|
68
|
+
metrics: Record<string, number | null>;
|
|
69
|
+
deprioritizedProviders: string[];
|
|
70
|
+
};
|
|
71
|
+
};
|
|
72
|
+
|
|
73
|
+
/** The locally recorded generation row — the shape GET /v1/generation serves (REST API docs,
|
|
74
|
+
* "Look up a generation"), folded into the kernel projection on every chat completion. */
|
|
75
|
+
export type GenerationRecord = {
|
|
76
|
+
id: string;
|
|
77
|
+
total_cost: number;
|
|
78
|
+
upstream_inference_cost: number;
|
|
79
|
+
usage: number;
|
|
80
|
+
created_at: string;
|
|
81
|
+
model: string;
|
|
82
|
+
is_byok: boolean;
|
|
83
|
+
provider_name: string;
|
|
84
|
+
streamed: boolean;
|
|
85
|
+
finish_reason: string;
|
|
86
|
+
latency: number;
|
|
87
|
+
generation_time: number;
|
|
88
|
+
tokens_prompt: number;
|
|
89
|
+
tokens_completion: number;
|
|
90
|
+
native_tokens_prompt: number;
|
|
91
|
+
native_tokens_completion: number;
|
|
92
|
+
native_tokens_reasoning: number;
|
|
93
|
+
native_tokens_cached: number;
|
|
94
|
+
native_tokens_cache_creation: number;
|
|
95
|
+
billable_web_search_calls: number;
|
|
96
|
+
};
|
|
@@ -0,0 +1,372 @@
|
|
|
1
|
+
// AI SDK gateway protocol (/v3/ai) — the surface `@ai-sdk/gateway` (and therefore the `ai`
|
|
2
|
+
// package's `createGateway`, the AI SDK's default provider) actually speaks. This is a DIFFERENT
|
|
3
|
+
// wire protocol from the OpenAI-compatible /v1 surface: the SDK's GatewayLanguageModel POSTs the
|
|
4
|
+
// raw LanguageModelV3 call options to `{baseURL}/language-model` with the model id carried in the
|
|
5
|
+
// `ai-language-model-id` header (plus `ai-language-model-specification-version: 3` and
|
|
6
|
+
// `ai-language-model-streaming`), and expects a LanguageModelV3 generate result (content parts /
|
|
7
|
+
// finishReason {unified,raw} / nested usage) back — or an SSE stream of LanguageModelV3 stream
|
|
8
|
+
// parts. `GET {baseURL}/config` serves the model/provider config document `getAvailableModels()`
|
|
9
|
+
// consumes. Protocol derived from the shipped SDK itself (@ai-sdk/gateway dist:
|
|
10
|
+
// gateway-language-model.ts getUrl()/getModelConfigHeaders(), gateway-fetch-metadata.ts
|
|
11
|
+
// gatewayAvailableModelsResponseSchema, and @ai-sdk/provider's LanguageModelV3 types) — there is
|
|
12
|
+
// no separately published wire spec.
|
|
13
|
+
//
|
|
14
|
+
// Design rule: this module is a TRANSLATION layer only. A /v3/ai language-model call is mapped
|
|
15
|
+
// onto the twin's existing deterministic chat machinery (`chatCompletion`, injected as `chat` to
|
|
16
|
+
// avoid an import cycle) so accounting/generation records flow through the SAME kernel fold as
|
|
17
|
+
// /v1 chats — one generation log, one credits/report surface, whichever protocol the consumer
|
|
18
|
+
// speaks.
|
|
19
|
+
import { AI_GATEWAY_CATALOG } from './ai-gateway-models.ts';
|
|
20
|
+
import type { AiGatewayRequest, AiGatewayResponse, ChatMessage, SseEvent } from './ai-gateway-types.ts';
|
|
21
|
+
|
|
22
|
+
type Body = Record<string, unknown>;
|
|
23
|
+
type ChatFn = (req: AiGatewayRequest, params: Record<string, unknown>) => Promise<AiGatewayResponse>;
|
|
24
|
+
|
|
25
|
+
const FIXED_NOW = '1970-01-01T00:00:00.000Z';
|
|
26
|
+
|
|
27
|
+
function nowIso(occurredAt?: string): string {
|
|
28
|
+
return new Date(Date.parse(occurredAt ?? FIXED_NOW)).toISOString();
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
// Same vendor error envelope as /v1 — the SDK's createGatewayErrorFromResponse parses
|
|
32
|
+
// { error: { message, type, param, code } } (param may be an OBJECT for model_not_found).
|
|
33
|
+
function v3Error(status: number, message: string, opts: { type?: string; param?: unknown; code?: string } = {}): AiGatewayResponse {
|
|
34
|
+
return {
|
|
35
|
+
status,
|
|
36
|
+
body: {
|
|
37
|
+
error: {
|
|
38
|
+
message,
|
|
39
|
+
type: opts.type ?? 'invalid_request_error',
|
|
40
|
+
...(opts.param !== undefined ? { param: opts.param } : {}),
|
|
41
|
+
...(opts.code !== undefined ? { code: opts.code } : {}),
|
|
42
|
+
},
|
|
43
|
+
},
|
|
44
|
+
};
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
// ── GET /v3/ai/config ─────────────────────────────────────────────────────────────────────
|
|
48
|
+
// The config document `provider.getAvailableModels()` fetches. Shape per the SDK's
|
|
49
|
+
// gatewayAvailableModelsResponseSchema: models[] with id/name/description, decimal-string
|
|
50
|
+
// pricing, a v3 specification block, and modelType.
|
|
51
|
+
export function v3Config(): AiGatewayResponse {
|
|
52
|
+
return {
|
|
53
|
+
status: 200,
|
|
54
|
+
body: {
|
|
55
|
+
models: AI_GATEWAY_CATALOG.map(({ model }) => ({
|
|
56
|
+
id: model.id,
|
|
57
|
+
name: model.name,
|
|
58
|
+
description: model.description,
|
|
59
|
+
pricing: {
|
|
60
|
+
input: model.pricing.input,
|
|
61
|
+
output: model.pricing.output ?? '0',
|
|
62
|
+
...(model.pricing.input_cache_read ? { input_cache_read: model.pricing.input_cache_read } : {}),
|
|
63
|
+
...(model.pricing.input_cache_write ? { input_cache_write: model.pricing.input_cache_write } : {}),
|
|
64
|
+
},
|
|
65
|
+
specification: {
|
|
66
|
+
specificationVersion: 'v3',
|
|
67
|
+
provider: model.owned_by,
|
|
68
|
+
modelId: model.id,
|
|
69
|
+
},
|
|
70
|
+
modelType: model.type,
|
|
71
|
+
})),
|
|
72
|
+
},
|
|
73
|
+
};
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
// ── LanguageModelV3 prompt → OpenAI-compatible messages ──────────────────────────────────
|
|
77
|
+
|
|
78
|
+
function toolResultOutputText(output: Body): string {
|
|
79
|
+
switch (output.type) {
|
|
80
|
+
case 'text':
|
|
81
|
+
case 'error-text':
|
|
82
|
+
return String(output.value ?? '');
|
|
83
|
+
case 'json':
|
|
84
|
+
case 'error-json':
|
|
85
|
+
return JSON.stringify(output.value ?? null);
|
|
86
|
+
case 'execution-denied':
|
|
87
|
+
return output.reason ? `execution denied: ${String(output.reason)}` : 'execution denied';
|
|
88
|
+
case 'content':
|
|
89
|
+
return (Array.isArray(output.value) ? output.value as Body[] : [])
|
|
90
|
+
.map((p) => (p?.type === 'text' ? String(p.text ?? '') : ''))
|
|
91
|
+
.filter(Boolean)
|
|
92
|
+
.join(' ');
|
|
93
|
+
default:
|
|
94
|
+
return JSON.stringify(output);
|
|
95
|
+
}
|
|
96
|
+
}
|
|
97
|
+
|
|
98
|
+
function filePartToOpenAi(part: Body): { part: Record<string, unknown> } | { error: string } {
|
|
99
|
+
const mediaType = typeof part.mediaType === 'string' ? part.mediaType : 'application/octet-stream';
|
|
100
|
+
const data = part.data;
|
|
101
|
+
// Over the wire the SDK sends file data as a string: base64, or a URL (the client's
|
|
102
|
+
// maybeEncodeFileParts converts Uint8Array → data: URL; URL objects JSON-serialize to strings).
|
|
103
|
+
if (typeof data !== 'string' || !data) return { error: 'file parts must carry string data (base64 or URL) over the wire' };
|
|
104
|
+
const url = data.startsWith('http://') || data.startsWith('https://') || data.startsWith('data:')
|
|
105
|
+
? data
|
|
106
|
+
: `data:${mediaType};base64,${data}`;
|
|
107
|
+
if (mediaType.startsWith('image/')) return { part: { type: 'image_url', image_url: { url } } };
|
|
108
|
+
return { part: { type: 'file', file: { data: url, media_type: mediaType, ...(typeof part.filename === 'string' ? { filename: part.filename } : {}) } } };
|
|
109
|
+
}
|
|
110
|
+
|
|
111
|
+
function translatePrompt(prompt: unknown): { messages: ChatMessage[] } | { response: AiGatewayResponse } {
|
|
112
|
+
if (!Array.isArray(prompt) || prompt.length === 0) {
|
|
113
|
+
return { response: v3Error(400, 'prompt must be a non-empty array of LanguageModelV3 messages', { param: 'prompt' }) };
|
|
114
|
+
}
|
|
115
|
+
const messages: ChatMessage[] = [];
|
|
116
|
+
for (const raw of prompt) {
|
|
117
|
+
if (!raw || typeof raw !== 'object') return { response: v3Error(400, 'each prompt message must be an object', { param: 'prompt' }) };
|
|
118
|
+
const message = raw as Body;
|
|
119
|
+
const role = message.role;
|
|
120
|
+
if (role === 'system') {
|
|
121
|
+
messages.push({ role: 'system', content: String(message.content ?? '') });
|
|
122
|
+
continue;
|
|
123
|
+
}
|
|
124
|
+
if (role === 'user') {
|
|
125
|
+
const parts = Array.isArray(message.content) ? message.content as Body[] : [];
|
|
126
|
+
const content: Record<string, unknown>[] = [];
|
|
127
|
+
for (const part of parts) {
|
|
128
|
+
if (part?.type === 'text') content.push({ type: 'text', text: String(part.text ?? '') });
|
|
129
|
+
else if (part?.type === 'file') {
|
|
130
|
+
const mapped = filePartToOpenAi(part);
|
|
131
|
+
if ('error' in mapped) return { response: v3Error(400, mapped.error, { param: 'prompt' }) };
|
|
132
|
+
content.push(mapped.part);
|
|
133
|
+
} else return { response: v3Error(400, `unsupported user content part type: ${String(part?.type)}`, { param: 'prompt' }) };
|
|
134
|
+
}
|
|
135
|
+
messages.push({ role: 'user', content });
|
|
136
|
+
continue;
|
|
137
|
+
}
|
|
138
|
+
if (role === 'assistant') {
|
|
139
|
+
const parts = Array.isArray(message.content) ? message.content as Body[] : [];
|
|
140
|
+
const text = parts.filter((p) => p?.type === 'text').map((p) => String(p.text ?? '')).join('');
|
|
141
|
+
const toolCalls = parts.filter((p) => p?.type === 'tool-call').map((p) => ({
|
|
142
|
+
id: String(p.toolCallId ?? ''),
|
|
143
|
+
type: 'function',
|
|
144
|
+
function: {
|
|
145
|
+
name: String(p.toolName ?? ''),
|
|
146
|
+
arguments: typeof p.input === 'string' ? p.input : JSON.stringify(p.input ?? {}),
|
|
147
|
+
},
|
|
148
|
+
}));
|
|
149
|
+
// Reasoning parts in the history contribute nothing the stub needs; they are dropped
|
|
150
|
+
// (the OpenAI-compat machinery has no assistant-reasoning input field either).
|
|
151
|
+
messages.push({
|
|
152
|
+
role: 'assistant',
|
|
153
|
+
content: text || (toolCalls.length > 0 ? null : text),
|
|
154
|
+
...(toolCalls.length > 0 ? { tool_calls: toolCalls } : {}),
|
|
155
|
+
});
|
|
156
|
+
continue;
|
|
157
|
+
}
|
|
158
|
+
if (role === 'tool') {
|
|
159
|
+
const parts = Array.isArray(message.content) ? message.content as Body[] : [];
|
|
160
|
+
for (const part of parts) {
|
|
161
|
+
if (part?.type !== 'tool-result') continue; // tool-approval-response: provider-executed tools are not modeled (open todo)
|
|
162
|
+
messages.push({
|
|
163
|
+
role: 'tool',
|
|
164
|
+
tool_call_id: String(part.toolCallId ?? ''),
|
|
165
|
+
content: toolResultOutputText((part.output ?? {}) as Body),
|
|
166
|
+
});
|
|
167
|
+
}
|
|
168
|
+
continue;
|
|
169
|
+
}
|
|
170
|
+
return { response: v3Error(400, `unsupported prompt message role: ${String(role)}`, { param: 'prompt' }) };
|
|
171
|
+
}
|
|
172
|
+
return { messages };
|
|
173
|
+
}
|
|
174
|
+
|
|
175
|
+
// ── LanguageModelV3 call options → OpenAI-compatible chat params ─────────────────────────
|
|
176
|
+
|
|
177
|
+
function translateCallOptions(modelId: string, streaming: boolean, options: Body): { params: Record<string, unknown> } | { response: AiGatewayResponse } {
|
|
178
|
+
const translated = translatePrompt(options.prompt);
|
|
179
|
+
if ('response' in translated) return translated;
|
|
180
|
+
|
|
181
|
+
const params: Record<string, unknown> = { model: modelId, messages: translated.messages, stream: streaming };
|
|
182
|
+
|
|
183
|
+
if (options.maxOutputTokens !== undefined) params.max_tokens = options.maxOutputTokens;
|
|
184
|
+
if (options.temperature !== undefined) params.temperature = options.temperature;
|
|
185
|
+
if (options.topP !== undefined) params.top_p = options.topP;
|
|
186
|
+
if (options.frequencyPenalty !== undefined) params.frequency_penalty = options.frequencyPenalty;
|
|
187
|
+
if (options.presencePenalty !== undefined) params.presence_penalty = options.presencePenalty;
|
|
188
|
+
if (options.stopSequences !== undefined) params.stop = options.stopSequences;
|
|
189
|
+
|
|
190
|
+
if (options.tools !== undefined) {
|
|
191
|
+
if (!Array.isArray(options.tools)) return { response: v3Error(400, 'tools must be an array', { param: 'tools' }) };
|
|
192
|
+
const tools: Record<string, unknown>[] = [];
|
|
193
|
+
for (const raw of options.tools as Body[]) {
|
|
194
|
+
if (raw?.type === 'function') {
|
|
195
|
+
tools.push({
|
|
196
|
+
type: 'function',
|
|
197
|
+
function: {
|
|
198
|
+
name: String(raw.name ?? ''),
|
|
199
|
+
...(raw.description !== undefined ? { description: raw.description } : {}),
|
|
200
|
+
parameters: raw.inputSchema ?? {},
|
|
201
|
+
},
|
|
202
|
+
});
|
|
203
|
+
} else {
|
|
204
|
+
// Provider-executed tools (gateway.parallel_search / gateway.perplexity_search) are not
|
|
205
|
+
// modeled yet — fail loudly rather than silently dropping the tool (open todo:
|
|
206
|
+
// ai-gateway.v3.provider_tools).
|
|
207
|
+
return { response: v3Error(400, `provider-executed tools are not modeled by the twin yet: ${String((raw as Body)?.id ?? (raw as Body)?.name)}`, { param: 'tools', code: 'not_found' }) };
|
|
208
|
+
}
|
|
209
|
+
}
|
|
210
|
+
params.tools = tools;
|
|
211
|
+
}
|
|
212
|
+
|
|
213
|
+
const tc = options.toolChoice as Body | undefined;
|
|
214
|
+
if (tc !== undefined) {
|
|
215
|
+
if (tc?.type === 'auto' || tc?.type === 'none' || tc?.type === 'required') params.tool_choice = tc.type;
|
|
216
|
+
else if (tc?.type === 'tool' && typeof tc.toolName === 'string') params.tool_choice = { type: 'function', function: { name: tc.toolName } };
|
|
217
|
+
else return { response: v3Error(400, 'toolChoice must be auto/none/required or { type: "tool", toolName }', { param: 'toolChoice' }) };
|
|
218
|
+
}
|
|
219
|
+
|
|
220
|
+
const rf = options.responseFormat as Body | undefined;
|
|
221
|
+
if (rf !== undefined && rf?.type === 'json') {
|
|
222
|
+
// V3's { type:'json', schema? } maps onto the gateway's legacy json response_format, which
|
|
223
|
+
// the chat machinery already synthesizes schema-conformant output for.
|
|
224
|
+
params.response_format = { type: 'json', ...(rf.schema !== undefined ? { schema: rf.schema } : {}) };
|
|
225
|
+
}
|
|
226
|
+
|
|
227
|
+
// providerOptions.gateway (order/only/sort/models/byok/caching) flows through verbatim — the
|
|
228
|
+
// chat machinery already parses exactly this shape.
|
|
229
|
+
if (options.providerOptions !== undefined) params.providerOptions = options.providerOptions;
|
|
230
|
+
|
|
231
|
+
return { params };
|
|
232
|
+
}
|
|
233
|
+
|
|
234
|
+
// ── OpenAI-compatible chat body → LanguageModelV3 generate result ────────────────────────
|
|
235
|
+
|
|
236
|
+
function mapFinishReason(raw: string): Body {
|
|
237
|
+
const unified = raw === 'tool_calls' ? 'tool-calls' : raw === 'stop' || raw === 'length' ? raw : 'other';
|
|
238
|
+
return { unified, raw };
|
|
239
|
+
}
|
|
240
|
+
|
|
241
|
+
function mapUsage(usage: Body): Body {
|
|
242
|
+
const prompt = Number(usage.prompt_tokens ?? 0);
|
|
243
|
+
const completion = Number(usage.completion_tokens ?? 0);
|
|
244
|
+
const cached = Number((usage.prompt_tokens_details as Body | undefined)?.cached_tokens ?? 0);
|
|
245
|
+
const reasoning = Number((usage.completion_tokens_details as Body | undefined)?.reasoning_tokens ?? 0);
|
|
246
|
+
return {
|
|
247
|
+
inputTokens: { total: prompt, noCache: prompt - cached, cacheRead: cached, cacheWrite: 0 },
|
|
248
|
+
outputTokens: { total: completion, text: completion - reasoning, reasoning },
|
|
249
|
+
raw: usage,
|
|
250
|
+
};
|
|
251
|
+
}
|
|
252
|
+
|
|
253
|
+
function contentParts(message: Body): Body[] {
|
|
254
|
+
const parts: Body[] = [];
|
|
255
|
+
if (typeof message.reasoning === 'string') parts.push({ type: 'reasoning', text: message.reasoning });
|
|
256
|
+
if (Array.isArray(message.tool_calls)) {
|
|
257
|
+
for (const call of message.tool_calls as Body[]) {
|
|
258
|
+
const fn = (call.function ?? {}) as Body;
|
|
259
|
+
parts.push({ type: 'tool-call', toolCallId: String(call.id ?? ''), toolName: String(fn.name ?? ''), input: String(fn.arguments ?? '{}') });
|
|
260
|
+
}
|
|
261
|
+
} else {
|
|
262
|
+
parts.push({ type: 'text', text: String(message.content ?? '') });
|
|
263
|
+
}
|
|
264
|
+
return parts;
|
|
265
|
+
}
|
|
266
|
+
|
|
267
|
+
function v3GenerateBody(chatBody: Body, occurredAt?: string): Body {
|
|
268
|
+
const choice = (chatBody.choices as Body[])[0]!;
|
|
269
|
+
const message = choice.message as Body;
|
|
270
|
+
return {
|
|
271
|
+
content: contentParts(message),
|
|
272
|
+
finishReason: mapFinishReason(String(choice.finish_reason)),
|
|
273
|
+
usage: mapUsage((chatBody.usage ?? {}) as Body),
|
|
274
|
+
...(chatBody.providerMetadata !== undefined ? { providerMetadata: chatBody.providerMetadata } : {}),
|
|
275
|
+
response: { id: chatBody.id, modelId: chatBody.model, timestamp: nowIso(occurredAt) },
|
|
276
|
+
warnings: [],
|
|
277
|
+
};
|
|
278
|
+
}
|
|
279
|
+
|
|
280
|
+
// SSE stream of LanguageModelV3 stream parts: response-metadata first (the generation id rides
|
|
281
|
+
// on it), reasoning-start/-delta/-end, then text-start/-delta/-end OR
|
|
282
|
+
// tool-input-start/-delta/-end + a tool-call part, a finish part carrying finishReason + usage
|
|
283
|
+
// (+ providerMetadata.gateway), and `data: [DONE]`.
|
|
284
|
+
function streamV3(chatBody: Body, sink: (event: SseEvent) => void, occurredAt?: string): void {
|
|
285
|
+
const choice = (chatBody.choices as Body[])[0]!;
|
|
286
|
+
const message = choice.message as Body;
|
|
287
|
+
sink({ data: { type: 'response-metadata', id: chatBody.id, modelId: chatBody.model, timestamp: nowIso(occurredAt) } });
|
|
288
|
+
if (typeof message.reasoning === 'string') {
|
|
289
|
+
const reasoning = message.reasoning;
|
|
290
|
+
sink({ data: { type: 'reasoning-start', id: 'reasoning-0' } });
|
|
291
|
+
for (let i = 0; i < reasoning.length; i += 48) {
|
|
292
|
+
sink({ data: { type: 'reasoning-delta', id: 'reasoning-0', delta: reasoning.slice(i, i + 48) } });
|
|
293
|
+
}
|
|
294
|
+
sink({ data: { type: 'reasoning-end', id: 'reasoning-0' } });
|
|
295
|
+
}
|
|
296
|
+
if (Array.isArray(message.tool_calls)) {
|
|
297
|
+
for (const call of message.tool_calls as Body[]) {
|
|
298
|
+
const fn = (call.function ?? {}) as Body;
|
|
299
|
+
const id = String(call.id ?? '');
|
|
300
|
+
const args = String(fn.arguments ?? '{}');
|
|
301
|
+
sink({ data: { type: 'tool-input-start', id, toolName: String(fn.name ?? '') } });
|
|
302
|
+
sink({ data: { type: 'tool-input-delta', id, delta: args } });
|
|
303
|
+
sink({ data: { type: 'tool-input-end', id } });
|
|
304
|
+
sink({ data: { type: 'tool-call', toolCallId: id, toolName: String(fn.name ?? ''), input: args } });
|
|
305
|
+
}
|
|
306
|
+
} else {
|
|
307
|
+
const text = String(message.content ?? '');
|
|
308
|
+
sink({ data: { type: 'text-start', id: 'text-0' } });
|
|
309
|
+
for (let i = 0; i < text.length; i += 24) {
|
|
310
|
+
sink({ data: { type: 'text-delta', id: 'text-0', delta: text.slice(i, i + 24) } });
|
|
311
|
+
}
|
|
312
|
+
sink({ data: { type: 'text-end', id: 'text-0' } });
|
|
313
|
+
}
|
|
314
|
+
sink({
|
|
315
|
+
data: {
|
|
316
|
+
type: 'finish',
|
|
317
|
+
finishReason: mapFinishReason(String(choice.finish_reason)),
|
|
318
|
+
usage: mapUsage((chatBody.usage ?? {}) as Body),
|
|
319
|
+
...(chatBody.providerMetadata !== undefined ? { providerMetadata: chatBody.providerMetadata } : {}),
|
|
320
|
+
},
|
|
321
|
+
});
|
|
322
|
+
sink({ done: true });
|
|
323
|
+
}
|
|
324
|
+
|
|
325
|
+
// ── POST /v3/ai/language-model ────────────────────────────────────────────────────────────
|
|
326
|
+
export async function handleV3LanguageModel(req: AiGatewayRequest, chat: ChatFn): Promise<AiGatewayResponse> {
|
|
327
|
+
const headers = req.headers ?? {};
|
|
328
|
+
const modelId = headers['ai-language-model-id'];
|
|
329
|
+
if (!modelId) {
|
|
330
|
+
return v3Error(400, "Invalid request: missing required header 'ai-language-model-id'", { param: 'ai-language-model-id', code: 'missing_parameter' });
|
|
331
|
+
}
|
|
332
|
+
const specVersion = headers['ai-language-model-specification-version'];
|
|
333
|
+
if (specVersion !== undefined && specVersion !== '3') {
|
|
334
|
+
return v3Error(400, `Unsupported ai-language-model-specification-version: ${specVersion} (the twin models specification version 3)`, { param: 'ai-language-model-specification-version' });
|
|
335
|
+
}
|
|
336
|
+
const streaming = headers['ai-language-model-streaming'] === 'true';
|
|
337
|
+
|
|
338
|
+
let options: Body = {};
|
|
339
|
+
if (req.body?.trim()) {
|
|
340
|
+
try {
|
|
341
|
+
const parsed = JSON.parse(req.body);
|
|
342
|
+
if (parsed && typeof parsed === 'object' && !Array.isArray(parsed)) options = parsed as Body;
|
|
343
|
+
} catch {
|
|
344
|
+
return v3Error(400, 'request body must be a JSON object of LanguageModelV3 call options');
|
|
345
|
+
}
|
|
346
|
+
}
|
|
347
|
+
|
|
348
|
+
const translated = translateCallOptions(modelId, streaming, options);
|
|
349
|
+
if ('response' in translated) return translated.response;
|
|
350
|
+
|
|
351
|
+
// Run the SAME chat machinery/kernel fold as /v1 (validation, routing, scenario scripting,
|
|
352
|
+
// generation record, credits) — but WITHOUT the OpenAI SSE sink: the v3 stream shape is
|
|
353
|
+
// emitted below from the unary body.
|
|
354
|
+
const { sseSink: _sseSink, ...chatReq } = req;
|
|
355
|
+
const result = await chat(chatReq, translated.params);
|
|
356
|
+
|
|
357
|
+
if (result.status !== 200) {
|
|
358
|
+
const errorBody = result.body as Body;
|
|
359
|
+
const err = (errorBody?.error ?? {}) as Body;
|
|
360
|
+
// The /v1 surface reports unknown models OpenAI-style (type invalid_request_error, code
|
|
361
|
+
// model_not_found); the AI SDK protocol's typed errors key on error.type, with the model id
|
|
362
|
+
// in an OBJECT param — remap so the SDK surfaces GatewayModelNotFoundError.
|
|
363
|
+
if (result.status === 404 && err.code === 'model_not_found') {
|
|
364
|
+
return v3Error(404, String(err.message ?? 'Model not found'), { type: 'model_not_found', param: { modelId } });
|
|
365
|
+
}
|
|
366
|
+
return result; // same { error: { message, type, param, code } } envelope the SDK parses
|
|
367
|
+
}
|
|
368
|
+
|
|
369
|
+
const chatBody = result.body as Body;
|
|
370
|
+
if (streaming && req.sseSink) streamV3(chatBody, req.sseSink, req.occurredAt);
|
|
371
|
+
return { status: 200, body: v3GenerateBody(chatBody, req.occurredAt) };
|
|
372
|
+
}
|
package/src/cli.ts
ADDED
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
import { keepProcessAlive } from '@volter/world-core/lifecycle';
|
|
3
|
+
import { hasFlag, optionValue } from '@volter/world-core/args';
|
|
4
|
+
import { createAiGatewayTwinServer } from './ai-gateway-server.ts';
|
|
5
|
+
|
|
6
|
+
const [cmd, ...rest] = process.argv.slice(2);
|
|
7
|
+
const port = Number(optionValue(rest, '--port', '0')) || undefined;
|
|
8
|
+
const root = optionValue(rest, '--root') || undefined;
|
|
9
|
+
const readOnly = hasFlag(rest, '--read-only');
|
|
10
|
+
const scenario = optionValue(rest, '--scenario') || undefined; // scripted completions (JSON scenario file)
|
|
11
|
+
|
|
12
|
+
if (cmd === 'serve') {
|
|
13
|
+
const server = await createAiGatewayTwinServer({ readOnly, ...(root ? { root } : {}), ...(port ? { port } : {}), ...(scenario ? { scenarioPath: scenario } : {}) });
|
|
14
|
+
process.stdout.write(`ai-gateway twin (deterministic stub output)${readOnly ? ' [read-only]' : ''}${scenario ? ` [scenario: ${scenario}]` : ''} at http://127.0.0.1:${server.port}\n`);
|
|
15
|
+
await keepProcessAlive();
|
|
16
|
+
} else if (cmd === 'conformance') {
|
|
17
|
+
const { checkAiGatewayConformance } = await import('./ai-gateway-conformance.ts'); // lazy — dev-only
|
|
18
|
+
const report = await checkAiGatewayConformance({ ...(root ? { root } : {}) });
|
|
19
|
+
process.stdout.write(`${JSON.stringify(report, null, 2)}\n`);
|
|
20
|
+
if (!report.ok) process.exitCode = 1;
|
|
21
|
+
} else {
|
|
22
|
+
process.stdout.write('Usage: world-ai-gateway serve|conformance [--port N] [--root DIR] [--read-only] [--scenario FILE]\n');
|
|
23
|
+
}
|
package/src/index.ts
ADDED
|
@@ -0,0 +1,67 @@
|
|
|
1
|
+
export { handleAiGatewayTwinRequest } from './ai-gateway-twin.ts';
|
|
2
|
+
export type { AiGatewayRequest, AiGatewayResponse, SseEvent, SseSink, GenerationRecord, GatewayRouting, ModelAttempt, ProviderAttempt } from './ai-gateway-types.ts';
|
|
3
|
+
export { createAiGatewayTwinFetch, createAiGatewayTwinServer, type AiGatewayTwinFetchOptions } from './ai-gateway-server.ts';
|
|
4
|
+
export { AI_GATEWAY_CATALOG, AI_GATEWAY_MODELS, findAiGatewayModel } from './ai-gateway-models.ts';
|
|
5
|
+
export type { AiGatewayModel, AiGatewayCatalogEntry } from './ai-gateway-models.ts';
|
|
6
|
+
export { contentToText, countPromptTokens, estimateTokens, lastUserText, stableHash, stubAssistantText, synthesizeJsonSchemaValue } from './ai-gateway-stub.ts';
|
|
7
|
+
export { aiGatewayScenarioAdapter, createAiGatewayScenarioEngine, loadAiGatewayScenarioDocument, realizeAiGatewayRespond } from './ai-gateway-scenario.ts';
|
|
8
|
+
export type { AiGatewayScenarioEngine, AiGatewayScenarioRequest, AiGatewayScenarioRespond, ScenarioToolCall, ScriptedResult } from './ai-gateway-scenario.ts';
|
|
9
|
+
export {
|
|
10
|
+
aiGatewayRequestForAction,
|
|
11
|
+
pullAiGatewayState,
|
|
12
|
+
pushAiGatewayAction,
|
|
13
|
+
syncAiGatewayFromReal,
|
|
14
|
+
} from './ai-gateway-connector.ts';
|
|
15
|
+
export type { AiGatewayExecute } from './ai-gateway-connector.ts';
|
|
16
|
+
|
|
17
|
+
import { registerPack, type TwinPack } from '@volter/world-core';
|
|
18
|
+
|
|
19
|
+
import { performAiGatewayAction, syncAiGatewayFromRemote } from './ai-gateway-connector.ts';
|
|
20
|
+
|
|
21
|
+
export const pack: TwinPack = {
|
|
22
|
+
// PROTOCOL 2 (docs/contributing/architecture.md#protocol-2-the-pack-is-a-plugin): the pack is a plugin — its wire, its tree, and its half of the real
|
|
23
|
+
// state system. Moved 2026-09-08. The gateway has NO write API: inference is the only "write", and
|
|
24
|
+
// replaying one would spend real money — so a recorded generation is RECONCILED against the gateway
|
|
25
|
+
// rather than written, and everything else settles with that reason.
|
|
26
|
+
protocol: '2',
|
|
27
|
+
refresh: { every: '15m', onDemand: { atMost: '60s' } },
|
|
28
|
+
stateSystem: { perform: performAiGatewayAction, refresh: syncAiGatewayFromRemote },
|
|
29
|
+
// the round trip: a chat completion — the twin records the generation it answered, which is the row a
|
|
30
|
+
// caller makes here, and a fresh one each time
|
|
31
|
+
roundTrip: { method: 'POST', path: '/v1/chat/completions', body: { model: 'openai/gpt-5', messages: [{ role: 'user', content: 'round trip' }] }, headers: { authorization: 'Bearer round-trip' } },
|
|
32
|
+
parityOrigin: 'http://twin',
|
|
33
|
+
vendor: 'ai-gateway',
|
|
34
|
+
transport: 'rest',
|
|
35
|
+
archetype: 'generative',
|
|
36
|
+
bin: 'world-ai-gateway',
|
|
37
|
+
// The subject types the twin SERVES from its own state — the R2 resource-level claim.
|
|
38
|
+
// `model` and `credits` are deliberately NOT here: both are connector-side projections
|
|
39
|
+
// that only `pullAiGatewayState` / `syncAiGatewayFromReal` fold in from a real gateway
|
|
40
|
+
// through an injected client, and no route reads either back — GET /v1/models serves the
|
|
41
|
+
// static AI_GATEWAY_MODELS catalog and GET /v1/credits a balance DERIVED from the
|
|
42
|
+
// generation rows, neither from a stored subject.
|
|
43
|
+
resources: ['generation'],
|
|
44
|
+
specSource: 'Vercel AI Gateway docs (OpenAI-compatible Chat Completions/Embeddings + REST API Reference: models/credits/generation/report) + the shipped @ai-sdk/gateway SDK for the /v3/ai AI SDK gateway protocol (config + language-model); model output is a deterministic stub',
|
|
45
|
+
description: 'Vercel AI Gateway twin — OpenAI-compatible chat completions/streaming with faithful provider-routing metadata, the AI SDK gateway protocol (/v3/ai: config + language-model for `createGateway`), models catalog with per-provider endpoints, billed embeddings, credits/generation/report surfaces; generative output is a labeled deterministic stub.',
|
|
46
|
+
// The gateway serves TWO API prefixes on one host: '/v1/' (OpenAI-compat + REST reference)
|
|
47
|
+
// and '/v3/ai' (the AI SDK gateway protocol `createGateway` speaks). `apiPathPrefix` is a
|
|
48
|
+
// single startsWith prefix (control-plane VendorRoute), so '/v' is the narrowest value that
|
|
49
|
+
// routes BOTH surfaces to the twin.
|
|
50
|
+
browserRouting: { apiPathPrefix: '/v', loaderHost: 'https://ai-gateway.vercel.sh' },
|
|
51
|
+
// INTERCEPTION RULING — hostsNone, the pack's own home for it: SDK is base-URL-configured
|
|
52
|
+
// (@ai-sdk/gateway baseURL / AI_GATEWAY_BASE_URL) — worlds wire the twin through explicit
|
|
53
|
+
// endpoint config; no injector entry grounded yet.
|
|
54
|
+
hostsNone:
|
|
55
|
+
"SDK is base-URL-configured (@ai-sdk/gateway baseURL / AI_GATEWAY_BASE_URL) — worlds wire the twin through explicit endpoint config; no injector entry grounded yet",
|
|
56
|
+
// Adoption, moved off the central maps unchanged (descriptor-first back-migration, adding-a-twin.md §3,
|
|
57
|
+
// 2026-08-31).
|
|
58
|
+
adoption: {
|
|
59
|
+
// No Python distribution: Vercel's AI Gateway is addressed through the TypeScript AI SDK
|
|
60
|
+
// (`@ai-sdk/gateway`); a Python caller points an OpenAI-compatible client at the gateway base
|
|
61
|
+
// URL, and that distribution (`openai`) belongs to the openai pack.
|
|
62
|
+
pypi: [],
|
|
63
|
+
sdks: ['@ai-sdk/gateway'], envStems: ['AIGATEWAY'],
|
|
64
|
+
},
|
|
65
|
+
};
|
|
66
|
+
// registered at import: the kernel learns the pack's state system (protocol 2)
|
|
67
|
+
registerPack(pack);
|