@volter/twin-ai-gateway 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +202 -0
- package/README.md +176 -0
- package/dist/src/ai-gateway-capabilities.d.ts +4 -0
- package/dist/src/ai-gateway-capabilities.js +772 -0
- package/dist/src/ai-gateway-conformance.d.ts +11 -0
- package/dist/src/ai-gateway-conformance.js +72 -0
- package/dist/src/ai-gateway-connector.d.ts +50 -0
- package/dist/src/ai-gateway-connector.js +97 -0
- package/dist/src/ai-gateway-models.d.ts +27 -0
- package/dist/src/ai-gateway-models.js +65 -0
- package/dist/src/ai-gateway-perform-harness.d.ts +5 -0
- package/dist/src/ai-gateway-perform-harness.js +17 -0
- package/dist/src/ai-gateway-scenario.d.ts +36 -0
- package/dist/src/ai-gateway-scenario.js +125 -0
- package/dist/src/ai-gateway-server.d.ts +16 -0
- package/dist/src/ai-gateway-server.js +107 -0
- package/dist/src/ai-gateway-stub.d.ts +20 -0
- package/dist/src/ai-gateway-stub.js +124 -0
- package/dist/src/ai-gateway-twin.d.ts +2 -0
- package/dist/src/ai-gateway-twin.js +994 -0
- package/dist/src/ai-gateway-types.d.ts +94 -0
- package/dist/src/ai-gateway-types.js +1 -0
- package/dist/src/ai-gateway-v3.d.ts +5 -0
- package/dist/src/ai-gateway-v3.js +367 -0
- package/dist/src/cli.d.ts +2 -0
- package/dist/src/cli.js +24 -0
- package/dist/src/index.d.ts +12 -0
- package/dist/src/index.js +54 -0
- package/package.json +66 -0
- package/src/ai-gateway-capabilities.ts +894 -0
- package/src/ai-gateway-conformance.ts +77 -0
- package/src/ai-gateway-connector.ts +100 -0
- package/src/ai-gateway-models.ts +105 -0
- package/src/ai-gateway-perform-harness.ts +17 -0
- package/src/ai-gateway-scenario.ts +137 -0
- package/src/ai-gateway-server.ts +122 -0
- package/src/ai-gateway-stub.ts +115 -0
- package/src/ai-gateway-twin.ts +1068 -0
- package/src/ai-gateway-types.ts +96 -0
- package/src/ai-gateway-v3.ts +372 -0
- package/src/cli.ts +23 -0
- package/src/index.ts +67 -0
|
@@ -0,0 +1,77 @@
|
|
|
1
|
+
import { mkdtempSync, rmSync } from 'node:fs';
|
|
2
|
+
import { tmpdir } from 'node:os';
|
|
3
|
+
import { join } from 'node:path';
|
|
4
|
+
import { handleAiGatewayTwinRequest } from './ai-gateway-twin.ts';
|
|
5
|
+
import type { SseEvent } from './ai-gateway-types.ts';
|
|
6
|
+
|
|
7
|
+
export type AiGatewayConformanceReport = {
|
|
8
|
+
ok: boolean;
|
|
9
|
+
checksRun: number;
|
|
10
|
+
violations: Array<{ check: string; detail: string }>;
|
|
11
|
+
};
|
|
12
|
+
|
|
13
|
+
type Body = Record<string, any>;
|
|
14
|
+
|
|
15
|
+
export async function checkAiGatewayConformance(opts: { root?: string } = {}): Promise<AiGatewayConformanceReport> {
|
|
16
|
+
const root = opts.root ?? mkdtempSync(join(tmpdir(), 'ai-gateway-conf-'));
|
|
17
|
+
const owned = opts.root === undefined;
|
|
18
|
+
const violations: Array<{ check: string; detail: string }> = [];
|
|
19
|
+
let checksRun = 0;
|
|
20
|
+
const fail = (check: string, detail: string) => violations.push({ check, detail });
|
|
21
|
+
const h = (method: string, path: string, body?: unknown, sseSink?: (event: SseEvent) => void) =>
|
|
22
|
+
handleAiGatewayTwinRequest({ method, path, body: body === undefined ? undefined : JSON.stringify(body), root, ...(sseSink ? { sseSink } : {}) });
|
|
23
|
+
|
|
24
|
+
try {
|
|
25
|
+
checksRun++;
|
|
26
|
+
const chat = await h('POST', '/v1/chat/completions', { model: 'anthropic/claude-sonnet-4.6', messages: [{ role: 'user', content: 'hello' }] });
|
|
27
|
+
const cb = chat.body as Body;
|
|
28
|
+
if (chat.status !== 200 || cb.object !== 'chat.completion' || cb.choices?.[0]?.message?.role !== 'assistant') fail('chat.envelope', 'invalid chat completion envelope');
|
|
29
|
+
if (!String(cb.id).startsWith('gen_')) fail('chat.generation_id', 'chat completion id is not a gen_ generation id');
|
|
30
|
+
if (typeof cb.providerMetadata?.gateway?.cost !== 'string' || cb.providerMetadata?.gateway?.routing?.finalProvider !== 'anthropic') {
|
|
31
|
+
fail('chat.provider_metadata', 'providerMetadata.gateway routing/cost missing');
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
checksRun++;
|
|
35
|
+
const events: SseEvent[] = [];
|
|
36
|
+
await h('POST', '/v1/chat/completions', { model: 'anthropic/claude-sonnet-4.6', stream: true, messages: [{ role: 'user', content: 'stream' }] }, (event) => events.push(event));
|
|
37
|
+
if (!events.some((event) => event.data?.object === 'chat.completion.chunk') || events.at(-1)?.done !== true) fail('streaming', 'missing chunk or [DONE]');
|
|
38
|
+
|
|
39
|
+
checksRun++;
|
|
40
|
+
const models = await h('GET', '/v1/models');
|
|
41
|
+
if ((models.body as Body).object !== 'list' || !Array.isArray((models.body as Body).data) || (models.body as Body).data.length === 0) fail('models.list', 'missing models data');
|
|
42
|
+
const one = await h('GET', '/v1/models/anthropic/claude-sonnet-4.6');
|
|
43
|
+
if ((one.body as Body).id !== 'anthropic/claude-sonnet-4.6') fail('models.retrieve', 'missing model');
|
|
44
|
+
|
|
45
|
+
checksRun++;
|
|
46
|
+
const gen = await h('GET', `/v1/generation?id=${encodeURIComponent(cb.id)}`);
|
|
47
|
+
if ((gen.body as Body).data?.model !== 'anthropic/claude-sonnet-4.6') fail('generation.lookup', 'generation lookup failed');
|
|
48
|
+
const creditsRes = await h('GET', '/v1/credits');
|
|
49
|
+
if (typeof (creditsRes.body as Body).balance !== 'string' || typeof (creditsRes.body as Body).total_used !== 'string') fail('credits', 'credits shape invalid');
|
|
50
|
+
|
|
51
|
+
checksRun++;
|
|
52
|
+
const bad = await h('POST', '/v1/chat/completions', { model: 'anthropic/claude-sonnet-4.6', messages: [] });
|
|
53
|
+
if (bad.status !== 400 || (bad.body as Body).error?.type !== 'invalid_request_error') fail('error.envelope', 'bad chat request did not return the vendor error envelope');
|
|
54
|
+
|
|
55
|
+
checksRun++;
|
|
56
|
+
const embed = await h('POST', '/v1/embeddings', { model: 'openai/text-embedding-3-small', input: 'hi' });
|
|
57
|
+
if (embed.status !== 200 || !Array.isArray((embed.body as Body).data?.[0]?.embedding)) fail('embeddings.create', 'embeddings envelope invalid');
|
|
58
|
+
if (!String((embed.body as Body).providerMetadata?.gateway?.generationId ?? '').startsWith('gen_')) fail('embeddings.accounting', 'embeddings call did not record a billed generation');
|
|
59
|
+
|
|
60
|
+
checksRun++;
|
|
61
|
+
const v3 = await handleAiGatewayTwinRequest({
|
|
62
|
+
method: 'POST',
|
|
63
|
+
path: '/v3/ai/language-model',
|
|
64
|
+
headers: { 'ai-language-model-id': 'anthropic/claude-sonnet-4.6', 'ai-language-model-streaming': 'false' },
|
|
65
|
+
body: JSON.stringify({ prompt: [{ role: 'user', content: [{ type: 'text', text: 'v3 conformance' }] }] }),
|
|
66
|
+
root,
|
|
67
|
+
});
|
|
68
|
+
const vb = v3.body as Body;
|
|
69
|
+
if (v3.status !== 200 || (vb.finishReason as Body)?.unified !== 'stop' || !Array.isArray(vb.content)) fail('v3.language_model', 'AI SDK gateway protocol generate result invalid');
|
|
70
|
+
const config = await h('GET', '/v3/ai/config');
|
|
71
|
+
if (config.status !== 200 || !Array.isArray((config.body as Body).models)) fail('v3.config', 'AI SDK gateway config document invalid');
|
|
72
|
+
} finally {
|
|
73
|
+
if (owned) rmSync(root, { recursive: true, force: true });
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
return { ok: violations.length === 0, checksRun, violations };
|
|
77
|
+
}
|
|
@@ -0,0 +1,100 @@
|
|
|
1
|
+
// Connector: pull/push against the REAL AI Gateway over an INJECTED client (the real fetch-based
|
|
2
|
+
// executor in prod, a fake in tests). No other module may reach the live vendor — the vendor's
|
|
3
|
+
// read surface here is the unauthenticated model catalog (GET /v1/models) plus the team's
|
|
4
|
+
// credits balance (GET /v1/credits).
|
|
5
|
+
import { observeResources } from '@volter/world-core';
|
|
6
|
+
import type { PerformContext, PushOutcome, RemoteExecute, SyncResource, TwinAction } from '@volter/world-core';
|
|
7
|
+
import type { GenerationRecord } from './ai-gateway-types.ts';
|
|
8
|
+
|
|
9
|
+
export type AiGatewayExecute = (req: { method: string; path: string; body?: unknown }) => Promise<{ status: number; data: unknown }>;
|
|
10
|
+
const DEFAULT_OCCURRED_AT = '1970-01-01T00:00:00.000Z';
|
|
11
|
+
|
|
12
|
+
/** Fetch the gateway's model catalog + credits through the injected client and map to
|
|
13
|
+
* SyncResource[] (no fold). */
|
|
14
|
+
async function collectAiGatewayState(execute: AiGatewayExecute): Promise<SyncResource[]> {
|
|
15
|
+
const resources: SyncResource[] = [];
|
|
16
|
+
const models = await execute({ method: 'GET', path: '/v1/models' });
|
|
17
|
+
const data = (models.data as { data?: Array<Record<string, unknown>> }).data ?? [];
|
|
18
|
+
for (const model of data) {
|
|
19
|
+
if (typeof model.id === 'string') resources.push({ type: 'model', id: `model_${model.id}`, fields: model });
|
|
20
|
+
}
|
|
21
|
+
const credits = await execute({ method: 'GET', path: '/v1/credits' });
|
|
22
|
+
if (credits.status >= 200 && credits.status < 300 && credits.data && typeof credits.data === 'object') {
|
|
23
|
+
resources.push({ type: 'credits', id: 'credits_account', fields: credits.data as Record<string, unknown> });
|
|
24
|
+
}
|
|
25
|
+
return resources;
|
|
26
|
+
}
|
|
27
|
+
|
|
28
|
+
/** One fold onto the head: protocol 2's observe, one batch, one instant. */
|
|
29
|
+
function fold(resources: SyncResource[], opts: { root?: string; occurredAt?: string }) {
|
|
30
|
+
const at = opts.occurredAt ?? DEFAULT_OCCURRED_AT;
|
|
31
|
+
return observeResources('ai-gateway', resources.map((r) => ({ type: r.type, id: r.id, fields: r.fields })), {
|
|
32
|
+
...(opts.root !== undefined ? { root: opts.root } : {}), at, batch: `obs:ai-gateway:${at}`,
|
|
33
|
+
});
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
export async function pullAiGatewayState(execute: AiGatewayExecute, opts: { root?: string; occurredAt?: string } = {}): Promise<void> {
|
|
37
|
+
const resources = await collectAiGatewayState(execute);
|
|
38
|
+
fold(resources, opts);
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
/**
|
|
42
|
+
* D7 consumer-facing pull entry point: gather all AI Gateway read domains (model catalog +
|
|
43
|
+
* credits) and fold them into the twin through a SINGLE observed batch, returning the standard result
|
|
44
|
+
* shape. Idempotent: a re-pull of identical state appends no new deltas.
|
|
45
|
+
*/
|
|
46
|
+
export async function syncAiGatewayFromReal(
|
|
47
|
+
execute: AiGatewayExecute,
|
|
48
|
+
opts: { root?: string; occurredAt?: string } = {},
|
|
49
|
+
): Promise<{ observed: number; deltasAppended: number }> {
|
|
50
|
+
const resources = await collectAiGatewayState(execute);
|
|
51
|
+
const result = fold(resources, opts);
|
|
52
|
+
return { observed: result.observed, deltasAppended: result.appended };
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
/** Map a local twin action to the real-gateway request that CONFIRMS it. The gateway exposes no
|
|
56
|
+
* write API (inference is the only "write", and replaying it against the live vendor would
|
|
57
|
+
* spend real money) — so push means reconciliation: verify a locally recorded generation
|
|
58
|
+
* exists upstream via GET /v1/generation. Unknown operation types return null (unpushable). */
|
|
59
|
+
export function aiGatewayRequestForAction(action: TwinAction): { method: string; path: string; body?: unknown } | null {
|
|
60
|
+
if (action.operation === 'generation.record') {
|
|
61
|
+
const record = action.fields as unknown as GenerationRecord;
|
|
62
|
+
return { method: 'GET', path: `/v1/generation?id=${encodeURIComponent(record.id)}` };
|
|
63
|
+
}
|
|
64
|
+
return null;
|
|
65
|
+
}
|
|
66
|
+
|
|
67
|
+
export async function pushAiGatewayAction(action: TwinAction, execute: AiGatewayExecute): Promise<boolean> {
|
|
68
|
+
const req = aiGatewayRequestForAction(action);
|
|
69
|
+
if (!req) return false;
|
|
70
|
+
const result = await execute(req);
|
|
71
|
+
return result.status >= 200 && result.status < 300;
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
// ── PROTOCOL 2: the pack's half of the real state system ────────────────────────────────────
|
|
75
|
+
/** The pack's executor over the kernel's. */
|
|
76
|
+
export function aiGatewayExecuteOver(execute: RemoteExecute): AiGatewayExecute {
|
|
77
|
+
return async ({ method, path, body }) => {
|
|
78
|
+
const res = await execute({ method, path, headers: { accept: 'application/json', authorization: 'Bearer twin', ...(body ? { 'content-type': 'application/json' } : {}) }, ...(body === undefined ? {} : { body: JSON.stringify(body) }) });
|
|
79
|
+
let data: unknown = {};
|
|
80
|
+
if (res.body) { try { data = JSON.parse(res.body); } catch { data = { error: res.body.slice(0, 200) }; } }
|
|
81
|
+
return { status: res.status, data };
|
|
82
|
+
};
|
|
83
|
+
}
|
|
84
|
+
/** The refresh adapter: pull the gateway's model catalogue and the account's credits. */
|
|
85
|
+
export async function syncAiGatewayFromRemote(execute: RemoteExecute, opts: { root?: string; origin?: string; occurredAt?: string } = {}): Promise<{ observed: number; deltasAppended: number }> {
|
|
86
|
+
return syncAiGatewayFromReal(aiGatewayExecuteOver(execute), {
|
|
87
|
+
...(opts.root !== undefined ? { root: opts.root } : {}),
|
|
88
|
+
...(opts.occurredAt !== undefined ? { occurredAt: opts.occurredAt } : {}),
|
|
89
|
+
});
|
|
90
|
+
}
|
|
91
|
+
/** The perform adapter. The gateway exposes NO write API — inference is the only "write", and replaying
|
|
92
|
+
* one against the live vendor would spend real money. So a recorded generation is RECONCILED rather than
|
|
93
|
+
* written: the head asks the gateway whether it has that generation, and a refusal is heard as one. */
|
|
94
|
+
export async function performAiGatewayAction(execute: RemoteExecute, action: TwinAction, _ctx: PerformContext): Promise<PushOutcome> {
|
|
95
|
+
const req = aiGatewayRequestForAction(action);
|
|
96
|
+
if (!req) return { externalId: action.subject.id, data: { performed: false, reason: `${action.operation ?? action.subject.type} is the twin's own record — the AI Gateway has no write API` } };
|
|
97
|
+
const ok = await pushAiGatewayAction(action, aiGatewayExecuteOver(execute));
|
|
98
|
+
if (!ok) throw new Error(`ai-gateway perform ${action.operation ?? action.subject.type} refused: the gateway does not hold that generation`);
|
|
99
|
+
return { externalId: action.subject.id, data: { performed: false, reason: 'reconciled against the gateway rather than written — inference is the only write, and replaying it would spend real money' } };
|
|
100
|
+
}
|
|
@@ -0,0 +1,105 @@
|
|
|
1
|
+
// Deterministic local model catalog for the AI Gateway twin. The SERVED shape follows the
|
|
2
|
+
// documented GET /v1/models response (OpenAI models format + gateway fields: name, description,
|
|
3
|
+
// context_window, max_tokens, type, tags, pricing — see Vercel "REST API Reference", List models).
|
|
4
|
+
// Model ids are `creator/model` slugs, exactly how the real gateway addresses models.
|
|
5
|
+
//
|
|
6
|
+
// `providers` is twin-internal routing data (which provider endpoints can serve the model, in
|
|
7
|
+
// default preference order) — it is NOT part of the served /v1/models row (the real gateway
|
|
8
|
+
// exposes per-provider data only via GET /v1/models/{creator}/{model}/endpoints).
|
|
9
|
+
|
|
10
|
+
export type AiGatewayModel = {
|
|
11
|
+
id: string;
|
|
12
|
+
object: 'model';
|
|
13
|
+
created: number;
|
|
14
|
+
released: number;
|
|
15
|
+
owned_by: string;
|
|
16
|
+
name: string;
|
|
17
|
+
description: string;
|
|
18
|
+
context_window: number;
|
|
19
|
+
max_tokens: number;
|
|
20
|
+
type: 'language' | 'embedding' | 'reranking' | 'image' | 'video';
|
|
21
|
+
tags: string[];
|
|
22
|
+
pricing: {
|
|
23
|
+
input: string;
|
|
24
|
+
output?: string;
|
|
25
|
+
input_cache_read?: string;
|
|
26
|
+
input_cache_write?: string;
|
|
27
|
+
};
|
|
28
|
+
};
|
|
29
|
+
|
|
30
|
+
export type AiGatewayCatalogEntry = {
|
|
31
|
+
model: AiGatewayModel;
|
|
32
|
+
/** Provider slugs able to serve this model, in the gateway's default preference order. */
|
|
33
|
+
providers: string[];
|
|
34
|
+
};
|
|
35
|
+
|
|
36
|
+
const lang = (
|
|
37
|
+
id: string,
|
|
38
|
+
ownedBy: string,
|
|
39
|
+
name: string,
|
|
40
|
+
contextWindow: number,
|
|
41
|
+
maxTokens: number,
|
|
42
|
+
input: string,
|
|
43
|
+
output: string,
|
|
44
|
+
providers: string[],
|
|
45
|
+
tags: string[] = ['tool-use', 'reasoning'],
|
|
46
|
+
): AiGatewayCatalogEntry => ({
|
|
47
|
+
model: {
|
|
48
|
+
id,
|
|
49
|
+
object: 'model',
|
|
50
|
+
created: 1755815280,
|
|
51
|
+
released: 1755815280,
|
|
52
|
+
owned_by: ownedBy,
|
|
53
|
+
name,
|
|
54
|
+
description: `Deterministic local catalog entry for ${name}.`,
|
|
55
|
+
context_window: contextWindow,
|
|
56
|
+
max_tokens: maxTokens,
|
|
57
|
+
type: 'language',
|
|
58
|
+
tags,
|
|
59
|
+
pricing: {
|
|
60
|
+
input,
|
|
61
|
+
output,
|
|
62
|
+
input_cache_read: '0.0000001',
|
|
63
|
+
input_cache_write: input,
|
|
64
|
+
},
|
|
65
|
+
},
|
|
66
|
+
providers,
|
|
67
|
+
});
|
|
68
|
+
|
|
69
|
+
export const AI_GATEWAY_CATALOG: AiGatewayCatalogEntry[] = [
|
|
70
|
+
lang('anthropic/claude-sonnet-4.6', 'anthropic', 'Claude Sonnet 4.6', 1_000_000, 64_000, '0.000003', '0.000015', ['anthropic', 'bedrock', 'vertex'], ['tool-use', 'reasoning', 'vision', 'file-input']),
|
|
71
|
+
lang('anthropic/claude-opus-4.8', 'anthropic', 'Claude Opus 4.8', 1_000_000, 64_000, '0.000005', '0.000025', ['anthropic', 'bedrock', 'vertex'], ['tool-use', 'reasoning', 'vision', 'file-input']),
|
|
72
|
+
lang('openai/gpt-5', 'openai', 'GPT-5', 400_000, 128_000, '0.00000125', '0.00001', ['openai', 'azure'], ['tool-use', 'reasoning', 'vision', 'file-input']),
|
|
73
|
+
lang('openai/gpt-5.2', 'openai', 'GPT-5.2', 400_000, 128_000, '0.00000125', '0.00001', ['openai', 'azure'], ['tool-use', 'reasoning', 'vision', 'file-input']),
|
|
74
|
+
lang('xai/grok-4-1', 'xai', 'Grok 4.1', 256_000, 64_000, '0.000003', '0.000015', ['xai'], ['tool-use', 'reasoning']),
|
|
75
|
+
lang('groq/llama-3.3-70b-versatile', 'groq', 'Llama 3.3 70B Versatile', 131_072, 32_768, '0.00000059', '0.00000079', ['groq'], ['tool-use']),
|
|
76
|
+
lang('zai/glm-4.6', 'zai', 'GLM 4.6', 200_000, 128_000, '0.0000006', '0.0000022', ['zai', 'novita'], ['tool-use', 'reasoning']),
|
|
77
|
+
lang('zai/glm-4.5-air', 'zai', 'GLM 4.5 Air', 128_000, 96_000, '0.0000002', '0.0000011', ['zai', 'novita'], ['tool-use', 'reasoning']),
|
|
78
|
+
lang('openai/gpt-oss-120b', 'openai', 'GPT OSS 120B', 131_072, 32_768, '0.0000001', '0.0000005', ['groq', 'cerebras', 'fireworks'], ['tool-use', 'reasoning']),
|
|
79
|
+
lang('openai/gpt-oss-20b', 'openai', 'GPT OSS 20B', 131_072, 32_768, '0.00000005', '0.0000002', ['groq', 'cerebras'], ['tool-use', 'reasoning']),
|
|
80
|
+
lang('minimax/minimax-m2', 'minimax', 'MiniMax M2', 200_000, 128_000, '0.0000003', '0.0000012', ['minimax', 'novita'], ['tool-use', 'reasoning']),
|
|
81
|
+
lang('google/gemini-3.1-pro-preview', 'google', 'Gemini 3.1 Pro Preview', 1_000_000, 64_000, '0.000002', '0.000012', ['google', 'vertex'], ['tool-use', 'reasoning', 'vision', 'file-input']),
|
|
82
|
+
{
|
|
83
|
+
model: {
|
|
84
|
+
id: 'openai/text-embedding-3-small',
|
|
85
|
+
object: 'model',
|
|
86
|
+
created: 1755815280,
|
|
87
|
+
released: 1755815280,
|
|
88
|
+
owned_by: 'openai',
|
|
89
|
+
name: 'Text Embedding 3 Small',
|
|
90
|
+
description: 'Deterministic local catalog entry for Text Embedding 3 Small.',
|
|
91
|
+
context_window: 8_191,
|
|
92
|
+
max_tokens: 0,
|
|
93
|
+
type: 'embedding',
|
|
94
|
+
tags: [],
|
|
95
|
+
pricing: { input: '0.00000002' },
|
|
96
|
+
},
|
|
97
|
+
providers: ['openai', 'azure'],
|
|
98
|
+
},
|
|
99
|
+
];
|
|
100
|
+
|
|
101
|
+
export const AI_GATEWAY_MODELS: AiGatewayModel[] = AI_GATEWAY_CATALOG.map((entry) => entry.model);
|
|
102
|
+
|
|
103
|
+
export function findAiGatewayModel(id: string): AiGatewayCatalogEntry | undefined {
|
|
104
|
+
return AI_GATEWAY_CATALOG.find((entry) => entry.model.id === id);
|
|
105
|
+
}
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
// A MINIATURE OF THE HEAD, for this pack's own claims and suites (protocol 2). Not on the serve path.
|
|
2
|
+
import { confirmAction, deployableEntries, worldNow } from '@volter/world-core';
|
|
3
|
+
import { pushAiGatewayAction, type AiGatewayExecute } from './ai-gateway-connector.ts';
|
|
4
|
+
|
|
5
|
+
export async function performPending(execute: AiGatewayExecute, opts: { root?: string; occurredAt?: string } = {}): Promise<number> {
|
|
6
|
+
let pushed = 0;
|
|
7
|
+
for (const entry of deployableEntries('ai-gateway', opts.root)) {
|
|
8
|
+
if (!(await pushAiGatewayAction(entry, execute))) continue;
|
|
9
|
+
confirmAction({
|
|
10
|
+
service: 'ai-gateway', actionId: entry.id, subject: entry.subject, fields: entry.fields ?? {},
|
|
11
|
+
occurredAt: opts.occurredAt ?? worldNow(), receipt: { status: 'deployed' },
|
|
12
|
+
...(opts.root !== undefined ? { root: opts.root } : {}),
|
|
13
|
+
});
|
|
14
|
+
pushed += 1;
|
|
15
|
+
}
|
|
16
|
+
return pushed;
|
|
17
|
+
}
|
|
@@ -0,0 +1,137 @@
|
|
|
1
|
+
// The ai-gateway pack's HALF of the scenario system on the kernel's ONE engine (@volter/world-core
|
|
2
|
+
// scenario.ts): the OpenAI-compat chat vocabulary + the scripted-turn respond shape the
|
|
3
|
+
// envelope builder consumes. The handler FILE (handlers/ai-gateway.json in a world dir) is
|
|
4
|
+
// the only write surface.
|
|
5
|
+
import { getActiveWorldStore, parseScenarioDocument, type PackScenarioAdapter, ScenarioError, type ScenarioDocument, ScenarioEngine, type ScenarioFeatures } from '@volter/world-core';
|
|
6
|
+
import { contentToText, lastUserText } from './ai-gateway-stub.ts';
|
|
7
|
+
import type { ChatMessage } from './ai-gateway-types.ts';
|
|
8
|
+
|
|
9
|
+
export type AiGatewayScenarioRequest = { model: string; messages: ChatMessage[]; tools?: unknown };
|
|
10
|
+
export type AiGatewayScenarioEngine = ScenarioEngine<AiGatewayScenarioRequest>;
|
|
11
|
+
|
|
12
|
+
/** A scripted tool call — the full arguments payload is emitted verbatim (and streamed as
|
|
13
|
+
* tool_calls argument deltas on the streaming path). */
|
|
14
|
+
export type ScenarioToolCall = { name: string; arguments: Record<string, unknown>; id?: string };
|
|
15
|
+
|
|
16
|
+
/** What the assistant says when a handler fires: any combination of reasoning/text/toolCalls
|
|
17
|
+
* (emitted in the vendor's field ordering). */
|
|
18
|
+
export type AiGatewayScenarioRespond = {
|
|
19
|
+
reasoning?: string;
|
|
20
|
+
text?: string;
|
|
21
|
+
toolCalls?: ScenarioToolCall | ScenarioToolCall[];
|
|
22
|
+
/** Defaults to 'tool_calls' when any tool call is present, else 'stop'. */
|
|
23
|
+
finishReason?: 'stop' | 'length' | 'tool_calls' | 'content_filter';
|
|
24
|
+
};
|
|
25
|
+
|
|
26
|
+
/** What a fired handler yields — the scripted assistant turn for the envelope builder. */
|
|
27
|
+
export type ScriptedResult = {
|
|
28
|
+
text: string | null;
|
|
29
|
+
reasoning: string | null;
|
|
30
|
+
toolCalls: ScenarioToolCall[];
|
|
31
|
+
finishReason: 'stop' | 'length' | 'tool_calls' | 'content_filter';
|
|
32
|
+
};
|
|
33
|
+
|
|
34
|
+
const RESPOND_KEYS = new Set(['reasoning', 'text', 'toolCalls', 'finishReason']);
|
|
35
|
+
const FINISH_REASONS = new Set(['stop', 'length', 'tool_calls', 'content_filter']);
|
|
36
|
+
const nonEmptyString = (cond: unknown): cond is string => typeof cond === 'string' && cond.length > 0;
|
|
37
|
+
|
|
38
|
+
/** The function names the last message's tool results answer: resolve each `tool` message's
|
|
39
|
+
* tool_call_id against the tool_calls of earlier assistant turns. */
|
|
40
|
+
function lastToolResultNames(messages: ChatMessage[]): Set<string> {
|
|
41
|
+
const last = messages[messages.length - 1];
|
|
42
|
+
const names = new Set<string>();
|
|
43
|
+
if (!last || last.role !== 'tool' || typeof last.tool_call_id !== 'string') return names;
|
|
44
|
+
for (const m of messages) {
|
|
45
|
+
if (m.role !== 'assistant' || !Array.isArray(m.tool_calls)) continue;
|
|
46
|
+
for (const tc of m.tool_calls) {
|
|
47
|
+
const call = tc as { id?: unknown; function?: { name?: unknown } };
|
|
48
|
+
if (call?.id === last.tool_call_id && typeof call?.function?.name === 'string') names.add(call.function.name);
|
|
49
|
+
}
|
|
50
|
+
}
|
|
51
|
+
return names;
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
function toolNames(tools: unknown): string[] {
|
|
55
|
+
if (!Array.isArray(tools)) return [];
|
|
56
|
+
return tools.map((t) => {
|
|
57
|
+
const fn = t as { function?: { name?: unknown }; name?: unknown };
|
|
58
|
+
return typeof fn?.function?.name === 'string' ? fn.function.name : typeof fn?.name === 'string' ? fn.name : null;
|
|
59
|
+
}).filter((n): n is string => n !== null);
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
export const aiGatewayScenarioAdapter: PackScenarioAdapter<AiGatewayScenarioRequest> = {
|
|
63
|
+
vendor: 'ai-gateway',
|
|
64
|
+
features: (req): ScenarioFeatures => ({
|
|
65
|
+
model: req.model,
|
|
66
|
+
lastUserText: lastUserText(req.messages).slice(0, 300),
|
|
67
|
+
tools: toolNames(req.tools),
|
|
68
|
+
lastMessageHasToolResult: req.messages[req.messages.length - 1]?.role === 'tool',
|
|
69
|
+
toolResultFor: [...lastToolResultNames(req.messages)],
|
|
70
|
+
}),
|
|
71
|
+
matchers: {
|
|
72
|
+
modelEquals: (req, cond) => nonEmptyString(cond) && req.model === cond,
|
|
73
|
+
userTextIncludes: (req, cond) => nonEmptyString(cond) && lastUserText(req.messages).toLowerCase().includes(cond.toLowerCase()),
|
|
74
|
+
anyTextIncludes: (req, cond) => nonEmptyString(cond) && req.messages.map((m) => contentToText(m.content)).join('\n').toLowerCase().includes(cond.toLowerCase()),
|
|
75
|
+
lastMessageHasToolResult: (req, cond) => typeof cond === 'boolean' && (req.messages[req.messages.length - 1]?.role === 'tool') === cond,
|
|
76
|
+
toolResultFor: (req, cond) => nonEmptyString(cond) && lastToolResultNames(req.messages).has(cond),
|
|
77
|
+
hasTool: (req, cond) => nonEmptyString(cond) && toolNames(req.tools).includes(cond),
|
|
78
|
+
},
|
|
79
|
+
text: (req) => req.messages.map((m) => contentToText(m.content)).join('\n'),
|
|
80
|
+
validateOn: (on) => {
|
|
81
|
+
for (const k of ['modelEquals', 'userTextIncludes', 'anyTextIncludes', 'toolResultFor', 'hasTool'] as const) if (on[k] !== undefined && (typeof on[k] !== 'string' || !on[k])) return `on.${k} is a non-empty string`;
|
|
82
|
+
for (const k of ['lastMessageHasToolResult'] as const) if (on[k] !== undefined && typeof on[k] !== 'boolean') return `on.${k} is a boolean`;
|
|
83
|
+
return null;
|
|
84
|
+
},
|
|
85
|
+
validateRespond: (respond) => {
|
|
86
|
+
if (typeof respond !== 'object' || respond === null || Array.isArray(respond)) return 'respond is an object { reasoning?, text?, toolCalls?, finishReason? }';
|
|
87
|
+
const r = respond as Record<string, unknown>;
|
|
88
|
+
for (const k of Object.keys(r)) if (!RESPOND_KEYS.has(k)) return `respond: unknown key "${k}" (valid: ${[...RESPOND_KEYS].join(', ')})`;
|
|
89
|
+
if (r.text !== undefined && typeof r.text !== 'string') return 'respond.text is a string';
|
|
90
|
+
if (r.reasoning !== undefined && typeof r.reasoning !== 'string') return 'respond.reasoning is a string';
|
|
91
|
+
if (r.finishReason !== undefined && (typeof r.finishReason !== 'string' || !FINISH_REASONS.has(r.finishReason))) return `respond.finishReason is one of ${[...FINISH_REASONS].join(', ')}`;
|
|
92
|
+
if (r.toolCalls !== undefined) {
|
|
93
|
+
for (const tc of Array.isArray(r.toolCalls) ? r.toolCalls : [r.toolCalls]) {
|
|
94
|
+
const t = tc as Record<string, unknown>;
|
|
95
|
+
if (!t || typeof t !== 'object' || Array.isArray(t)) return 'respond.toolCalls entries are objects';
|
|
96
|
+
if (typeof t.name !== 'string' || !t.name) return 'respond.toolCalls[].name is a non-empty string';
|
|
97
|
+
if (!t.arguments || typeof t.arguments !== 'object' || Array.isArray(t.arguments)) return 'respond.toolCalls[].arguments is an object';
|
|
98
|
+
if (t.id !== undefined && typeof t.id !== 'string') return 'respond.toolCalls[].id is a string';
|
|
99
|
+
}
|
|
100
|
+
}
|
|
101
|
+
const hasAny = r.text !== undefined || r.reasoning !== undefined || r.toolCalls !== undefined;
|
|
102
|
+
if (!hasAny) return 'respond needs text, reasoning, or toolCalls';
|
|
103
|
+
return null;
|
|
104
|
+
},
|
|
105
|
+
};
|
|
106
|
+
|
|
107
|
+
export function loadAiGatewayScenarioDocument(path: string): ScenarioDocument {
|
|
108
|
+
let parsed: unknown;
|
|
109
|
+
try {
|
|
110
|
+
// Read through the ACTIVE WorldStore, never the filesystem directly (runtime contract
|
|
111
|
+
// R12b): the handlers document is WORLD STATE, so a MemoryWorldStore / DO-backed world
|
|
112
|
+
// serves ITS OWN scenario instead of whatever happens to sit on the host disk — and the
|
|
113
|
+
// serve path stays workerd-clean. A missing document keeps the historical ENOENT wording,
|
|
114
|
+
// so the loud load-time failure reads byte-identically to the read it replaces.
|
|
115
|
+
const raw = getActiveWorldStore().read(path);
|
|
116
|
+
if (raw === null) throw new Error(`ENOENT: no such file or directory, open '${path}'`);
|
|
117
|
+
parsed = JSON.parse(raw);
|
|
118
|
+
} catch (e) {
|
|
119
|
+
throw new ScenarioError(`ai-gateway scenario: cannot read/parse ${path}: ${e instanceof Error ? e.message : String(e)}`);
|
|
120
|
+
}
|
|
121
|
+
return parseScenarioDocument(parsed, aiGatewayScenarioAdapter);
|
|
122
|
+
}
|
|
123
|
+
|
|
124
|
+
export function createAiGatewayScenarioEngine(document?: ScenarioDocument): AiGatewayScenarioEngine {
|
|
125
|
+
return new ScenarioEngine(aiGatewayScenarioAdapter, document);
|
|
126
|
+
}
|
|
127
|
+
|
|
128
|
+
/** Realize a fired handler's respond into the envelope builder's scripted-turn shape. */
|
|
129
|
+
export function realizeAiGatewayRespond(respond: AiGatewayScenarioRespond): ScriptedResult {
|
|
130
|
+
const toolCalls = respond.toolCalls === undefined ? [] : Array.isArray(respond.toolCalls) ? respond.toolCalls : [respond.toolCalls];
|
|
131
|
+
return {
|
|
132
|
+
text: respond.text ?? null,
|
|
133
|
+
reasoning: respond.reasoning ?? null,
|
|
134
|
+
toolCalls,
|
|
135
|
+
finishReason: respond.finishReason ?? (toolCalls.length > 0 ? 'tool_calls' : 'stop'),
|
|
136
|
+
};
|
|
137
|
+
}
|
|
@@ -0,0 +1,122 @@
|
|
|
1
|
+
// Vercel AI Gateway twin HTTP server.
|
|
2
|
+
//
|
|
3
|
+
// FETCH-FIRST (runtime contract R12b): the surface is the plain `createAiGatewayTwinFetch` and
|
|
4
|
+
// the SERVER is one line of `Bun.serve` around it, so the standalone (R1) and hosted lanes execute
|
|
5
|
+
// identical serving bytes. This is a CUSTOM fetch, not the kernel adapter
|
|
6
|
+
// (`createTwinFetchFromHandler`): the SSE streaming lane below genuinely exceeds the common shape
|
|
7
|
+
// — openai-server.ts is the reference for that.
|
|
8
|
+
import { serveHttp } from '@volter/world-core';
|
|
9
|
+
import { handleAiGatewayTwinRequest } from './ai-gateway-twin.ts';
|
|
10
|
+
import { twinManifest, worldNow } from '@volter/world-core';
|
|
11
|
+
import { createAiGatewayScenarioEngine, type AiGatewayScenarioEngine, loadAiGatewayScenarioDocument } from './ai-gateway-scenario.ts';
|
|
12
|
+
import type { SseEvent } from './ai-gateway-types.ts';
|
|
13
|
+
|
|
14
|
+
function wantsStream(body: string): boolean {
|
|
15
|
+
try {
|
|
16
|
+
return JSON.parse(body || '{}')?.stream === true;
|
|
17
|
+
} catch {
|
|
18
|
+
return false;
|
|
19
|
+
}
|
|
20
|
+
}
|
|
21
|
+
|
|
22
|
+
function encodeSse(event: SseEvent): string {
|
|
23
|
+
return event.done ? 'data: [DONE]\n\n' : `data: ${JSON.stringify(event.data)}\n\n`;
|
|
24
|
+
}
|
|
25
|
+
|
|
26
|
+
/** Options every AI Gateway-twin HTTP surface needs, independent of who owns the socket. */
|
|
27
|
+
export interface AiGatewayTwinFetchOptions {
|
|
28
|
+
root?: string;
|
|
29
|
+
readOnly?: boolean;
|
|
30
|
+
scenarioPath?: string;
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
export function createAiGatewayTwinFetch(options: AiGatewayTwinFetchOptions = {}): (request: Request) => Promise<Response> {
|
|
34
|
+
const readOnly = options.readOnly ?? false;
|
|
35
|
+
// Scenario scripting (ai-gateway-scenario.ts): a JSON scenario file — via the scenarioPath
|
|
36
|
+
// option or the TWIN_AI_GATEWAY_SCENARIO env var — scripts deterministic completions for
|
|
37
|
+
// POST /v1/chat/completions. A malformed scenario throws here, failing construction loudly.
|
|
38
|
+
const scenarioPath = options.scenarioPath ?? process.env.TWIN_AI_GATEWAY_SCENARIO;
|
|
39
|
+
const scenarioEngine: AiGatewayScenarioEngine | undefined = scenarioPath ? createAiGatewayScenarioEngine(loadAiGatewayScenarioDocument(scenarioPath)) : undefined;
|
|
40
|
+
return async function aiGatewayTwinFetch(request: Request): Promise<Response> {
|
|
41
|
+
const url = new URL(request.url);
|
|
42
|
+
// THE READ DOORS (TWIN-PROGRAMMING-MODEL): discovery + inspection, read-only.
|
|
43
|
+
if (request.method === 'GET' && url.pathname.replace(/\/+$/, '') === '/twin') {
|
|
44
|
+
return Response.json(twinManifest({
|
|
45
|
+
vendor: 'ai-gateway',
|
|
46
|
+
twinOf: 'Vercel AI Gateway (OpenAI-compat chat completions)',
|
|
47
|
+
stateSentence: 'Seed nothing — generations/credits accrue from ordinary API use with any key.',
|
|
48
|
+
behaviorSentence: 'Completions are scripted by MSW-shaped handlers in the world dir (handlers/ai-gateway.json): {on:{userTextIncludes|anyTextIncludes|modelEquals|hasTool|toolResultFor|lastMessageHasToolResult|nthCall}, respond:{text|reasoning|toolCalls, finishReason?}, once?, phase?}. Unmatched requests answer a labeled stub naming this door.',
|
|
49
|
+
exampleHandler: { on: { userTextIncludes: 'plan', hasTool: 'proposePlan' }, respond: { toolCalls: { name: 'proposePlan', arguments: { steps: 3 } } }, once: true },
|
|
50
|
+
engine: scenarioEngine as never,
|
|
51
|
+
}));
|
|
52
|
+
}
|
|
53
|
+
if (request.method === 'GET' && url.pathname.replace(/\/+$/, '') === '/twin/scenario') {
|
|
54
|
+
return Response.json(scenarioEngine ? scenarioEngine.status() : { vendor: 'ai-gateway', handlers: [], misses: 0, recentMisses: [] });
|
|
55
|
+
}
|
|
56
|
+
const path = url.pathname + url.search;
|
|
57
|
+
const body = request.method === 'GET' ? '' : await request.text();
|
|
58
|
+
const headers = Object.fromEntries(request.headers.entries());
|
|
59
|
+
|
|
60
|
+
// Streaming is signaled in the BODY on the OpenAI-compat surface (`stream: true`) and in a
|
|
61
|
+
// HEADER on the AI SDK gateway protocol (`ai-language-model-streaming: true`).
|
|
62
|
+
const isStreamingRequest = !readOnly && request.method.toUpperCase() === 'POST' && (
|
|
63
|
+
(url.pathname === '/v1/chat/completions' && wantsStream(body)) ||
|
|
64
|
+
(url.pathname === '/v3/ai/language-model' && request.headers.get('ai-language-model-streaming') === 'true')
|
|
65
|
+
);
|
|
66
|
+
|
|
67
|
+
if (isStreamingRequest) {
|
|
68
|
+
// Buffer the handler's SSE events first: a request that FAILS validation/routing must
|
|
69
|
+
// get its real 4xx status + JSON error body on the wire (never a 200 SSE stream
|
|
70
|
+
// wrapping an error — that would be a fake success).
|
|
71
|
+
const events: SseEvent[] = [];
|
|
72
|
+
const result = await handleAiGatewayTwinRequest({
|
|
73
|
+
method: request.method,
|
|
74
|
+
path,
|
|
75
|
+
body,
|
|
76
|
+
headers,
|
|
77
|
+
readOnly,
|
|
78
|
+
occurredAt: worldNow(),
|
|
79
|
+
...(options.root !== undefined ? { root: options.root } : {}),
|
|
80
|
+
...(scenarioEngine ? { scenarioEngine } : {}),
|
|
81
|
+
sseSink: (event) => events.push(event),
|
|
82
|
+
});
|
|
83
|
+
if (result.status >= 400) {
|
|
84
|
+
return new Response(JSON.stringify(result.body), { status: result.status, headers: { 'content-type': 'application/json' } });
|
|
85
|
+
}
|
|
86
|
+
const stream = new ReadableStream<Uint8Array>({
|
|
87
|
+
start(controller) {
|
|
88
|
+
const encoder = new TextEncoder();
|
|
89
|
+
for (const event of events) controller.enqueue(encoder.encode(encodeSse(event)));
|
|
90
|
+
controller.close();
|
|
91
|
+
},
|
|
92
|
+
});
|
|
93
|
+
return new Response(stream, {
|
|
94
|
+
headers: {
|
|
95
|
+
'content-type': 'text/event-stream; charset=utf-8',
|
|
96
|
+
'cache-control': 'no-cache',
|
|
97
|
+
},
|
|
98
|
+
});
|
|
99
|
+
}
|
|
100
|
+
|
|
101
|
+
const result = await handleAiGatewayTwinRequest({
|
|
102
|
+
method: request.method,
|
|
103
|
+
path,
|
|
104
|
+
body,
|
|
105
|
+
headers,
|
|
106
|
+
readOnly,
|
|
107
|
+
occurredAt: worldNow(),
|
|
108
|
+
...(options.root !== undefined ? { root: options.root } : {}),
|
|
109
|
+
...(scenarioEngine ? { scenarioEngine } : {}),
|
|
110
|
+
});
|
|
111
|
+
return new Response(JSON.stringify(result.body), { status: result.status, headers: { 'content-type': 'application/json' } });
|
|
112
|
+
};
|
|
113
|
+
}
|
|
114
|
+
|
|
115
|
+
export async function createAiGatewayTwinServer(options: { root?: string; port?: number; readOnly?: boolean; scenarioPath?: string } = {}): Promise<{ port: number; stop: () => void }> {
|
|
116
|
+
const server = await serveHttp({
|
|
117
|
+
port: options.port ?? 0,
|
|
118
|
+
idleTimeout: 60,
|
|
119
|
+
fetch: createAiGatewayTwinFetch(options),
|
|
120
|
+
});
|
|
121
|
+
return { port: server.port ?? options.port ?? 0, stop: () => server.stop(true) };
|
|
122
|
+
}
|