@volter/twin-xai 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +202 -0
- package/README.md +246 -0
- package/client/xai-device-auth.css +246 -0
- package/client/xai-device-auth.tsx +138 -0
- package/dist/client/xai-device-auth.bundle.js +18 -0
- package/dist/client/xai-device-auth.css +246 -0
- package/dist/client/xai-device-auth.d.ts +19 -0
- package/dist/client/xai-device-auth.js +50 -0
- package/dist/client/xai-device-auth.tsx +138 -0
- package/dist/src/cli.d.ts +2 -0
- package/dist/src/cli.js +28 -0
- package/dist/src/index.d.ts +17 -0
- package/dist/src/index.js +70 -0
- package/dist/src/xai-budget.d.ts +60 -0
- package/dist/src/xai-budget.js +139 -0
- package/dist/src/xai-capabilities.d.ts +4 -0
- package/dist/src/xai-capabilities.js +1072 -0
- package/dist/src/xai-conformance.d.ts +13 -0
- package/dist/src/xai-conformance.js +148 -0
- package/dist/src/xai-connector.d.ts +82 -0
- package/dist/src/xai-connector.js +174 -0
- package/dist/src/xai-device-auth-css.gen.d.ts +1 -0
- package/dist/src/xai-device-auth-css.gen.js +6 -0
- package/dist/src/xai-device-auth-ui.d.ts +13 -0
- package/dist/src/xai-device-auth-ui.js +72 -0
- package/dist/src/xai-models.d.ts +57 -0
- package/dist/src/xai-models.js +102 -0
- package/dist/src/xai-oauth.d.ts +30 -0
- package/dist/src/xai-oauth.js +279 -0
- package/dist/src/xai-scenario.d.ts +33 -0
- package/dist/src/xai-scenario.js +139 -0
- package/dist/src/xai-server.d.ts +36 -0
- package/dist/src/xai-server.js +232 -0
- package/dist/src/xai-stub.d.ts +69 -0
- package/dist/src/xai-stub.js +210 -0
- package/dist/src/xai-twin.d.ts +89 -0
- package/dist/src/xai-twin.js +883 -0
- package/dist/src/xai-types.d.ts +118 -0
- package/dist/src/xai-types.js +6 -0
- package/package.json +76 -0
- package/src/cli.ts +27 -0
- package/src/index.ts +120 -0
- package/src/xai-budget.ts +165 -0
- package/src/xai-capabilities.ts +1046 -0
- package/src/xai-conformance.ts +136 -0
- package/src/xai-connector.ts +212 -0
- package/src/xai-device-auth-css.gen.ts +6 -0
- package/src/xai-device-auth-ui.ts +90 -0
- package/src/xai-journey.uitest.ts +155 -0
- package/src/xai-models.ts +154 -0
- package/src/xai-oauth.ts +301 -0
- package/src/xai-scenario.ts +148 -0
- package/src/xai-server.ts +258 -0
- package/src/xai-stub.ts +213 -0
- package/src/xai-twin.ts +960 -0
- package/src/xai-types.ts +111 -0
|
@@ -0,0 +1,883 @@
|
|
|
1
|
+
// xAI (Grok) twin REQUEST HANDLER — the canonical xAI API surface for the twin.
|
|
2
|
+
// Contract: handleXaiTwinRequest({method, path, body}) -> {status, body}. It is the faithful
|
|
3
|
+
// xAI API the real `@ai-sdk/xai` provider / any OpenAI-compatible client (pointed at this
|
|
4
|
+
// baseURL) talks to UNMODIFIED.
|
|
5
|
+
//
|
|
6
|
+
// THE HONEST DESIGN: the twin cannot run Grok, so POST /v1/chat/completions, /v1/completions,
|
|
7
|
+
// and /v1/messages return a DETERMINISTIC STUB completion (xai-stub.ts) clearly labeled a twin
|
|
8
|
+
// stub — it NEVER pretends to be real model output. Live Search returns deterministic labeled
|
|
9
|
+
// citations (never real retrieval). But the ENTIRE PROTOCOL ENVELOPE is vendor-faithful:
|
|
10
|
+
// response shapes, streaming SSE chunks, tool_calls, finish_reason, reasoning_content /
|
|
11
|
+
// reasoning-token accounting, deterministic usage, and xAI's flat {code, error} error
|
|
12
|
+
// envelope. The genuinely stateful + static surface is real, not stubbed:
|
|
13
|
+
// • GET /v1/models (+ /:id) — static catalog + connector-observed models
|
|
14
|
+
// • GET /v1/language-models (+ /:id) — xAI-native rich catalog
|
|
15
|
+
// • GET /v1/image-generation-models (+ /:id) — image-model catalog
|
|
16
|
+
// • deferred completions — stateful (kernel action log): create →
|
|
17
|
+
// 202-pending poll → 200 result, consumed exactly once
|
|
18
|
+
// • GET /v1/api-key — deterministic key introspection
|
|
19
|
+
// • POST /v1/tokenize-text — deterministic pseudo-tokenizer
|
|
20
|
+
//
|
|
21
|
+
// State lives in the kernel action log (D1): deferred completions are local actions projected
|
|
22
|
+
// on read. No real xAI is ever called from this path (D4). Streaming uses an INJECTED sink —
|
|
23
|
+
// no real sockets / setTimeout (D5 verify is offline + deterministic).
|
|
24
|
+
import { applyTwinWrite, projectResources } from '@volter/world-core';
|
|
25
|
+
import { canonicalModelId, findImageGenerationModel, findLanguageModel, findModel, modelBehavior, XAI_IMAGE_GENERATION_MODELS, XAI_LANGUAGE_MODELS, XAI_MODELS, } from "./xai-models.js";
|
|
26
|
+
import { realizeXaiRespond } from "./xai-scenario.js";
|
|
27
|
+
import { decideXaiTwinDeviceAuthorization, exchangeXaiTwinToken, isCurrentXaiTwinAccessToken, recordXaiTwinUsage, startXaiTwinDeviceAuthorization, } from "./xai-oauth.js";
|
|
28
|
+
import { countPromptTokenDetails, estimateTokens, fnv1a, lastUserText, pseudoTokenize, stubAssistantText, stubCitations, stubJsonObject, stubReasoningText, stubToolArguments, stubToolCall, uuidFromSeed, } from "./xai-stub.js";
|
|
29
|
+
const SERVICE = 'xai';
|
|
30
|
+
// ── vendor-shaped errors: xAI's gRPC-transcoded FLAT {code, error} envelope ──────────────
|
|
31
|
+
function errBody(code, error) {
|
|
32
|
+
return { code, error };
|
|
33
|
+
}
|
|
34
|
+
function invalidArgument(message) {
|
|
35
|
+
return { status: 400, body: errBody('Client specified an invalid argument', message) };
|
|
36
|
+
}
|
|
37
|
+
function notFoundEntity(message) {
|
|
38
|
+
return { status: 404, body: errBody('Some requested entity was not found', message) };
|
|
39
|
+
}
|
|
40
|
+
function modelNotFound(model) {
|
|
41
|
+
return notFoundEntity(`The model ${model} does not exist or your team has no access to it. Consult https://docs.x.ai for available models.`);
|
|
42
|
+
}
|
|
43
|
+
function authError(message) {
|
|
44
|
+
// gRPC transcode of 401 is UNAUTHENTICATED — its canonical description, matching the
|
|
45
|
+
// convention every other status here follows (INVALID_ARGUMENT / NOT_FOUND /
|
|
46
|
+
// PERMISSION_DENIED / RESOURCE_EXHAUSTED). §9 caught the 401 wrongly reusing the 400's code.
|
|
47
|
+
return { status: 401, body: errBody('The request does not have valid authentication credentials for the operation', message) };
|
|
48
|
+
}
|
|
49
|
+
function blockedError(message) {
|
|
50
|
+
return { status: 403, body: errBody('The caller does not have permission', message) };
|
|
51
|
+
}
|
|
52
|
+
// ── modeled authentication (401/403) ────────────────────────────────────────────────────
|
|
53
|
+
// Real xAI requires a bearer credential on every request: missing/invalid → 401, a blocked
|
|
54
|
+
// key → 403. The twin can't validate against real keys, so it models the CHECKABLE failures:
|
|
55
|
+
// a missing credential, the reserved sentinel 'xai-invalid' (invalid-key path), and the
|
|
56
|
+
// reserved sentinel 'xai-blocked' (blocked-key path — still allowed to introspect itself via
|
|
57
|
+
// GET /v1/api-key, which is exactly what that endpoint is for). Any other non-empty key is
|
|
58
|
+
// accepted. Trusted in-process calls (verify/connector) carry NEITHER `headers` nor `apiKey`
|
|
59
|
+
// and are NOT auth-gated; the real SDK always sends a key → passes.
|
|
60
|
+
function presentedKey(req) {
|
|
61
|
+
const auth = req.headers?.['authorization'];
|
|
62
|
+
const bearer = typeof auth === 'string' && auth.toLowerCase().startsWith('bearer ') ? auth.slice(7).trim() : '';
|
|
63
|
+
return (req.apiKey ?? '').trim() || bearer;
|
|
64
|
+
}
|
|
65
|
+
function checkAuth(req, path) {
|
|
66
|
+
const key = presentedKey(req);
|
|
67
|
+
if (!key)
|
|
68
|
+
return authError('No API key provided. You can obtain an API key from https://console.x.ai.');
|
|
69
|
+
const tokenAuth = req.headers?.['x-xai-token-auth'];
|
|
70
|
+
const cliTransport = tokenAuth !== undefined || req.headers?.['x-grok-model-override'] !== undefined;
|
|
71
|
+
if (cliTransport && tokenAuth !== 'xai-grok-cli') {
|
|
72
|
+
return authError('The Grok CLI token-auth marker is invalid.');
|
|
73
|
+
}
|
|
74
|
+
if (cliTransport && req.requireTwinOauth && !isCurrentXaiTwinAccessToken(key, req.root, req.occurredAt)) {
|
|
75
|
+
return authError('The Grok CLI credential was not issued by this sealed xAI Twin world.');
|
|
76
|
+
}
|
|
77
|
+
if (key === 'xai-invalid')
|
|
78
|
+
return authError('Incorrect API key provided. You can obtain an API key from https://console.x.ai.');
|
|
79
|
+
if (key === 'xai-blocked' && path !== '/v1/api-key') {
|
|
80
|
+
return blockedError('Your API key is blocked. Consult your team administrator or the xAI console at https://console.x.ai.');
|
|
81
|
+
}
|
|
82
|
+
return null;
|
|
83
|
+
}
|
|
84
|
+
// ── modeled rate limiting (429) ─────────────────────────────────────────────────────────
|
|
85
|
+
// Rate limits are non-deterministic in production, so the twin exposes a DETERMINISTIC opt-in
|
|
86
|
+
// trigger: a request carrying `x-twin-force-rate-limit: 1` (or `true`) returns the faithful
|
|
87
|
+
// 429 envelope + retry-after. (No real timing/quotas — this is the checkable plumbing.)
|
|
88
|
+
function rateLimitError() {
|
|
89
|
+
return {
|
|
90
|
+
status: 429,
|
|
91
|
+
body: errBody('Resource has been exhausted', 'You have exceeded your team\'s rate limit for this model. Slow down your request rate or retry after the indicated delay.'),
|
|
92
|
+
headers: { 'retry-after': '1' },
|
|
93
|
+
};
|
|
94
|
+
}
|
|
95
|
+
function rateLimitTriggered(req) {
|
|
96
|
+
const v = req.headers?.['x-twin-force-rate-limit'];
|
|
97
|
+
return v === '1' || v === 'true';
|
|
98
|
+
}
|
|
99
|
+
function nowEpoch(occurredAt) {
|
|
100
|
+
return Math.floor((occurredAt ? Date.parse(occurredAt) : 0) / 1000);
|
|
101
|
+
}
|
|
102
|
+
function nowIso(occurredAt) {
|
|
103
|
+
return occurredAt ? new Date(Date.parse(occurredAt)).toISOString() : '1970-01-01T00:00:00.000Z';
|
|
104
|
+
}
|
|
105
|
+
// ── kernel helpers ──────────────────────────────────────────────────────────────────────
|
|
106
|
+
function rows(type, root) {
|
|
107
|
+
return projectResources(SERVICE, root).filter((r) => r.type === type);
|
|
108
|
+
}
|
|
109
|
+
function getRow(type, id, root) {
|
|
110
|
+
return rows(type, root).find((r) => r.id === id);
|
|
111
|
+
}
|
|
112
|
+
// ── request parsing ─────────────────────────────────────────────────────────────────────
|
|
113
|
+
function parseJson(body) {
|
|
114
|
+
if (!body || !body.trim())
|
|
115
|
+
return {};
|
|
116
|
+
try {
|
|
117
|
+
const v = JSON.parse(body);
|
|
118
|
+
return v && typeof v === 'object' ? v : {};
|
|
119
|
+
}
|
|
120
|
+
catch {
|
|
121
|
+
return {};
|
|
122
|
+
}
|
|
123
|
+
}
|
|
124
|
+
function parseForm(body) {
|
|
125
|
+
return new URLSearchParams(body ?? '');
|
|
126
|
+
}
|
|
127
|
+
const SYSTEM_FINGERPRINT = 'fp_twin_stub';
|
|
128
|
+
// ── Live Search parameters (xAI delta) ──────────────────────────────────────────────────
|
|
129
|
+
const SEARCH_MODES = new Set(['auto', 'on', 'off']);
|
|
130
|
+
const SEARCH_SOURCE_TYPES = new Set(['web', 'x', 'news', 'rss']);
|
|
131
|
+
function validateSearchParameters(raw) {
|
|
132
|
+
if (!raw || typeof raw !== 'object' || Array.isArray(raw)) {
|
|
133
|
+
return { error: invalidArgument('search_parameters must be an object') };
|
|
134
|
+
}
|
|
135
|
+
const sp = raw;
|
|
136
|
+
let mode = 'auto';
|
|
137
|
+
if (sp.mode !== undefined) {
|
|
138
|
+
if (typeof sp.mode !== 'string' || !SEARCH_MODES.has(sp.mode)) {
|
|
139
|
+
return { error: invalidArgument(`search_parameters.mode must be one of "auto", "on", "off"`) };
|
|
140
|
+
}
|
|
141
|
+
mode = sp.mode;
|
|
142
|
+
}
|
|
143
|
+
const sourceTypes = [];
|
|
144
|
+
if (sp.sources !== undefined) {
|
|
145
|
+
if (!Array.isArray(sp.sources))
|
|
146
|
+
return { error: invalidArgument('search_parameters.sources must be an array') };
|
|
147
|
+
for (const s of sp.sources) {
|
|
148
|
+
const t = s?.type;
|
|
149
|
+
if (typeof t !== 'string' || !SEARCH_SOURCE_TYPES.has(t)) {
|
|
150
|
+
return { error: invalidArgument(`search_parameters.sources[].type must be one of "web", "x", "news", "rss"`) };
|
|
151
|
+
}
|
|
152
|
+
sourceTypes.push(t);
|
|
153
|
+
}
|
|
154
|
+
}
|
|
155
|
+
const returnCitations = sp.return_citations !== false; // vendor default: true
|
|
156
|
+
return { search: { mode, sourceTypes, returnCitations } };
|
|
157
|
+
}
|
|
158
|
+
function validateChat(params) {
|
|
159
|
+
if (params.model === undefined || params.model === '')
|
|
160
|
+
return { error: invalidArgument('the model field is required') };
|
|
161
|
+
if (typeof params.model !== 'string')
|
|
162
|
+
return { error: invalidArgument('the model field must be a string') };
|
|
163
|
+
const behavior = modelBehavior(params.model);
|
|
164
|
+
if (!behavior) {
|
|
165
|
+
// an image-generation model cannot chat; anything else is an unknown entity (vendor 404).
|
|
166
|
+
if (findImageGenerationModel(params.model))
|
|
167
|
+
return { error: invalidArgument(`The model ${params.model} is an image generation model and does not support chat`) };
|
|
168
|
+
return { error: modelNotFound(params.model) };
|
|
169
|
+
}
|
|
170
|
+
const model = canonicalModelId(params.model);
|
|
171
|
+
if (!Array.isArray(params.messages))
|
|
172
|
+
return { error: invalidArgument('the messages field is required') };
|
|
173
|
+
if (params.messages.length === 0)
|
|
174
|
+
return { error: invalidArgument('the messages field must not be empty') };
|
|
175
|
+
const messages = params.messages;
|
|
176
|
+
for (const m of messages) {
|
|
177
|
+
if (!m || typeof m !== 'object' || typeof m.role !== 'string') {
|
|
178
|
+
return { error: invalidArgument('each message must have a valid role') };
|
|
179
|
+
}
|
|
180
|
+
}
|
|
181
|
+
let n = 1;
|
|
182
|
+
if (params.n !== undefined) {
|
|
183
|
+
n = Number(params.n);
|
|
184
|
+
if (!Number.isInteger(n) || n < 1)
|
|
185
|
+
return { error: invalidArgument('n must be an integer >= 1') };
|
|
186
|
+
}
|
|
187
|
+
// max_completion_tokens is the current name; max_tokens is the compatible alias.
|
|
188
|
+
const maxRaw = params.max_completion_tokens ?? params.max_tokens;
|
|
189
|
+
let maxTokens;
|
|
190
|
+
if (maxRaw !== undefined) {
|
|
191
|
+
maxTokens = Number(maxRaw);
|
|
192
|
+
if (!Number.isInteger(maxTokens) || maxTokens < 1)
|
|
193
|
+
return { error: invalidArgument('max_tokens must be an integer >= 1') };
|
|
194
|
+
}
|
|
195
|
+
const stopRaw = params.stop;
|
|
196
|
+
let stop;
|
|
197
|
+
if (stopRaw !== undefined) {
|
|
198
|
+
if (typeof stopRaw === 'string')
|
|
199
|
+
stop = [stopRaw];
|
|
200
|
+
else if (Array.isArray(stopRaw))
|
|
201
|
+
stop = stopRaw;
|
|
202
|
+
else
|
|
203
|
+
return { error: invalidArgument('stop must be a string or array of strings') };
|
|
204
|
+
}
|
|
205
|
+
// tool_choice: 'auto' | 'none' | 'required' | { type:'function', function:{ name } }.
|
|
206
|
+
let toolChoice;
|
|
207
|
+
const tcRaw = params.tool_choice;
|
|
208
|
+
if (tcRaw !== undefined) {
|
|
209
|
+
if (tcRaw === 'auto' || tcRaw === 'none' || tcRaw === 'required')
|
|
210
|
+
toolChoice = tcRaw;
|
|
211
|
+
else if (tcRaw && typeof tcRaw === 'object') {
|
|
212
|
+
const fn = tcRaw.function;
|
|
213
|
+
if (typeof fn?.name === 'string')
|
|
214
|
+
toolChoice = { name: fn.name };
|
|
215
|
+
else
|
|
216
|
+
return { error: invalidArgument('invalid tool_choice — a named choice requires function.name') };
|
|
217
|
+
}
|
|
218
|
+
else
|
|
219
|
+
return { error: invalidArgument('tool_choice must be "auto", "none", "required" or a named function') };
|
|
220
|
+
}
|
|
221
|
+
// response_format: { type:'text' | 'json_object' | 'json_schema', json_schema? } (structured outputs).
|
|
222
|
+
let responseFormat = { kind: 'text' };
|
|
223
|
+
const rf = params.response_format;
|
|
224
|
+
if (rf !== undefined) {
|
|
225
|
+
if (!rf || typeof rf !== 'object')
|
|
226
|
+
return { error: invalidArgument('response_format must be an object') };
|
|
227
|
+
const t = rf.type;
|
|
228
|
+
if (t === 'json_object')
|
|
229
|
+
responseFormat = { kind: 'json_object' };
|
|
230
|
+
else if (t === 'json_schema')
|
|
231
|
+
responseFormat = { kind: 'json_schema', schema: rf.json_schema };
|
|
232
|
+
else if (t === 'text' || t === undefined)
|
|
233
|
+
responseFormat = { kind: 'text' };
|
|
234
|
+
else
|
|
235
|
+
return { error: invalidArgument('response_format.type must be "text", "json_object", or "json_schema"') };
|
|
236
|
+
}
|
|
237
|
+
// stream_options.include_usage → emit a final usage-only chunk in the stream.
|
|
238
|
+
const so = params.stream_options;
|
|
239
|
+
const includeUsage = !!(so && typeof so === 'object' && so.include_usage === true);
|
|
240
|
+
// seed → reproducible sampling (the twin is already deterministic; the seed varies the id the
|
|
241
|
+
// way a real seed-distinct request does).
|
|
242
|
+
let seed;
|
|
243
|
+
if (params.seed !== undefined) {
|
|
244
|
+
seed = Number(params.seed);
|
|
245
|
+
if (!Number.isInteger(seed))
|
|
246
|
+
return { error: invalidArgument('seed must be an integer') };
|
|
247
|
+
}
|
|
248
|
+
// reasoning_effort — the xAI rule: ONLY grok-3-mini accepts it; grok-4-family models reason
|
|
249
|
+
// internally but REJECT the parameter (a real vendor 400), as do the non-reasoning models.
|
|
250
|
+
let reasoningEffort;
|
|
251
|
+
if (params.reasoning_effort !== undefined) {
|
|
252
|
+
if (params.reasoning_effort !== 'low' && params.reasoning_effort !== 'high') {
|
|
253
|
+
return { error: invalidArgument('reasoning_effort must be "low" or "high"') };
|
|
254
|
+
}
|
|
255
|
+
if (!behavior.supportsReasoningEffort) {
|
|
256
|
+
return { error: invalidArgument(`Argument not supported: reasoning_effort is not supported by model ${model}`) };
|
|
257
|
+
}
|
|
258
|
+
reasoningEffort = params.reasoning_effort;
|
|
259
|
+
}
|
|
260
|
+
// Live Search (xAI delta).
|
|
261
|
+
let search;
|
|
262
|
+
if (params.search_parameters !== undefined) {
|
|
263
|
+
const v = validateSearchParameters(params.search_parameters);
|
|
264
|
+
if ('error' in v)
|
|
265
|
+
return v;
|
|
266
|
+
search = v.search;
|
|
267
|
+
}
|
|
268
|
+
return {
|
|
269
|
+
args: {
|
|
270
|
+
model,
|
|
271
|
+
requestedModel: params.model,
|
|
272
|
+
messages,
|
|
273
|
+
tools: params.tools,
|
|
274
|
+
n,
|
|
275
|
+
...(maxTokens !== undefined ? { maxTokens } : {}),
|
|
276
|
+
...(stop !== undefined ? { stop } : {}),
|
|
277
|
+
stream: params.stream === true,
|
|
278
|
+
...(toolChoice !== undefined ? { toolChoice } : {}),
|
|
279
|
+
parallelToolCalls: params.parallel_tool_calls !== false,
|
|
280
|
+
responseFormat,
|
|
281
|
+
includeUsage,
|
|
282
|
+
...(seed !== undefined ? { seed } : {}),
|
|
283
|
+
...(reasoningEffort !== undefined ? { reasoningEffort } : {}),
|
|
284
|
+
...(search !== undefined ? { search } : {}),
|
|
285
|
+
deferred: params.deferred === true,
|
|
286
|
+
},
|
|
287
|
+
};
|
|
288
|
+
}
|
|
289
|
+
// A deterministic id seed from the request (so ids are stable + assertable). Hash of the
|
|
290
|
+
// prompt + model + seed (a distinct sampling seed changes the id like a real distinct request).
|
|
291
|
+
function stableSeed(args) {
|
|
292
|
+
return `chat|${args.model}|${JSON.stringify(args.messages)}${args.seed !== undefined ? `|seed=${args.seed}` : ''}`;
|
|
293
|
+
}
|
|
294
|
+
// Deterministic reasoning-token budgets: grok-3-mini spends per requested effort; the
|
|
295
|
+
// grok-4-family reasoning models spend a fixed internal budget (content never exposed).
|
|
296
|
+
const REASONING_BUDGET = { low: 16, high: 64 };
|
|
297
|
+
const INTERNAL_REASONING_TOKENS = 32;
|
|
298
|
+
function emptyUsageDetails() {
|
|
299
|
+
return {
|
|
300
|
+
prompt_tokens: 0,
|
|
301
|
+
completion_tokens: 0,
|
|
302
|
+
total_tokens: 0,
|
|
303
|
+
prompt_tokens_details: { text_tokens: 0, audio_tokens: 0, image_tokens: 0, cached_tokens: 0 },
|
|
304
|
+
completion_tokens_details: { reasoning_tokens: 0, audio_tokens: 0, accepted_prediction_tokens: 0, rejected_prediction_tokens: 0 },
|
|
305
|
+
num_sources_used: 0,
|
|
306
|
+
};
|
|
307
|
+
}
|
|
308
|
+
// Build ONE deterministic stub choice (index `idx`). Honors tools (tool_calls +
|
|
309
|
+
// finish_reason:'tool_calls'), tool_choice (none/required/named), parallel_tool_calls,
|
|
310
|
+
// response_format (structured outputs), max_tokens, stop sequences, and reasoning_content
|
|
311
|
+
// (grok-3-mini). A scripted scenario result (chat call only) overrides the generic stub.
|
|
312
|
+
function buildChoice(args, idx, scripted) {
|
|
313
|
+
const behavior = modelBehavior(args.model);
|
|
314
|
+
const reasoningTokens = behavior.supportsReasoningEffort
|
|
315
|
+
? (REASONING_BUDGET[args.reasoningEffort ?? 'low'] ?? 16)
|
|
316
|
+
: behavior.isReasoningModel ? INTERNAL_REASONING_TOKENS : 0;
|
|
317
|
+
// grok-3-mini exposes its (stubbed) reasoning trace; grok-4-family never does.
|
|
318
|
+
const reasoningContent = behavior.supportsReasoningEffort
|
|
319
|
+
? stubReasoningText(args.messages, args.model, args.reasoningEffort ?? 'low')
|
|
320
|
+
: undefined;
|
|
321
|
+
if (scripted) {
|
|
322
|
+
if (scripted.toolCalls.length) {
|
|
323
|
+
return {
|
|
324
|
+
choice: {
|
|
325
|
+
index: idx,
|
|
326
|
+
message: { role: 'assistant', content: scripted.text, ...(reasoningContent !== undefined ? { reasoning_content: reasoningContent } : {}), tool_calls: scripted.toolCalls, refusal: null },
|
|
327
|
+
logprobs: null,
|
|
328
|
+
finish_reason: scripted.finishReason,
|
|
329
|
+
},
|
|
330
|
+
completionTokens: estimateTokens(JSON.stringify(scripted.toolCalls) + (scripted.text ?? '')),
|
|
331
|
+
reasoningTokens,
|
|
332
|
+
};
|
|
333
|
+
}
|
|
334
|
+
const text = scripted.text ?? '';
|
|
335
|
+
return {
|
|
336
|
+
choice: { index: idx, message: { role: 'assistant', content: text, ...(reasoningContent !== undefined ? { reasoning_content: reasoningContent } : {}), refusal: null }, logprobs: null, finish_reason: scripted.finishReason },
|
|
337
|
+
completionTokens: estimateTokens(text),
|
|
338
|
+
reasoningTokens,
|
|
339
|
+
};
|
|
340
|
+
}
|
|
341
|
+
const toolsSource = args.tools;
|
|
342
|
+
const hasTools = Array.isArray(toolsSource) && toolsSource.length > 0;
|
|
343
|
+
// tool_choice gates whether the stub calls a tool: 'none' forbids it; a named/required choice
|
|
344
|
+
// forces it; 'auto'/default calls when tools exist.
|
|
345
|
+
const forbidTools = args.toolChoice === 'none';
|
|
346
|
+
const forcedName = typeof args.toolChoice === 'object' ? args.toolChoice.name : undefined;
|
|
347
|
+
if (hasTools && !forbidTools) {
|
|
348
|
+
const list = toolsSource;
|
|
349
|
+
const calls = [];
|
|
350
|
+
if (forcedName || args.parallelToolCalls === false) {
|
|
351
|
+
const tc = stubToolCall(toolsSource, idx + 1, forcedName);
|
|
352
|
+
if (tc)
|
|
353
|
+
calls.push(tc);
|
|
354
|
+
}
|
|
355
|
+
else {
|
|
356
|
+
for (let t = 0; t < list.length; t++) {
|
|
357
|
+
const tc = stubToolCall([list[t]], idx * 100 + t + 1);
|
|
358
|
+
if (tc)
|
|
359
|
+
calls.push(tc);
|
|
360
|
+
}
|
|
361
|
+
}
|
|
362
|
+
if (calls.length) {
|
|
363
|
+
return {
|
|
364
|
+
choice: {
|
|
365
|
+
index: idx,
|
|
366
|
+
message: { role: 'assistant', content: null, ...(reasoningContent !== undefined ? { reasoning_content: reasoningContent } : {}), tool_calls: calls, refusal: null },
|
|
367
|
+
logprobs: null,
|
|
368
|
+
finish_reason: 'tool_calls',
|
|
369
|
+
},
|
|
370
|
+
completionTokens: estimateTokens(JSON.stringify(calls)),
|
|
371
|
+
reasoningTokens,
|
|
372
|
+
};
|
|
373
|
+
}
|
|
374
|
+
}
|
|
375
|
+
// structured outputs: when response_format requests json_object/json_schema, the content is
|
|
376
|
+
// valid JSON (json_schema fills every declared property).
|
|
377
|
+
let text = args.responseFormat.kind === 'json_object'
|
|
378
|
+
? stubJsonObject(args.messages, args.model)
|
|
379
|
+
: args.responseFormat.kind === 'json_schema'
|
|
380
|
+
? stubJsonObject(args.messages, args.model, args.responseFormat.schema)
|
|
381
|
+
: stubAssistantText(args.messages, args.model);
|
|
382
|
+
let finish = 'stop';
|
|
383
|
+
// Truncate at the EARLIEST-occurring stop sequence across the whole `stop` list.
|
|
384
|
+
let stopAt = -1;
|
|
385
|
+
for (const s of args.stop ?? []) {
|
|
386
|
+
if (!s)
|
|
387
|
+
continue;
|
|
388
|
+
const i = text.indexOf(s);
|
|
389
|
+
if (i >= 0 && (stopAt < 0 || i < stopAt))
|
|
390
|
+
stopAt = i;
|
|
391
|
+
}
|
|
392
|
+
if (stopAt >= 0)
|
|
393
|
+
text = text.slice(0, stopAt);
|
|
394
|
+
if (args.maxTokens !== undefined && estimateTokens(text) > args.maxTokens) {
|
|
395
|
+
text = text.slice(0, args.maxTokens * 4);
|
|
396
|
+
finish = 'length';
|
|
397
|
+
}
|
|
398
|
+
return {
|
|
399
|
+
choice: { index: idx, message: { role: 'assistant', content: text, ...(reasoningContent !== undefined ? { reasoning_content: reasoningContent } : {}), refusal: null }, logprobs: null, finish_reason: finish },
|
|
400
|
+
completionTokens: estimateTokens(text),
|
|
401
|
+
reasoningTokens,
|
|
402
|
+
};
|
|
403
|
+
}
|
|
404
|
+
export function buildChatCompletion(args, occurredAt, scenarioEngine) {
|
|
405
|
+
// Scenario handlers: chat-completions calls only; a fired handler scripts the assistant
|
|
406
|
+
// turn (the envelope below stays vendor-faithful either way); a miss teaches in the stub.
|
|
407
|
+
let missTeach = '';
|
|
408
|
+
const scripted = scenarioEngine
|
|
409
|
+
? (() => {
|
|
410
|
+
const decision = scenarioEngine.next({ model: args.requestedModel, messages: args.messages, tools: args.tools });
|
|
411
|
+
if (decision.kind === 'handler')
|
|
412
|
+
return realizeXaiRespond(decision.respond);
|
|
413
|
+
missTeach = `\n[twin-scenario miss — no handler matched. Author one in the world dir's handlers/xai.json (GET /twin explains; GET /twin/scenario lists handlers + misses). Features seen: ${JSON.stringify(decision.miss.features)}]`;
|
|
414
|
+
return null;
|
|
415
|
+
})()
|
|
416
|
+
: null;
|
|
417
|
+
const promptDetails = countPromptTokenDetails(args.messages);
|
|
418
|
+
const promptTokens = promptDetails.text + promptDetails.image;
|
|
419
|
+
const choices = [];
|
|
420
|
+
let completionTokens = 0;
|
|
421
|
+
let reasoningTokens = 0;
|
|
422
|
+
for (let i = 0; i < args.n; i++) {
|
|
423
|
+
const { choice, completionTokens: ct, reasoningTokens: rt } = buildChoice(args, i, scripted);
|
|
424
|
+
choices.push(choice);
|
|
425
|
+
completionTokens += ct;
|
|
426
|
+
reasoningTokens += rt;
|
|
427
|
+
}
|
|
428
|
+
if (missTeach) {
|
|
429
|
+
const first = choices[0];
|
|
430
|
+
if (first?.message && typeof first.message.content === 'string')
|
|
431
|
+
first.message.content += missTeach;
|
|
432
|
+
}
|
|
433
|
+
// Live Search: mode 'off' disables; 'on'/'auto' consult the (stubbed, labeled) sources. The
|
|
434
|
+
// twin treats 'auto' as searching so the behavior is deterministic — documented in README.
|
|
435
|
+
const searching = args.search !== undefined && args.search.mode !== 'off';
|
|
436
|
+
const citations = searching ? stubCitations(lastUserText(args.messages), args.search.sourceTypes) : [];
|
|
437
|
+
const usage = {
|
|
438
|
+
prompt_tokens: promptTokens,
|
|
439
|
+
completion_tokens: completionTokens + reasoningTokens,
|
|
440
|
+
total_tokens: promptTokens + completionTokens + reasoningTokens,
|
|
441
|
+
prompt_tokens_details: { text_tokens: promptDetails.text, audio_tokens: 0, image_tokens: promptDetails.image, cached_tokens: 0 },
|
|
442
|
+
completion_tokens_details: { reasoning_tokens: reasoningTokens, audio_tokens: 0, accepted_prediction_tokens: 0, rejected_prediction_tokens: 0 },
|
|
443
|
+
num_sources_used: citations.length,
|
|
444
|
+
};
|
|
445
|
+
return {
|
|
446
|
+
id: uuidFromSeed(stableSeed(args)),
|
|
447
|
+
object: 'chat.completion',
|
|
448
|
+
created: nowEpoch(occurredAt),
|
|
449
|
+
model: args.model,
|
|
450
|
+
choices,
|
|
451
|
+
usage,
|
|
452
|
+
system_fingerprint: SYSTEM_FINGERPRINT,
|
|
453
|
+
...(searching && args.search.returnCitations ? { citations } : {}),
|
|
454
|
+
};
|
|
455
|
+
}
|
|
456
|
+
// Split text into deterministic streaming chunks (≤ ~20 chars each), preserving order.
|
|
457
|
+
function chunkText(text) {
|
|
458
|
+
if (!text)
|
|
459
|
+
return [];
|
|
460
|
+
const out = [];
|
|
461
|
+
for (let i = 0; i < text.length; i += 20)
|
|
462
|
+
out.push(text.slice(i, i + 20));
|
|
463
|
+
return out;
|
|
464
|
+
}
|
|
465
|
+
/**
|
|
466
|
+
* Emit the vendor-faithful chat streaming sequence into the injected sink (NO sockets, NO
|
|
467
|
+
* setTimeout). Real order: a first chunk with `delta:{role:'assistant'}`, then
|
|
468
|
+
* `delta:{content}` chunks (or tool_calls deltas), then a final chunk with `finish_reason`,
|
|
469
|
+
* then a usage-only chunk when stream_options.include_usage, then `[DONE]`. Deterministic +
|
|
470
|
+
* synchronous so a collector can assert the full sequence.
|
|
471
|
+
*/
|
|
472
|
+
export function streamChat(args, sink, occurredAt, scenarioEngine) {
|
|
473
|
+
const full = buildChatCompletion(args, occurredAt, scenarioEngine);
|
|
474
|
+
const base = { id: full.id, object: 'chat.completion.chunk', created: full.created, model: full.model, system_fingerprint: SYSTEM_FINGERPRINT };
|
|
475
|
+
for (const choice of full.choices) {
|
|
476
|
+
const idx = choice.index;
|
|
477
|
+
sink({ data: { ...base, choices: [{ index: idx, delta: { role: 'assistant', content: '' }, logprobs: null, finish_reason: null }] } });
|
|
478
|
+
if (choice.message.tool_calls && choice.message.tool_calls.length) {
|
|
479
|
+
// xAI DELTA vs OpenAI: each streamed tool call arrives as ONE COMPLETE delta (id, type,
|
|
480
|
+
// function.name AND the full arguments in a single chunk) — xAI does not split a call's
|
|
481
|
+
// arguments across deltas. The real @ai-sdk/xai chunk schema REQUIRES every tool_calls
|
|
482
|
+
// delta to carry all of these fields, so an OpenAI-style split-delta stream fails its
|
|
483
|
+
// validation (observed against @ai-sdk/xai 2.0.x).
|
|
484
|
+
choice.message.tool_calls.forEach((tc, tIdx) => {
|
|
485
|
+
sink({ data: { ...base, choices: [{ index: idx, delta: { tool_calls: [{ index: tIdx, id: tc.id, type: 'function', function: { name: tc.function.name, arguments: tc.function.arguments } }] }, logprobs: null, finish_reason: null }] } });
|
|
486
|
+
});
|
|
487
|
+
}
|
|
488
|
+
else {
|
|
489
|
+
for (const piece of chunkText(choice.message.content ?? '')) {
|
|
490
|
+
sink({ data: { ...base, choices: [{ index: idx, delta: { content: piece }, logprobs: null, finish_reason: null }] } });
|
|
491
|
+
}
|
|
492
|
+
}
|
|
493
|
+
sink({ data: { ...base, choices: [{ index: idx, delta: {}, logprobs: null, finish_reason: choice.finish_reason }], ...(full.citations ? { citations: full.citations } : {}) } });
|
|
494
|
+
}
|
|
495
|
+
if (args.includeUsage) {
|
|
496
|
+
sink({ data: { ...base, choices: [], usage: full.usage } });
|
|
497
|
+
}
|
|
498
|
+
sink({ done: true });
|
|
499
|
+
return full;
|
|
500
|
+
}
|
|
501
|
+
// ── deferred completions (xAI delta: deferred:true → request_id → poll → result, once) ────
|
|
502
|
+
// The vendor runs the inference asynchronously: POST returns { request_id }, and
|
|
503
|
+
// GET /v1/chat/deferred-completion/{request_id} returns 202 while pending and 200 with the
|
|
504
|
+
// chat completion when done — and the result can be RETRIEVED ONLY ONCE. The twin has nothing
|
|
505
|
+
// to run in the background, so it persists the request (kernel action log) and models the
|
|
506
|
+
// lifecycle deterministically: the FIRST poll is the pending 202, the SECOND poll serves the
|
|
507
|
+
// completed result and consumes it, and any later poll (or an unknown id) is the vendor 404.
|
|
508
|
+
async function createDeferredCompletion(args, req) {
|
|
509
|
+
const seq = rows('deferred_completion', req.root).length + 1;
|
|
510
|
+
const requestId = uuidFromSeed(`deferred|${seq}|${stableSeed(args)}`);
|
|
511
|
+
await applyTwinWrite(SERVICE, {
|
|
512
|
+
operation: 'deferred_completion.create',
|
|
513
|
+
subjectType: 'deferred_completion',
|
|
514
|
+
subjectId: requestId,
|
|
515
|
+
fields: { object: 'deferred_completion', status: 'pending', polls: 0, consumed: false, created_at: nowEpoch(req.occurredAt), _args: JSON.stringify(args) },
|
|
516
|
+
...(req.occurredAt ? { occurredAt: req.occurredAt } : {}),
|
|
517
|
+
actor: { kind: 'agent' },
|
|
518
|
+
}, req.root);
|
|
519
|
+
return { status: 200, body: { request_id: requestId } };
|
|
520
|
+
}
|
|
521
|
+
async function retrieveDeferredCompletion(requestId, req) {
|
|
522
|
+
const row = getRow('deferred_completion', requestId, req.root);
|
|
523
|
+
if (!row || row.consumed === true) {
|
|
524
|
+
return notFoundEntity(`The deferred completion ${requestId} does not exist, has expired, or has already been retrieved.`);
|
|
525
|
+
}
|
|
526
|
+
if (Number(row.polls ?? 0) === 0) {
|
|
527
|
+
// first poll → still pending (202, empty body), like the vendor mid-inference.
|
|
528
|
+
if (!req.readOnly) {
|
|
529
|
+
await applyTwinWrite(SERVICE, {
|
|
530
|
+
operation: 'deferred_completion.poll',
|
|
531
|
+
subjectType: 'deferred_completion',
|
|
532
|
+
subjectId: requestId,
|
|
533
|
+
fields: { polls: 1 },
|
|
534
|
+
...(req.occurredAt ? { occurredAt: req.occurredAt } : {}),
|
|
535
|
+
actor: { kind: 'agent' },
|
|
536
|
+
}, req.root);
|
|
537
|
+
}
|
|
538
|
+
return { status: 202, body: {} };
|
|
539
|
+
}
|
|
540
|
+
const args = JSON.parse(String(row._args ?? '{}'));
|
|
541
|
+
const completion = buildChatCompletion(args, req.occurredAt);
|
|
542
|
+
// A READ-ONLY twin must never mutate the kernel log — serving the result is a read; the
|
|
543
|
+
// consume is a WRITE, so it is skipped (the lifecycle simply cannot advance in a read-only
|
|
544
|
+
// mirror, and repeated read-only polls keep serving the still-unconsumed result). §9 found
|
|
545
|
+
// this consume unguarded — a read-only GET was consuming the row.
|
|
546
|
+
if (!req.readOnly) {
|
|
547
|
+
await applyTwinWrite(SERVICE, {
|
|
548
|
+
operation: 'deferred_completion.consume',
|
|
549
|
+
subjectType: 'deferred_completion',
|
|
550
|
+
subjectId: requestId,
|
|
551
|
+
fields: { status: 'done', consumed: true },
|
|
552
|
+
...(req.occurredAt ? { occurredAt: req.occurredAt } : {}),
|
|
553
|
+
actor: { kind: 'agent' },
|
|
554
|
+
}, req.root);
|
|
555
|
+
}
|
|
556
|
+
return { status: 200, body: completion };
|
|
557
|
+
}
|
|
558
|
+
// ── legacy completions (POST /v1/completions) ───────────────────────────────────────────
|
|
559
|
+
function handleLegacyCompletions(params, occurredAt) {
|
|
560
|
+
if (params.model === undefined || params.model === '')
|
|
561
|
+
return invalidArgument('the model field is required');
|
|
562
|
+
if (typeof params.model !== 'string')
|
|
563
|
+
return invalidArgument('the model field must be a string');
|
|
564
|
+
if (!modelBehavior(params.model))
|
|
565
|
+
return modelNotFound(params.model);
|
|
566
|
+
const model = canonicalModelId(params.model);
|
|
567
|
+
if (params.prompt === undefined)
|
|
568
|
+
return invalidArgument('the prompt field is required');
|
|
569
|
+
const prompt = typeof params.prompt === 'string'
|
|
570
|
+
? params.prompt
|
|
571
|
+
: Array.isArray(params.prompt) ? params.prompt.map(String).join('\n') : '';
|
|
572
|
+
const messages = [{ role: 'user', content: prompt }];
|
|
573
|
+
let text = stubAssistantText(messages, model);
|
|
574
|
+
let finish = 'stop';
|
|
575
|
+
const maxRaw = params.max_tokens;
|
|
576
|
+
if (maxRaw !== undefined) {
|
|
577
|
+
const maxTokens = Number(maxRaw);
|
|
578
|
+
if (!Number.isInteger(maxTokens) || maxTokens < 1)
|
|
579
|
+
return invalidArgument('max_tokens must be an integer >= 1');
|
|
580
|
+
if (estimateTokens(text) > maxTokens) {
|
|
581
|
+
text = text.slice(0, maxTokens * 4);
|
|
582
|
+
finish = 'length';
|
|
583
|
+
}
|
|
584
|
+
}
|
|
585
|
+
const promptTokens = estimateTokens(prompt);
|
|
586
|
+
const completionTokens = estimateTokens(text);
|
|
587
|
+
const usage = emptyUsageDetails();
|
|
588
|
+
usage.prompt_tokens = promptTokens;
|
|
589
|
+
usage.prompt_tokens_details.text_tokens = promptTokens;
|
|
590
|
+
usage.completion_tokens = completionTokens;
|
|
591
|
+
usage.total_tokens = promptTokens + completionTokens;
|
|
592
|
+
const out = {
|
|
593
|
+
id: uuidFromSeed(`cmpl|${model}|${prompt}`),
|
|
594
|
+
object: 'text_completion',
|
|
595
|
+
created: nowEpoch(occurredAt),
|
|
596
|
+
model,
|
|
597
|
+
choices: [{ text, index: 0, logprobs: null, finish_reason: finish }],
|
|
598
|
+
usage,
|
|
599
|
+
system_fingerprint: SYSTEM_FINGERPRINT,
|
|
600
|
+
};
|
|
601
|
+
return { status: 200, body: out };
|
|
602
|
+
}
|
|
603
|
+
// ── Anthropic-compatible Messages (POST /v1/messages) ───────────────────────────────────
|
|
604
|
+
// xAI serves an Anthropic-style Messages endpoint so Anthropic-shaped clients can drive Grok.
|
|
605
|
+
// The twin models the request rules (model / messages / REQUIRED max_tokens) and the response
|
|
606
|
+
// envelope (content blocks, stop_reason, usage) faithfully; the text is the labeled stub and
|
|
607
|
+
// Anthropic-style tools yield a faithful tool_use block.
|
|
608
|
+
function handleMessagesCompat(params, occurredAt) {
|
|
609
|
+
if (params.model === undefined || params.model === '')
|
|
610
|
+
return invalidArgument('the model field is required');
|
|
611
|
+
if (typeof params.model !== 'string')
|
|
612
|
+
return invalidArgument('the model field must be a string');
|
|
613
|
+
if (!modelBehavior(params.model))
|
|
614
|
+
return modelNotFound(params.model);
|
|
615
|
+
const model = canonicalModelId(params.model);
|
|
616
|
+
if (!Array.isArray(params.messages) || params.messages.length === 0)
|
|
617
|
+
return invalidArgument('the messages field is required and must not be empty');
|
|
618
|
+
const maxTokens = Number(params.max_tokens);
|
|
619
|
+
if (params.max_tokens === undefined || !Number.isInteger(maxTokens) || maxTokens < 1) {
|
|
620
|
+
return invalidArgument('the max_tokens field is required and must be a positive integer');
|
|
621
|
+
}
|
|
622
|
+
const messages = params.messages;
|
|
623
|
+
const system = typeof params.system === 'string' ? params.system : '';
|
|
624
|
+
const inputTokens = estimateTokens(system) + countPromptTokenDetails(messages).text + countPromptTokenDetails(messages).image;
|
|
625
|
+
const content = [];
|
|
626
|
+
let stopReason = 'end_turn';
|
|
627
|
+
const tools = Array.isArray(params.tools) ? params.tools : [];
|
|
628
|
+
if (tools.length > 0) {
|
|
629
|
+
const first = tools[0];
|
|
630
|
+
const name = typeof first.name === 'string' ? first.name : 'unknown_tool';
|
|
631
|
+
// Anthropic tools carry `input_schema` (not `function.parameters`); stubToolArguments
|
|
632
|
+
// understands both, so tool_use.input validates against the declared schema.
|
|
633
|
+
let input = {};
|
|
634
|
+
try {
|
|
635
|
+
input = JSON.parse(stubToolArguments(first));
|
|
636
|
+
}
|
|
637
|
+
catch {
|
|
638
|
+
input = {};
|
|
639
|
+
}
|
|
640
|
+
content.push({ type: 'tool_use', id: `toolu_twin_${fnv1a(name).toString(36)}`, name, input });
|
|
641
|
+
stopReason = 'tool_use';
|
|
642
|
+
}
|
|
643
|
+
else {
|
|
644
|
+
let text = stubAssistantText(messages, model);
|
|
645
|
+
if (estimateTokens(text) > maxTokens) {
|
|
646
|
+
text = text.slice(0, maxTokens * 4);
|
|
647
|
+
stopReason = 'max_tokens';
|
|
648
|
+
}
|
|
649
|
+
content.push({ type: 'text', text });
|
|
650
|
+
}
|
|
651
|
+
const outputTokens = estimateTokens(JSON.stringify(content));
|
|
652
|
+
const out = {
|
|
653
|
+
id: `msg_twin_${fnv1a(`${model}|${JSON.stringify(messages)}`).toString(36)}`,
|
|
654
|
+
type: 'message',
|
|
655
|
+
role: 'assistant',
|
|
656
|
+
model,
|
|
657
|
+
content,
|
|
658
|
+
stop_reason: stopReason,
|
|
659
|
+
stop_sequence: null,
|
|
660
|
+
usage: { input_tokens: inputTokens, output_tokens: outputTokens },
|
|
661
|
+
};
|
|
662
|
+
return { status: 200, body: out };
|
|
663
|
+
}
|
|
664
|
+
// ── image generations (POST /v1/images/generations) ─────────────────────────────────────
|
|
665
|
+
// xAI's image endpoint is OpenAI-shaped but REJECTS OpenAI's size/quality/style parameters
|
|
666
|
+
// (a real vendor 400 — a faithful compatibility delta). The response shape (data[].url |
|
|
667
|
+
// b64_json + revised_prompt) is faithful with clearly-labeled stub values.
|
|
668
|
+
function handleImageGenerations(params, occurredAt) {
|
|
669
|
+
for (const unsupported of ['size', 'quality', 'style']) {
|
|
670
|
+
if (params[unsupported] !== undefined)
|
|
671
|
+
return invalidArgument(`Argument not supported: ${unsupported}`);
|
|
672
|
+
}
|
|
673
|
+
if (params.prompt === undefined || params.prompt === '')
|
|
674
|
+
return invalidArgument('the prompt field is required');
|
|
675
|
+
const modelRaw = typeof params.model === 'string' && params.model ? params.model : 'grok-2-image-1212';
|
|
676
|
+
const imageModel = findImageGenerationModel(modelRaw);
|
|
677
|
+
if (!imageModel) {
|
|
678
|
+
if (modelBehavior(modelRaw))
|
|
679
|
+
return invalidArgument(`The model ${modelRaw} is not an image generation model`);
|
|
680
|
+
return modelNotFound(modelRaw);
|
|
681
|
+
}
|
|
682
|
+
let n = 1;
|
|
683
|
+
if (params.n !== undefined) {
|
|
684
|
+
n = Number(params.n);
|
|
685
|
+
if (!Number.isInteger(n) || n < 1 || n > 10)
|
|
686
|
+
return invalidArgument('n must be an integer between 1 and 10');
|
|
687
|
+
}
|
|
688
|
+
const format = params.response_format ?? 'url';
|
|
689
|
+
if (format !== 'url' && format !== 'b64_json')
|
|
690
|
+
return invalidArgument('response_format must be "url" or "b64_json"');
|
|
691
|
+
const prompt = String(params.prompt);
|
|
692
|
+
const data = Array.from({ length: n }, (_, i) => {
|
|
693
|
+
const seed = fnv1a(`${prompt}|${i}`).toString(36);
|
|
694
|
+
const revised = `[twin-stub:${imageModel.id}] ${prompt}`;
|
|
695
|
+
if (format === 'b64_json') {
|
|
696
|
+
const marker = `[twin-stub-image] no real pixels; prompt: ${prompt} (#${i})`;
|
|
697
|
+
return { b64_json: Buffer.from(marker, 'utf8').toString('base64'), revised_prompt: revised };
|
|
698
|
+
}
|
|
699
|
+
return { url: `https://twin.invalid/xai-image-stub/${seed}-${i}.png`, revised_prompt: revised };
|
|
700
|
+
});
|
|
701
|
+
return { status: 200, body: { created: nowEpoch(occurredAt), data } };
|
|
702
|
+
}
|
|
703
|
+
// ── api-key introspection (GET /v1/api-key) ─────────────────────────────────────────────
|
|
704
|
+
// Returns metadata about the PRESENTED key (redacted value, acls, blocked/disabled flags).
|
|
705
|
+
// Keys are SYNTHETIC twin values — never real xAI credentials. The reserved sentinel
|
|
706
|
+
// 'xai-blocked' reports api_key_blocked:true (and every other route 403s for it).
|
|
707
|
+
function handleApiKey(req) {
|
|
708
|
+
const key = presentedKey(req) || 'xai-twin-trusted';
|
|
709
|
+
const redacted = key.length > 8 ? `${key.slice(0, 4)}...${key.slice(-4)}` : `${key.slice(0, 2)}...`;
|
|
710
|
+
return {
|
|
711
|
+
status: 200,
|
|
712
|
+
body: {
|
|
713
|
+
redacted_api_key: redacted,
|
|
714
|
+
user_id: uuidFromSeed(`user|${key}`),
|
|
715
|
+
name: 'twin api key',
|
|
716
|
+
create_time: nowIso(req.occurredAt),
|
|
717
|
+
modify_time: nowIso(req.occurredAt),
|
|
718
|
+
modified_by: uuidFromSeed(`user|${key}`),
|
|
719
|
+
team_id: uuidFromSeed(`team|${key}`),
|
|
720
|
+
acls: ['api-key:model:*', 'api-key:endpoint:*'],
|
|
721
|
+
api_key_id: uuidFromSeed(`key|${key}`),
|
|
722
|
+
team_blocked: false,
|
|
723
|
+
api_key_blocked: key === 'xai-blocked',
|
|
724
|
+
api_key_disabled: false,
|
|
725
|
+
},
|
|
726
|
+
};
|
|
727
|
+
}
|
|
728
|
+
// ── tokenize-text (POST /v1/tokenize-text) ──────────────────────────────────────────────
|
|
729
|
+
function handleTokenizeText(params) {
|
|
730
|
+
if (params.text === undefined || typeof params.text !== 'string' || params.text === '') {
|
|
731
|
+
return invalidArgument('the text field is required');
|
|
732
|
+
}
|
|
733
|
+
if (params.model === undefined || params.model === '')
|
|
734
|
+
return invalidArgument('the model field is required');
|
|
735
|
+
if (typeof params.model !== 'string' || !modelBehavior(params.model))
|
|
736
|
+
return modelNotFound(String(params.model));
|
|
737
|
+
return { status: 200, body: { token_ids: pseudoTokenize(params.text) } };
|
|
738
|
+
}
|
|
739
|
+
// ── models catalogs (static + connector-observed) ───────────────────────────────────────
|
|
740
|
+
function observedModels(root) {
|
|
741
|
+
const catalogIds = new Set(XAI_MODELS.map((m) => m.id));
|
|
742
|
+
return rows('model', root)
|
|
743
|
+
.filter((r) => !catalogIds.has(String(r.id)))
|
|
744
|
+
.map((r) => ({ id: String(r.id), object: 'model', created: Number(r.created ?? 0), owned_by: String(r.owned_by ?? 'xai') }));
|
|
745
|
+
}
|
|
746
|
+
function observedLanguageModels(root) {
|
|
747
|
+
const catalogIds = new Set(XAI_LANGUAGE_MODELS.map((m) => m.id));
|
|
748
|
+
return rows('language_model', root)
|
|
749
|
+
.filter((r) => !catalogIds.has(String(r.id)))
|
|
750
|
+
.map((r) => {
|
|
751
|
+
const { type: _t, updatedAt: _u, ...rest } = r;
|
|
752
|
+
return rest;
|
|
753
|
+
});
|
|
754
|
+
}
|
|
755
|
+
// ── public entry: cross-cutting protocol (auth / rate-limit) then route ──────────────────
|
|
756
|
+
export async function handleXaiTwinRequest(req) {
|
|
757
|
+
const method = req.method.toUpperCase();
|
|
758
|
+
const path = (req.path.split('?')[0] ?? '/').replace(/\/+$/, '') || '/';
|
|
759
|
+
// OAuth device authorization is a separate public protocol surface. Its token exchange is
|
|
760
|
+
// authenticated by one-use device/refresh credentials in the form body, not an API bearer.
|
|
761
|
+
// Keeping it in this handler means identity, refresh, CLI inference, and usage all project the
|
|
762
|
+
// same kernel state rather than behaving like disconnected fixtures.
|
|
763
|
+
if (path === '/oauth2/device/code' && method === 'POST') {
|
|
764
|
+
if (req.readOnly)
|
|
765
|
+
return { status: 405, body: { error: 'invalid_request', error_description: 'twin is read-only' } };
|
|
766
|
+
return startXaiTwinDeviceAuthorization({
|
|
767
|
+
body: parseForm(req.body),
|
|
768
|
+
occurredAt: req.occurredAt,
|
|
769
|
+
origin: req.origin ?? 'http://127.0.0.1',
|
|
770
|
+
root: req.root,
|
|
771
|
+
});
|
|
772
|
+
}
|
|
773
|
+
if (path === '/oauth2/token' && method === 'POST') {
|
|
774
|
+
if (req.readOnly)
|
|
775
|
+
return { status: 405, body: { error: 'invalid_request', error_description: 'twin is read-only' } };
|
|
776
|
+
return exchangeXaiTwinToken({ body: parseForm(req.body), occurredAt: req.occurredAt, root: req.root });
|
|
777
|
+
}
|
|
778
|
+
if (path === '/twin/oauth/authorizations' && method === 'POST') {
|
|
779
|
+
if (req.readOnly)
|
|
780
|
+
return { status: 405, body: { error: 'invalid_request', error_description: 'twin is read-only' } };
|
|
781
|
+
const control = parseJson(req.body);
|
|
782
|
+
if (typeof control.user_code !== 'string' || typeof control.approved !== 'boolean') {
|
|
783
|
+
return invalidArgument('user_code (string) and approved (boolean) are required');
|
|
784
|
+
}
|
|
785
|
+
return decideXaiTwinDeviceAuthorization({ approved: control.approved, occurredAt: req.occurredAt, root: req.root, userCode: control.user_code });
|
|
786
|
+
}
|
|
787
|
+
// Modeled authentication (401/403). Gated only when the request carries an auth surface.
|
|
788
|
+
if (req.headers !== undefined || req.apiKey !== undefined) {
|
|
789
|
+
const authErr = checkAuth(req, path);
|
|
790
|
+
if (authErr)
|
|
791
|
+
return authErr;
|
|
792
|
+
}
|
|
793
|
+
// Modeled rate limiting (429) — deterministic opt-in trigger header.
|
|
794
|
+
if (rateLimitTriggered(req))
|
|
795
|
+
return rateLimitError();
|
|
796
|
+
const seg = path.replace(/^\/+/, '').split('/'); // ["v1","chat","completions",...]
|
|
797
|
+
const params = parseJson(req.body);
|
|
798
|
+
// D3: a read-only twin rejects any mutation with a vendor-shaped error.
|
|
799
|
+
if (req.readOnly && method !== 'GET') {
|
|
800
|
+
return { status: 405, body: errBody('The operation is not allowed', 'twin is read-only; omit readOnly to accept writes') };
|
|
801
|
+
}
|
|
802
|
+
// ---- model catalogs ----
|
|
803
|
+
if (path === '/v1/models' && method === 'GET') {
|
|
804
|
+
return { status: 200, body: { object: 'list', data: [...XAI_MODELS, ...observedModels(req.root)] } };
|
|
805
|
+
}
|
|
806
|
+
if (seg[1] === 'models' && seg.length === 3 && method === 'GET') {
|
|
807
|
+
const mid = decodeURIComponent(seg[2]);
|
|
808
|
+
const m = findModel(mid) ?? observedModels(req.root).find((om) => om.id === mid);
|
|
809
|
+
return m ? { status: 200, body: m } : modelNotFound(mid);
|
|
810
|
+
}
|
|
811
|
+
if (path === '/v1/language-models' && method === 'GET') {
|
|
812
|
+
return { status: 200, body: { models: [...XAI_LANGUAGE_MODELS, ...observedLanguageModels(req.root)] } };
|
|
813
|
+
}
|
|
814
|
+
if (seg[1] === 'language-models' && seg.length === 3 && method === 'GET') {
|
|
815
|
+
const mid = decodeURIComponent(seg[2]);
|
|
816
|
+
const m = findLanguageModel(mid) ?? observedLanguageModels(req.root).find((om) => om.id === mid);
|
|
817
|
+
return m ? { status: 200, body: m } : modelNotFound(mid);
|
|
818
|
+
}
|
|
819
|
+
if (path === '/v1/image-generation-models' && method === 'GET') {
|
|
820
|
+
return { status: 200, body: { models: XAI_IMAGE_GENERATION_MODELS } };
|
|
821
|
+
}
|
|
822
|
+
if (seg[1] === 'image-generation-models' && seg.length === 3 && method === 'GET') {
|
|
823
|
+
const mid = decodeURIComponent(seg[2]);
|
|
824
|
+
const m = findImageGenerationModel(mid);
|
|
825
|
+
return m ? { status: 200, body: m } : modelNotFound(mid);
|
|
826
|
+
}
|
|
827
|
+
// ---- chat completions (the generative stub; envelope is faithful) ----
|
|
828
|
+
if (path === '/v1/chat/completions' && method === 'POST') {
|
|
829
|
+
// The official Grok CLI proxy routes by x-grok-model-override rather than
|
|
830
|
+
// trusting the JSON model field. API callers omit it and keep ordinary
|
|
831
|
+
// the ordinary inference API behavior; CLI callers get the exact deployment selection they
|
|
832
|
+
// requested, including the built-in grok-build alias.
|
|
833
|
+
const modelOverride = req.headers?.['x-grok-model-override'];
|
|
834
|
+
const chatParams = typeof modelOverride === 'string' && modelOverride.trim()
|
|
835
|
+
? { ...params, model: modelOverride.trim() }
|
|
836
|
+
: params;
|
|
837
|
+
const validated = validateChat(chatParams);
|
|
838
|
+
if ('error' in validated)
|
|
839
|
+
return validated.error;
|
|
840
|
+
const args = validated.args;
|
|
841
|
+
// deferred:true → { request_id } now, the completion on poll (stateful kernel row).
|
|
842
|
+
if (args.deferred)
|
|
843
|
+
return createDeferredCompletion(args, req);
|
|
844
|
+
if (args.stream && req.sseSink) {
|
|
845
|
+
const completion = streamChat(args, req.sseSink, req.occurredAt, req.scenarioEngine);
|
|
846
|
+
if (isCurrentXaiTwinAccessToken(presentedKey(req), req.root, req.occurredAt)) {
|
|
847
|
+
await recordXaiTwinUsage({ completionTokens: completion.usage.completion_tokens, occurredAt: req.occurredAt, promptTokens: completion.usage.prompt_tokens, requestId: completion.id, root: req.root });
|
|
848
|
+
}
|
|
849
|
+
return { status: 200, body: completion };
|
|
850
|
+
}
|
|
851
|
+
const completion = buildChatCompletion(args, req.occurredAt, req.scenarioEngine);
|
|
852
|
+
if (isCurrentXaiTwinAccessToken(presentedKey(req), req.root, req.occurredAt)) {
|
|
853
|
+
await recordXaiTwinUsage({ completionTokens: completion.usage.completion_tokens, occurredAt: req.occurredAt, promptTokens: completion.usage.prompt_tokens, requestId: completion.id, root: req.root });
|
|
854
|
+
}
|
|
855
|
+
return { status: 200, body: completion };
|
|
856
|
+
}
|
|
857
|
+
// deferred completion retrieval (202 pending → 200 once → 404 after).
|
|
858
|
+
if (seg[1] === 'chat' && seg[2] === 'deferred-completion' && seg.length === 4 && method === 'GET') {
|
|
859
|
+
return retrieveDeferredCompletion(decodeURIComponent(seg[3]), req);
|
|
860
|
+
}
|
|
861
|
+
// ---- legacy completions ----
|
|
862
|
+
if (path === '/v1/completions' && method === 'POST') {
|
|
863
|
+
return handleLegacyCompletions(params, req.occurredAt);
|
|
864
|
+
}
|
|
865
|
+
// ---- Anthropic-compatible messages ----
|
|
866
|
+
if (path === '/v1/messages' && method === 'POST') {
|
|
867
|
+
return handleMessagesCompat(params, req.occurredAt);
|
|
868
|
+
}
|
|
869
|
+
// ---- image generations ----
|
|
870
|
+
if (path === '/v1/images/generations' && method === 'POST') {
|
|
871
|
+
return handleImageGenerations(params, req.occurredAt);
|
|
872
|
+
}
|
|
873
|
+
// ---- api-key introspection ----
|
|
874
|
+
if (path === '/v1/api-key' && method === 'GET') {
|
|
875
|
+
return handleApiKey(req);
|
|
876
|
+
}
|
|
877
|
+
// ---- tokenizer ----
|
|
878
|
+
if (path === '/v1/tokenize-text' && method === 'POST') {
|
|
879
|
+
return handleTokenizeText(params);
|
|
880
|
+
}
|
|
881
|
+
// Unknown route → vendor-faithful 404 (never a fabricated success — D2).
|
|
882
|
+
return notFoundEntity(`Unknown request URL: ${method} ${path}. Please check the URL against https://docs.x.ai.`);
|
|
883
|
+
}
|