@volter/twin-fireworks 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (39) hide show
  1. package/LICENSE +202 -0
  2. package/README.md +184 -0
  3. package/dist/src/cli.d.ts +2 -0
  4. package/dist/src/cli.js +28 -0
  5. package/dist/src/fireworks-budget.d.ts +54 -0
  6. package/dist/src/fireworks-budget.js +146 -0
  7. package/dist/src/fireworks-capabilities.d.ts +4 -0
  8. package/dist/src/fireworks-capabilities.js +1205 -0
  9. package/dist/src/fireworks-conformance.d.ts +14 -0
  10. package/dist/src/fireworks-conformance.js +514 -0
  11. package/dist/src/fireworks-connector.d.ts +168 -0
  12. package/dist/src/fireworks-connector.js +641 -0
  13. package/dist/src/fireworks-models.d.ts +11 -0
  14. package/dist/src/fireworks-models.js +53 -0
  15. package/dist/src/fireworks-scenario.d.ts +55 -0
  16. package/dist/src/fireworks-scenario.js +171 -0
  17. package/dist/src/fireworks-server.d.ts +16 -0
  18. package/dist/src/fireworks-server.js +144 -0
  19. package/dist/src/fireworks-stub.d.ts +26 -0
  20. package/dist/src/fireworks-stub.js +78 -0
  21. package/dist/src/fireworks-twin.d.ts +51 -0
  22. package/dist/src/fireworks-twin.js +1426 -0
  23. package/dist/src/fireworks-types.d.ts +212 -0
  24. package/dist/src/fireworks-types.js +4 -0
  25. package/dist/src/index.d.ts +9 -0
  26. package/dist/src/index.js +105 -0
  27. package/package.json +52 -0
  28. package/src/cli.ts +27 -0
  29. package/src/fireworks-budget.ts +172 -0
  30. package/src/fireworks-capabilities.ts +1229 -0
  31. package/src/fireworks-conformance.ts +542 -0
  32. package/src/fireworks-connector.ts +700 -0
  33. package/src/fireworks-models.ts +63 -0
  34. package/src/fireworks-scenario.ts +191 -0
  35. package/src/fireworks-server.ts +153 -0
  36. package/src/fireworks-stub.ts +83 -0
  37. package/src/fireworks-twin.ts +1427 -0
  38. package/src/fireworks-types.ts +165 -0
  39. package/src/index.ts +134 -0
@@ -0,0 +1,1426 @@
1
+ // Fireworks twin REQUEST HANDLER — the canonical Fireworks API surface for the twin.
2
+ // Contract: handleFireworksTwinRequest({method, path, body}) -> {status, body}. It is the
3
+ // faithful Fireworks API the official clients (the `fireworks` Python SDK, `@ai-sdk/fireworks`,
4
+ // plain OpenAI clients pointed at api.fireworks.ai/inference/v1) talk to UNMODIFIED.
5
+ //
6
+ // Fireworks serves TWO planes on ONE host, and the twin serves both under the paths the vendor
7
+ // actually publishes:
8
+ // • INFERENCE — `/inference/v1/…` (OpenAI-compatible: chat/completions, completions, responses,
9
+ // embeddings, rerank) plus the Anthropic-compatible `/inference/v1/messages`. The twin cannot
10
+ // run a model, so generative output is a DETERMINISTIC STUB (fireworks-stub.ts), clearly
11
+ // labeled — never pretending to be real inference. The wire around it is faithful, INCLUDING
12
+ // Fireworks' documented OpenAI differences:
13
+ // - streaming usage arrives in the final chunk BY DEFAULT (OpenAI makes it opt-in);
14
+ // - `context_length_exceeded_behavior` defaults to `truncate`, not `error`;
15
+ // - `service_tier` accepts only 'priority' — every other value is treated as 'default'
16
+ // (documented; NOT an error);
17
+ // - validation failures answer the FastAPI `422 HTTPValidationError` envelope the vendor's
18
+ // own spec declares on these operations.
19
+ // • CONTROL — the Gateway REST API under `/v1/accounts/{account_id}/…`: deployments, datasets,
20
+ // batch-inference and supervised-fine-tuning jobs, users + their apiKeys, secrets, models.
21
+ // Stateful over the kernel action log; google.rpc-style errors (gatewayStatus {code,message});
22
+ // list envelopes `{<plural>, nextPageToken, totalSize}`; create ids passed as QUERY params
23
+ // (deployments/datasets/users) or inside the body (datasets), per the vendor's own spec.
24
+ //
25
+ // State lives in the kernel action log (D1): all writes are local actions, reads are the
26
+ // projection. No real Fireworks API is ever called from this path (D4). Streaming uses an
27
+ // INJECTED sink — no real sockets / setTimeout (D5 verify is offline + deterministic).
28
+ import { applyTwinWrite, projectResources } from '@volter/world-core';
29
+ import { FIREWORKS_MODEL_IDS } from "./fireworks-models.js";
30
+ import { contentToText, countPromptTokens, estimateTokens, fnv1a, pseudoEmbedding, stubAssistantText, stubReasoningText, stubRelevanceScore, } from "./fireworks-stub.js";
31
+ import { realizeFireworksRespond } from "./fireworks-scenario.js";
32
+ const SERVICE = 'fireworks';
33
+ /** The inference base path. The vendor's inference server root is `https://api.fireworks.ai/inference`
34
+ * and every OpenAI-compat operation hangs off `/v1/…` under it. */
35
+ export const FIREWORKS_INFERENCE_PREFIX = '/inference/v1';
36
+ /** The control-plane base path: everything Gateway REST hangs off `/v1/accounts/{account_id}/…`. */
37
+ export const FIREWORKS_ACCOUNTS_PREFIX = '/v1/accounts';
38
+ // ── vendor-shaped errors ──────────────────────────────────────────────────────────────
39
+ // TWO error envelopes, one per plane — mixing them would be a wire lie:
40
+ // • the inference plane's OpenAI-compat surface answers `{ error: { message, … } }` (the
41
+ // OpenAI shape the clients parse); the Anthropic-compat /v1/messages answers
42
+ // `{ type: 'error', error: { type, message }, request_id }` (the Anthropic shape);
43
+ // • the control plane answers google.rpc-style `{ code: <gatewayCode>, message }` — the shape
44
+ // the vendor's own spec embeds on every resource (`gatewayStatus`) and its gatewayCode enum.
45
+ function errBody(message, extra = {}) {
46
+ return { error: { message, ...extra } };
47
+ }
48
+ function invalidRequest(message, extra = {}) {
49
+ return { status: 400, body: errBody(message, { code: 400, ...extra }) };
50
+ }
51
+ function notFound(message) {
52
+ return { status: 404, body: errBody(message, { code: 404 }) };
53
+ }
54
+ function authError(message) {
55
+ return { status: 401, body: errBody(message, { code: 401 }) };
56
+ }
57
+ /** The FastAPI validation envelope the vendor's own spec declares on the inference operations
58
+ * (`422 HTTPValidationError` → `detail: [{loc, msg, type}]`). */
59
+ function validationError(loc, msg, type) {
60
+ return { status: 422, body: { detail: [{ loc, msg, type }] } };
61
+ }
62
+ /** The google.rpc-style control-plane error. `code` is a gatewayCode string, mapped from the
63
+ * HTTP status the way the vendor's own status embedding implies (NOT_FOUND → 'NOT_FOUND', …). */
64
+ function gatewayError(status, message) {
65
+ const code = status === 400 ? 'INVALID_ARGUMENT' :
66
+ status === 401 ? 'UNAUTHENTICATED' :
67
+ status === 403 ? 'PERMISSION_DENIED' :
68
+ status === 404 ? 'NOT_FOUND' :
69
+ status === 409 ? 'ALREADY_EXISTS' :
70
+ 'UNKNOWN';
71
+ return { status, body: { code, message } };
72
+ }
73
+ // ── modeled authentication (401) ────────────────────────────────────────────────────────
74
+ // Real Fireworks requires a bearer credential on every request (docs.fireworks.ai/api-reference/
75
+ // introduction: "All requests … must include an Authorization header with a valid Bearer token")
76
+ // and answers 401 Unauthorized when it is missing or invalid (the vendor's own error-code table).
77
+ // The twin can't validate against real keys, so it models the CHECKABLE failures: a missing
78
+ // credential, and a reserved sentinel for the invalid-key path. Any other non-empty key is
79
+ // accepted. Trusted in-process calls carry NEITHER `headers` nor `apiKey` and are NOT auth-gated;
80
+ // the official clients always send a key → they pass.
81
+ function checkAuth(req) {
82
+ const auth = req.headers?.['authorization'];
83
+ const bearer = typeof auth === 'string' && auth.toLowerCase().startsWith('bearer ') ? auth.slice(7).trim() : '';
84
+ const key = (req.apiKey ?? '').trim() || bearer;
85
+ if (!key || key === 'fw_invalid' || key === 'invalid') {
86
+ // Each plane answers 401 in its OWN envelope — a control-plane 401 carrying the OpenAI shape
87
+ // would mix the envelopes the header above calls a wire lie.
88
+ if (req.path.startsWith(`${FIREWORKS_INFERENCE_PREFIX}/messages`)) {
89
+ return anthropicError(401, 'authentication_error', 'Unauthorized: missing or invalid API key');
90
+ }
91
+ if (req.path.startsWith(`${FIREWORKS_ACCOUNTS_PREFIX}/`)) {
92
+ return gatewayError(401, 'Unauthorized: missing or invalid API key');
93
+ }
94
+ return authError('Unauthorized: missing or invalid API key');
95
+ }
96
+ return null;
97
+ }
98
+ function nowEpoch(occurredAt) {
99
+ return Math.floor((occurredAt ? Date.parse(occurredAt) : 0) / 1000);
100
+ }
101
+ // ── kernel helpers ──────────────────────────────────────────────────────────────────────
102
+ /**
103
+ * Rows of `type`, SCOPED TO ONE ACCOUNT. Every control-plane resource is stored with the
104
+ * `accountId` it was created (or pulled) under, and every read goes through this — a resource
105
+ * addressed under a different account's path is invisible here, which is what makes the
106
+ * cross-account GET/PATCH/DELETE answer the vendor's own NOT_FOUND. `accountId === undefined`
107
+ * means the caller reads a plane with NO account segment (the inference plane's stored
108
+ * responses) — those rows are not account-scoped and the filter is skipped for them.
109
+ */
110
+ /** A row is gone when the TWIN deleted it (`_deleted`, the twin's own marker) or when the
111
+ * CONNECTOR observed it vanish from the vendor (`deleted`, the kernel's tombstone field).
112
+ * Reading only the first served a resource the pull had correctly tombstoned (round-four review). */
113
+ export function isTombstoned(r) {
114
+ return r._deleted === true || r.deleted === true;
115
+ }
116
+ function rows(type, accountId, root) {
117
+ return projectResources(SERVICE, root).filter((r) => r.type === type && (accountId === undefined || r._account === accountId));
118
+ }
119
+ /**
120
+ * Mint the next local id for `type`. Derived from the ID SET ALREADY IN STATE (a scan of the
121
+ * projection), never a row count — a count-mint silently clobbers a pulled vendor id sitting in a
122
+ * gap above the count (ADDING_A_TWIN.md §5). The `_twin_` infix namespaces LOCAL mints, so a
123
+ * pulled Fireworks id can never be matched by this regex and therefore can never be re-minted;
124
+ * the scan includes TOMBSTONED rows, so the counter RATCHETS across delete→recreate. The scan is
125
+ * PER ACCOUNT (rows() filters by `_account`) and matches the id SUFFIX after the account
126
+ * namespace, so two tenants each mint their own `dep_twin_1` without colliding.
127
+ */
128
+ function nextId(type, prefix, accountId, root) {
129
+ let max = 0;
130
+ for (const r of rows(type, accountId, root)) {
131
+ const local = accountId !== undefined && typeof r.id === 'string' && r.id.startsWith(`${accountId}/`)
132
+ ? r.id.slice(accountId.length + 1)
133
+ : String(r.id);
134
+ const m = new RegExp(`^${prefix}_twin_(\\d+)$`).exec(local);
135
+ if (m)
136
+ max = Math.max(max, Number(m[1]));
137
+ }
138
+ return `${prefix}_twin_${max + 1}`;
139
+ }
140
+ /**
141
+ * Resolve one row by its BARE (wire) id inside one account. The kernel subject is the
142
+ * ACCOUNT-NAMESPACED id `${accountId}/${id}` — matching the vendor's own name grammar
143
+ * `accounts/{account}/{collection}/{id}` — so (type, subject) is unique per tenant and account
144
+ * B's create of the same id can never land on account A's row. Pulled rows carry the same
145
+ * namespaced id (the connector's mappers emit it), so created and pulled rows share one id space.
146
+ */
147
+ function getRow(type, id, accountId, root) {
148
+ const full = accountId === undefined ? id : `${accountId}/${id}`;
149
+ return rows(type, accountId, root).find((r) => r.id === full);
150
+ }
151
+ /** Strip the kernel's housekeeping fields and the twin's private underscore-prefixed fields. */
152
+ function strip(r) {
153
+ const out = {};
154
+ for (const [k, v] of Object.entries(r)) {
155
+ if (k === 'type' || k === 'updatedAt' || k === 'deleted' || k.startsWith('_'))
156
+ continue;
157
+ out[k] = v;
158
+ }
159
+ return out;
160
+ }
161
+ // ── request parsing ─────────────────────────────────────────────────────────────────────
162
+ function parseJson(body) {
163
+ if (!body || !body.trim())
164
+ return {};
165
+ try {
166
+ const v = JSON.parse(body);
167
+ return v && typeof v === 'object' ? v : {};
168
+ }
169
+ catch {
170
+ return {};
171
+ }
172
+ }
173
+ /** A deterministic id suffix from the request (so ids are stable + assertable). */
174
+ function stableSuffix(seedText) {
175
+ return fnv1a(seedText).toString(36);
176
+ }
177
+ // ── model catalog (the 404 the vendor answers for an unknown model id) ──────────────────
178
+ /**
179
+ * The model ids the INFERENCE plane accepts: the static catalog (fireworks-models.ts, sourced)
180
+ * PLUS every model row the control plane holds (a create/pull under /v1/accounts/.../models —
181
+ * an account-owned model is addressable for inference exactly as the vendor allows, including
182
+ * the fine-tunes). A pulled/created row of the same id as a static entry just re-states it.
183
+ * Real Fireworks answers 404 "Model id not found" for an id outside this set — the twin must
184
+ * refuse the same way, never answer 200 for any string.
185
+ */
186
+ function servedModelIds(root) {
187
+ const ids = new Set(FIREWORKS_MODEL_IDS);
188
+ for (const r of projectResources(SERVICE, root)) {
189
+ if (r.type !== 'model' || isTombstoned(r))
190
+ continue;
191
+ // A control-plane-created model is addressed ONLY by its FULL resource name on the
192
+ // inference plane (`accounts/{account}/models/{id}`) — the `name` the create handler
193
+ // minted. The bare suffix is NOT an alias: the vendor's own inference API takes full
194
+ // resource names for account-owned models (the static catalog's short aliases are
195
+ // documented serverless serving paths, a different thing), and accepting the bare suffix
196
+ // leaked one account's model into EVERY account's inference plane — a model created only
197
+ // under SECRETACCT answered 200 as the bare id while its qualified name 404'd under
198
+ // another account (inverted tenancy). The kernel subject stays the account-namespaced id
199
+ // (`{account}/{id}`); only the wire `name` joins the catalog.
200
+ if (typeof r.name === 'string')
201
+ ids.add(r.name);
202
+ }
203
+ return ids;
204
+ }
205
+ /** The vendor's own refusal for an unknown model id (its error table: 404 covers "the model
206
+ * doesn't exist, the model is not deployed, or you don't have permission to access it"). */
207
+ function unknownModel(model) {
208
+ return notFound(`Model id not found: ${model}`);
209
+ }
210
+ // ── chat completions: validate the request the way Fireworks does ───────────────────────
211
+ const SERVICE_TIERS = ['auto', 'default', 'flex', 'priority'];
212
+ const CONTEXT_BEHAVIORS = ['error', 'truncate'];
213
+ /**
214
+ * The inference sampling parameters' documented ranges, from the pinned spec's own field
215
+ * descriptions (temperature "0 to 2", n "between 1 and 128", top_k "between 0 and 100",
216
+ * top_p "Required range: `0 <= x <= 1`" — CompletionRequest's own wording; ChatCompletionRequest
217
+ * carries the same nucleus-sampling semantics, frequency/presence_penalty "between -2 and 2")
218
+ * plus the OpenAI-compat types (role enum, stream boolean). The vendor's server (FastAPI)
219
+ * answers each violation with the 422 HTTPValidationError envelope and the pydantic error
220
+ * `type` — the twin answers the same, per parameter and constraint. `loc` matches where the
221
+ * field rides in the request body.
222
+ */
223
+ const SAMPLING_RULES = [
224
+ { field: 'temperature', kind: 'number', min: 0, max: 2 },
225
+ { field: 'top_p', kind: 'number', min: 0, max: 1 },
226
+ { field: 'n', kind: 'number', min: 1, max: 128, integer: true },
227
+ { field: 'top_k', kind: 'number', min: 0, max: 100, integer: true },
228
+ { field: 'frequency_penalty', kind: 'number', min: -2, max: 2 },
229
+ { field: 'presence_penalty', kind: 'number', min: -2, max: 2 },
230
+ ];
231
+ /** Validate the sampling/typing rules; returns the FastAPI 422 envelope on violation. Shared by
232
+ * the chat and legacy-completions doors so the two surfaces cannot drift. The message ROLE enum
233
+ * is checked per message by the chat door's own loop (the role rides inside `messages`). */
234
+ function validateSampling(params) {
235
+ for (const rule of SAMPLING_RULES) {
236
+ const v = params[rule.field];
237
+ if (v === undefined || v === null)
238
+ continue;
239
+ if (rule.kind === 'number') {
240
+ // pydantic v2's kinds: a JSON string/bool where a number belongs is int_parsing (for an
241
+ // integer field) or float_parsing (for a float field) — int_type/float_type are
242
+ // python-typed-argument errors that never fire on a JSON body; a float where an integer
243
+ // belongs is int_from_float.
244
+ if (typeof v !== 'number' || Number.isNaN(v)) {
245
+ return validationError(['body', rule.field], `Input should be a valid ${rule.integer ? 'integer' : 'number'}`, rule.integer ? 'int_parsing' : 'float_parsing');
246
+ }
247
+ if (rule.integer && !Number.isInteger(v))
248
+ return validationError(['body', rule.field], 'Input should be a valid integer', 'int_from_float');
249
+ if (rule.min !== undefined && v < rule.min)
250
+ return validationError(['body', rule.field], `Input should be greater than or equal to ${rule.min}`, 'greater_than_equal');
251
+ if (rule.max !== undefined && v > rule.max)
252
+ return validationError(['body', rule.field], `Input should be less than or equal to ${rule.max}`, 'less_than_equal');
253
+ }
254
+ }
255
+ if (params.stream !== undefined && params.stream !== null && typeof params.stream !== 'boolean') {
256
+ return validationError(['body', 'stream'], 'Input should be a valid boolean', 'bool_type');
257
+ }
258
+ return null;
259
+ }
260
+ /** The OpenAI-compat message role enum (the spec's ChatCompletionRequestMessage.role). */
261
+ const MESSAGE_ROLES = ['system', 'user', 'assistant', 'tool', 'function', 'developer'];
262
+ /**
263
+ * Fireworks' documented OpenAI differences are the fidelity surface here:
264
+ * • `max_tokens`/`max_completion_tokens` are MUTUALLY EXCLUSIVE (the spec's own note: "Alias for
265
+ * max_tokens. Cannot be specified together with max_tokens.") — OpenAI allows both;
266
+ * • `service_tier` is a closed enum whose values are all ACCEPTED but only 'priority' is
267
+ * honored — the others are treated as 'default', never an error;
268
+ * • `context_length_exceeded_behavior` defaults to 'truncate' (OpenAI's is 'error').
269
+ * The twin has no real context window, so the truncate path is modeled as the documented
270
+ * SEMANTIC (max_tokens lowered to fit) whenever a stub context is exceeded.
271
+ */
272
+ function validateChat(params) {
273
+ if (params.model === undefined)
274
+ return { error: validationError(['body', 'model'], 'Field required', 'missing') };
275
+ if (typeof params.model !== 'string' || !params.model)
276
+ return { error: validationError(['body', 'model'], 'Input should be a valid string', 'string_type') };
277
+ if (!Array.isArray(params.messages))
278
+ return { error: validationError(['body', 'messages'], 'Field required', 'missing') };
279
+ if (params.messages.length === 0)
280
+ return { error: invalidRequest("'messages' must not be empty") };
281
+ const messages = params.messages;
282
+ for (const [i, m] of messages.entries()) {
283
+ if (!m || typeof m !== 'object' || typeof m.role !== 'string') {
284
+ return { error: validationError(['body', 'messages'], 'Input should be a valid dictionary', 'model_attributes_type') };
285
+ }
286
+ if (!MESSAGE_ROLES.includes(m.role)) {
287
+ // FastAPI's loc carries the message INDEX (`body.messages.<i>.role`), not just the field.
288
+ return { error: validationError(['body', 'messages', i, 'role'], `Input should be one of ${MESSAGE_ROLES.map((r) => `'${r}'`).join(', ')}`, 'enum') };
289
+ }
290
+ }
291
+ // The sampling/typing table (temperature/n/top_k/penalties/stream) — the same rules the legacy
292
+ // door applies, so the two inference surfaces cannot drift.
293
+ const sampling = validateSampling(params);
294
+ if (sampling)
295
+ return { error: sampling };
296
+ if (params.max_tokens !== undefined && params.max_tokens !== null && params.max_completion_tokens !== undefined && params.max_completion_tokens !== null) {
297
+ return { error: invalidRequest("'max_tokens' and 'max_completion_tokens' cannot both be specified — 'max_completion_tokens' is an alias for 'max_tokens'") };
298
+ }
299
+ const maxRaw = params.max_completion_tokens ?? params.max_tokens;
300
+ let maxTokens;
301
+ if (maxRaw !== undefined && maxRaw !== null) {
302
+ // pydantic parses the JSON value BEFORE any range check: a string is int_parsing, a float
303
+ // int_from_float, and only a parsed integer then hits the ≥1 bound.
304
+ maxTokens = Number(maxRaw);
305
+ if (typeof maxRaw !== 'number' || Number.isNaN(maxRaw))
306
+ return { error: validationError(['body', 'max_tokens'], 'Input should be a valid integer', 'int_parsing') };
307
+ if (!Number.isInteger(maxTokens))
308
+ return { error: validationError(['body', 'max_tokens'], 'Input should be a valid integer', 'int_from_float') };
309
+ if (maxTokens < 1)
310
+ return { error: validationError(['body', 'max_tokens'], 'Input should be a valid integer greater than or equal to 1', 'greater_than_equal') };
311
+ }
312
+ let stop;
313
+ if (params.stop !== undefined && params.stop !== null) {
314
+ if (typeof params.stop === 'string')
315
+ stop = [params.stop];
316
+ else if (Array.isArray(params.stop))
317
+ stop = params.stop;
318
+ else
319
+ return { error: validationError(['body', 'stop'], 'Input should be a valid string or array of strings', 'string_type') };
320
+ }
321
+ let serviceTier = 'default';
322
+ if (params.service_tier !== undefined && params.service_tier !== null) {
323
+ if (typeof params.service_tier !== 'string' || !SERVICE_TIERS.includes(params.service_tier)) {
324
+ return { error: validationError(['body', 'service_tier'], `Input should be one of ${SERVICE_TIERS.map((t) => `'${t}'`).join(', ')}`, 'enum') };
325
+ }
326
+ // The vendor's own semantics: "Only 'priority' is supported, while all other values will be
327
+ // treated as 'default' tier." Every enum value is ACCEPTED; 'auto'/'flex' do NOT error.
328
+ serviceTier = params.service_tier === 'priority' ? 'priority' : 'default';
329
+ }
330
+ let contextBehavior = 'truncate';
331
+ if (params.context_length_exceeded_behavior !== undefined && params.context_length_exceeded_behavior !== null) {
332
+ if (typeof params.context_length_exceeded_behavior !== 'string' || !CONTEXT_BEHAVIORS.includes(params.context_length_exceeded_behavior)) {
333
+ return { error: validationError(['body', 'context_length_exceeded_behavior'], "Input should be one of 'error', 'truncate'", 'enum') };
334
+ }
335
+ contextBehavior = params.context_length_exceeded_behavior;
336
+ }
337
+ let responseFormat;
338
+ const rf = params.response_format;
339
+ if (rf && typeof rf === 'object') {
340
+ if (rf.type === 'json_object')
341
+ responseFormat = { type: 'json_object' };
342
+ else if (rf.type === 'json_schema')
343
+ responseFormat = { type: 'json_schema', schema: rf.json_schema };
344
+ else if (rf.type !== undefined && rf.type !== 'text')
345
+ return { error: invalidRequest("'response_format.type' must be one of 'text', 'json_object', 'json_schema'") };
346
+ }
347
+ const stream = params.stream === true;
348
+ // Fireworks streams usage BY DEFAULT (the documented OpenAI difference); `include_usage: false`
349
+ // is the opt-OUT — the inverse of OpenAI's opt-IN.
350
+ const streamOptions = params.stream_options;
351
+ const includeUsage = streamOptions?.include_usage !== false;
352
+ return {
353
+ args: {
354
+ model: params.model,
355
+ messages,
356
+ promptText: messages.map((m) => contentToText(m.content)).join('\n'),
357
+ ...(params.tools !== undefined ? { tools: params.tools } : {}),
358
+ ...(maxTokens !== undefined ? { maxTokens } : {}),
359
+ ...(stop !== undefined ? { stop } : {}),
360
+ stream,
361
+ includeUsage,
362
+ serviceTier,
363
+ contextBehavior,
364
+ ...(responseFormat !== undefined ? { responseFormat } : {}),
365
+ ...(params.reasoning_effort !== undefined && params.reasoning_effort !== null ? { reasoningEffort: params.reasoning_effort } : {}),
366
+ },
367
+ };
368
+ }
369
+ /** Build ONE deterministic stub chat completion. */
370
+ function buildChatCompletion(args, occurredAt, decision) {
371
+ const promptTokens = countPromptTokens(args.messages);
372
+ // Scenario scripting first — the DECISION was made (and any fault honored) by the request
373
+ // handler through the engine's serve(); only a content decision reaches here. A matching
374
+ // handler scripts the answer; a miss answers the labeled stub with a pointer naming the miss
375
+ // and the features seen.
376
+ let scripted = null;
377
+ let missTeach = '';
378
+ if (decision) {
379
+ if (decision.kind === 'handler') {
380
+ scripted = realizeFireworksRespond(decision.respond);
381
+ }
382
+ else {
383
+ missTeach = `\n[twin-scenario miss — no handler matched. Author one in the world dir's handlers/fireworks.json (GET /twin explains; GET /twin/scenario lists handlers + misses). Features seen: ${JSON.stringify(decision.miss.features)}]`;
384
+ }
385
+ }
386
+ if (scripted) {
387
+ const message = { role: 'assistant', content: scripted.text ?? '', ...(scripted.toolCalls.length ? { tool_calls: scripted.toolCalls } : {}) };
388
+ if (scripted.reasoning !== null)
389
+ message.reasoning_content = scripted.reasoning;
390
+ const completionTokens = estimateTokens(JSON.stringify(scripted.toolCalls.length ? scripted.toolCalls : scripted.text ?? ''));
391
+ return {
392
+ id: `chatcmpl-twin-${stableSuffix(JSON.stringify(args.messages) + args.model)}`,
393
+ object: 'chat.completion',
394
+ created: nowEpoch(occurredAt),
395
+ model: args.model,
396
+ choices: [{ index: 0, message, finish_reason: scripted.finishReason, logprobs: null }],
397
+ usage: { prompt_tokens: promptTokens, total_tokens: promptTokens + completionTokens, completion_tokens: completionTokens },
398
+ };
399
+ }
400
+ let text = args.responseFormat?.type === 'json_object' || args.responseFormat?.type === 'json_schema'
401
+ ? `${stubAssistantText(args.model, args.promptText)}\n{"twin_stub":true}`
402
+ : stubAssistantText(args.model, args.promptText);
403
+ if (missTeach)
404
+ text += missTeach;
405
+ let finish = 'stop';
406
+ // Truncate at the EARLIEST-occurring stop sequence across the whole `stop` list.
407
+ let stopAt = -1;
408
+ for (const s of args.stop ?? []) {
409
+ if (!s)
410
+ continue;
411
+ const i = text.indexOf(s);
412
+ if (i >= 0 && (stopAt < 0 || i < stopAt))
413
+ stopAt = i;
414
+ }
415
+ if (stopAt >= 0) {
416
+ text = text.slice(0, stopAt);
417
+ finish = 'stop';
418
+ }
419
+ // THE VENDOR DIFFERENCE: exceeding the (stub) context truncates max_tokens by default instead
420
+ // of erroring. The stub context is generous and fixed; the behavior is the fidelity surface.
421
+ const stubContextTokens = 8192;
422
+ if (promptTokens + (args.maxTokens ?? 0) > stubContextTokens) {
423
+ if (args.contextBehavior === 'error') {
424
+ return invalidRequest(`This model's maximum context length is ${stubContextTokens} tokens. However, you requested ${(args.maxTokens ?? 0) + promptTokens} tokens in the messages, which exceeds the model's context limit.`);
425
+ }
426
+ // truncate: max_tokens is lowered to fit.
427
+ if (args.maxTokens !== undefined)
428
+ args.maxTokens = Math.max(1, stubContextTokens - promptTokens);
429
+ }
430
+ if (args.maxTokens !== undefined && estimateTokens(text) > args.maxTokens) {
431
+ text = text.slice(0, args.maxTokens * 4);
432
+ finish = 'length';
433
+ }
434
+ const message = { role: 'assistant', content: text };
435
+ // reasoning_effort (any truthy non-'none'/'false' value) surfaces Fireworks' separate
436
+ // `reasoning_content` field — the field the vendor's reasoning models answer with.
437
+ const effortOn = args.reasoningEffort !== undefined && args.reasoningEffort !== 'none' && args.reasoningEffort !== false;
438
+ if (effortOn)
439
+ message.reasoning_content = stubReasoningText(args.model, args.promptText);
440
+ const completionTokens = estimateTokens(text) + estimateTokens(message.reasoning_content ?? '');
441
+ const usage = { prompt_tokens: promptTokens, total_tokens: promptTokens + completionTokens, completion_tokens: completionTokens };
442
+ const id = `chatcmpl-twin-${stableSuffix(JSON.stringify(args.messages) + args.model)}`;
443
+ return {
444
+ id,
445
+ object: 'chat.completion',
446
+ created: nowEpoch(occurredAt),
447
+ model: args.model,
448
+ choices: [{ index: 0, message, finish_reason: finish, logprobs: null }],
449
+ // usage is carried unconditionally (the vendor's streaming behavior implies it is always
450
+ // computed; the schema marks it nullable only for the echo/logprobs edge paths).
451
+ usage,
452
+ };
453
+ }
454
+ /** Split text into deterministic streaming chunks (≤ ~20 chars each), preserving order. */
455
+ function chunkText(text) {
456
+ if (!text)
457
+ return [];
458
+ const out = [];
459
+ for (let i = 0; i < text.length; i += 20)
460
+ out.push(text.slice(i, i + 20));
461
+ return out;
462
+ }
463
+ /**
464
+ * Emit the vendor-faithful Fireworks streaming sequence into the injected sink (NO sockets, NO
465
+ * setTimeout). THE VENDOR DIFFERENCE (docs.fireworks.ai/tools-sdks/openai-compatibility): "For
466
+ * streaming responses, the `usage` field is returned in the very last chunk on the response (i.e.
467
+ * the one having `finish_reason` set)" — BY DEFAULT, no `stream_options.include_usage` needed
468
+ * (OpenAI makes it opt-in; Fireworks makes it opt-OUT via `include_usage: false`).
469
+ */
470
+ function streamChat(args, sink, occurredAt, decision) {
471
+ const built = buildChatCompletion(args, occurredAt, decision);
472
+ if ('status' in built)
473
+ return built;
474
+ const full = built;
475
+ const base = { id: full.id, object: 'chat.completion.chunk', created: full.created, model: full.model };
476
+ sink({ data: { ...base, choices: [{ index: 0, delta: { role: 'assistant', content: '' }, finish_reason: null }], usage: null } });
477
+ if (full.choices[0].message.reasoning_content) {
478
+ sink({ data: { ...base, choices: [{ index: 0, delta: { reasoning_content: full.choices[0].message.reasoning_content }, finish_reason: null }], usage: null } });
479
+ }
480
+ for (const piece of chunkText(full.choices[0].message.content ?? '')) {
481
+ sink({ data: { ...base, choices: [{ index: 0, delta: { content: piece }, finish_reason: null }], usage: null } });
482
+ }
483
+ // The final chunk carries finish_reason AND — by default — the usage.
484
+ sink({
485
+ data: {
486
+ ...base,
487
+ choices: [{ index: 0, delta: {}, finish_reason: full.choices[0].finish_reason }],
488
+ ...(args.includeUsage ? { usage: full.usage } : { usage: null }),
489
+ },
490
+ });
491
+ sink({ done: true });
492
+ return full;
493
+ }
494
+ // ── legacy completions (/v1/completions) ────────────────────────────────────────────────
495
+ /** Stream a legacy completion the OpenAI-compatible way: chunks carry the delta in
496
+ * `choices[].text` (`object: 'text_completion'` throughout — the legacy stream has no separate
497
+ * chunk object), ending with the finish chunk that — the Fireworks default — carries `usage`. */
498
+ function streamCompletion(full, sink, includeUsage) {
499
+ const base = { id: full.id, object: 'text_completion', created: full.created, model: full.model };
500
+ for (const choice of full.choices) {
501
+ sink({ data: { ...base, choices: [{ index: choice.index, text: choice.text, finish_reason: null, logprobs: choice.logprobs }], usage: null } });
502
+ }
503
+ sink({
504
+ data: {
505
+ ...base,
506
+ choices: full.choices.map((c) => ({ index: c.index, text: '', finish_reason: c.finish_reason, logprobs: c.logprobs })),
507
+ ...(includeUsage ? { usage: full.usage } : { usage: null }),
508
+ },
509
+ });
510
+ sink({ done: true });
511
+ return full;
512
+ }
513
+ function handleCompletion(params, occurredAt, sink, root) {
514
+ if (params.model === undefined)
515
+ return { status: 422, body: { detail: [{ loc: ['body', 'model'], msg: 'Field required', type: 'missing' }] } };
516
+ if (params.prompt === undefined)
517
+ return { status: 422, body: { detail: [{ loc: ['body', 'prompt'], msg: 'Field required', type: 'missing' }] } };
518
+ if (typeof params.model !== 'string' || !params.model)
519
+ return { status: 422, body: { detail: [{ loc: ['body', 'model'], msg: 'Input should be a valid string', type: 'string_type' }] } };
520
+ if (!servedModelIds(root).has(params.model))
521
+ return unknownModel(params.model);
522
+ // The same sampling/typing table the chat door applies.
523
+ const sampling = validateSampling(params);
524
+ if (sampling)
525
+ return sampling;
526
+ const prompts = Array.isArray(params.prompt) ? params.prompt.map(String) : [String(params.prompt)];
527
+ const maxRaw = params.max_completion_tokens ?? params.max_tokens;
528
+ if (params.max_tokens !== undefined && params.max_tokens !== null && params.max_completion_tokens !== undefined && params.max_completion_tokens !== null) {
529
+ return invalidRequest("'max_tokens' and 'max_completion_tokens' cannot both be specified — 'max_completion_tokens' is an alias for 'max_tokens'");
530
+ }
531
+ let maxTokens;
532
+ if (maxRaw !== undefined && maxRaw !== null) {
533
+ // The same pydantic parse order the chat door applies (string → int_parsing, float →
534
+ // int_from_float, then the ≥1 bound).
535
+ maxTokens = Number(maxRaw);
536
+ if (typeof maxRaw !== 'number' || Number.isNaN(maxRaw))
537
+ return { status: 422, body: { detail: [{ loc: ['body', 'max_tokens'], msg: 'Input should be a valid integer', type: 'int_parsing' }] } };
538
+ if (!Number.isInteger(maxTokens))
539
+ return { status: 422, body: { detail: [{ loc: ['body', 'max_tokens'], msg: 'Input should be a valid integer', type: 'int_from_float' }] } };
540
+ if (maxTokens < 1)
541
+ return { status: 422, body: { detail: [{ loc: ['body', 'max_tokens'], msg: 'Input should be a valid integer greater than or equal to 1', type: 'greater_than_equal' }] } };
542
+ }
543
+ const model = String(params.model);
544
+ const promptTokens = prompts.reduce((sum, p) => sum + estimateTokens(p), 0);
545
+ const choices = prompts.map((p, i) => {
546
+ let text = stubAssistantText(model, p);
547
+ let finish = 'stop';
548
+ if (maxTokens !== undefined && estimateTokens(text) > maxTokens) {
549
+ text = text.slice(0, maxTokens * 4);
550
+ finish = 'length';
551
+ }
552
+ return { index: i, text, finish_reason: finish, logprobs: null };
553
+ });
554
+ const completionTokens = choices.reduce((sum, c) => sum + estimateTokens(c.text), 0);
555
+ const body = {
556
+ id: `cmpl-twin-${stableSuffix(JSON.stringify(prompts) + model)}`,
557
+ object: 'text_completion',
558
+ created: nowEpoch(occurredAt),
559
+ model,
560
+ choices,
561
+ // The spec marks `usage` REQUIRED on Completion (nullable only on chat).
562
+ usage: { prompt_tokens: promptTokens, total_tokens: promptTokens + completionTokens, completion_tokens: completionTokens },
563
+ };
564
+ // The server hands a sink only for `stream:true`; ignoring it would answer a 200 event stream
565
+ // with zero frames — a fake success on the wire.
566
+ const streamOptions = params.stream_options;
567
+ if (sink)
568
+ return { status: 200, body: streamCompletion(body, sink, streamOptions?.include_usage !== false) };
569
+ return { status: 200, body };
570
+ }
571
+ // ── Responses API (/v1/responses — stateful CRUD over the kernel log) ───────────────────
572
+ /**
573
+ * Fireworks' Responses API stores responses server-side (`store` defaults true; `store:false`
574
+ * answers with a NULL id per the vendor's own schema: "Will be None if store=False"). The twin
575
+ * mirrors that: stored responses live in the kernel action log and are retrievable/deletable;
576
+ * a store=false response is answered id-less and NOT stored.
577
+ */
578
+ function responseView(r) {
579
+ return { id: r.id, ...strip(r) };
580
+ }
581
+ async function createResponse(params, req) {
582
+ if (params.model === undefined)
583
+ return { status: 422, body: { detail: [{ loc: ['body', 'model'], msg: 'Field required', type: 'missing' }] } };
584
+ if (params.input === undefined)
585
+ return { status: 422, body: { detail: [{ loc: ['body', 'input'], msg: 'Field required', type: 'missing' }] } };
586
+ if (typeof params.model !== 'string' || !params.model)
587
+ return { status: 422, body: { detail: [{ loc: ['body', 'model'], msg: 'Input should be a valid string', type: 'string_type' }] } };
588
+ const store = params.store !== false;
589
+ const inputText = typeof params.input === 'string' ? params.input : JSON.stringify(params.input);
590
+ const promptTokens = estimateTokens(inputText);
591
+ const text = stubAssistantText(String(params.model), inputText);
592
+ const completionTokens = estimateTokens(text);
593
+ const outputItem = {
594
+ type: 'message',
595
+ id: `msg_twin_${stableSuffix(text)}`,
596
+ role: 'assistant',
597
+ status: 'completed',
598
+ content: [{ type: 'output_text', text }],
599
+ };
600
+ const base = {
601
+ object: 'response',
602
+ created_at: nowEpoch(req.occurredAt),
603
+ status: 'completed',
604
+ model: params.model,
605
+ output: [outputItem],
606
+ usage: { prompt_tokens: promptTokens, total_tokens: promptTokens + completionTokens, completion_tokens: completionTokens },
607
+ ...(params.instructions !== undefined ? { instructions: params.instructions } : {}),
608
+ ...(params.metadata !== undefined ? { metadata: params.metadata } : {}),
609
+ ...(params.previous_response_id !== undefined ? { previous_response_id: params.previous_response_id } : {}),
610
+ ...(params.temperature !== undefined ? { temperature: params.temperature } : {}),
611
+ ...(params.max_output_tokens !== undefined ? { max_output_tokens: params.max_output_tokens } : {}),
612
+ store,
613
+ };
614
+ if (!store) {
615
+ // The vendor's own contract: id is null when store=false, and nothing is retrievable later.
616
+ return { status: 200, body: { ...base, id: null } };
617
+ }
618
+ const id = `resp_twin_${stableSuffix(inputText + String(params.model))}`;
619
+ await applyTwinWrite(SERVICE, {
620
+ operation: 'response.create',
621
+ subjectType: 'response',
622
+ subjectId: id,
623
+ fields: base,
624
+ ...(req.occurredAt ? { occurredAt: req.occurredAt } : {}),
625
+ actor: { kind: 'agent' },
626
+ }, req.root);
627
+ return { status: 200, body: responseView(getRow('response', id, undefined, req.root) ?? { ...base, id }) };
628
+ }
629
+ function handleListResponses(req) {
630
+ const url = new URL(req.path, 'http://twin');
631
+ const limit = Number(url.searchParams.get('limit') ?? 20);
632
+ const data = rows('response', undefined, req.root).filter((r) => !isTombstoned(r)).map(responseView);
633
+ const page = data.slice(0, Number.isFinite(limit) && limit > 0 ? limit : 20);
634
+ return { status: 200, body: { object: 'list', data: page, has_more: data.length > page.length, first_id: page[0]?.id ?? null, last_id: page[page.length - 1]?.id ?? null } };
635
+ }
636
+ function handleGetResponse(id, req) {
637
+ const r = getRow('response', id, undefined, req.root);
638
+ if (!r || isTombstoned(r))
639
+ return notFound(`No response found with id '${id}'.`);
640
+ return { status: 200, body: responseView(r) };
641
+ }
642
+ async function handleDeleteResponse(id, req) {
643
+ const r = getRow('response', id, undefined, req.root);
644
+ if (!r || isTombstoned(r))
645
+ return notFound(`No response found with id '${id}'.`);
646
+ await applyTwinWrite(SERVICE, {
647
+ operation: 'response.delete', subjectType: 'response', subjectId: id, fields: { _deleted: true },
648
+ ...(req.occurredAt ? { occurredAt: req.occurredAt } : {}), actor: { kind: 'agent' },
649
+ }, req.root);
650
+ return { status: 200, body: { id, object: 'response', deleted: true } };
651
+ }
652
+ // ── Anthropic-compatible /v1/messages ───────────────────────────────────────────────────
653
+ /** The Anthropic error envelope — a DIFFERENT shape from the OpenAI-compat plane's. */
654
+ function anthropicError(status, type, message) {
655
+ return { status, body: { type: 'error', error: { type, message }, request_id: null } };
656
+ }
657
+ function validateAnthropicMessages(params) {
658
+ if (params.model === undefined)
659
+ return { error: anthropicError(400, 'invalid_request_error', 'model: Field required') };
660
+ if (!Array.isArray(params.messages) || params.messages.length === 0) {
661
+ return { error: anthropicError(400, 'invalid_request_error', 'messages: Field required') };
662
+ }
663
+ for (const m of params.messages) {
664
+ if (!m || typeof m !== 'object' || (m.role !== 'user' && m.role !== 'assistant')) {
665
+ return { error: anthropicError(400, 'invalid_request_error', "messages: each message must have role 'user' or 'assistant'") };
666
+ }
667
+ }
668
+ return { ok: true };
669
+ }
670
+ function buildAnthropicMessage(params, occurredAt) {
671
+ const bad = validateAnthropicMessages(params);
672
+ if ('error' in bad)
673
+ return bad.error;
674
+ const messages = params.messages;
675
+ const promptText = messages.map((m) => contentToText(m.content)).join('\n');
676
+ const systemText = typeof params.system === 'string' ? params.system : Array.isArray(params.system) ? params.system.map((b) => contentToText(b.text)).join('\n') : '';
677
+ const inputTokens = estimateTokens(promptText + systemText);
678
+ // max_tokens is OPTIONAL on Fireworks (required on Anthropic — the documented difference).
679
+ const maxTokens = typeof params.max_tokens === 'number' ? params.max_tokens : undefined;
680
+ let text = stubAssistantText(String(params.model), promptText.trim() || systemText);
681
+ let stopReason = 'end_turn';
682
+ if (maxTokens !== undefined && estimateTokens(text) > maxTokens) {
683
+ text = text.slice(0, maxTokens * 4);
684
+ stopReason = 'max_tokens';
685
+ }
686
+ const content = [{ type: 'text', text, citations: null }];
687
+ const outputTokens = estimateTokens(text);
688
+ const body = {
689
+ id: `msg_twin_${stableSuffix(promptText + String(params.model))}`,
690
+ type: 'message',
691
+ role: 'assistant',
692
+ content,
693
+ model: String(params.model),
694
+ stop_reason: stopReason,
695
+ stop_sequence: null,
696
+ // Usage is included in BOTH streaming and non-streaming responses (the documented difference
697
+ // from Anthropic, where streaming omits it until the final delta).
698
+ usage: { input_tokens: inputTokens, output_tokens: outputTokens },
699
+ };
700
+ return { status: 200, body };
701
+ }
702
+ /** The Anthropic SSE sequence: message_start → content_block_start → content_block_delta* →
703
+ * content_block_stop → message_delta (carrying the ACTUAL usage — the one message_delta per
704
+ * stream) → message_stop. */
705
+ function streamAnthropicMessage(params, sink, occurredAt) {
706
+ const built = buildAnthropicMessage(params, occurredAt);
707
+ if (built.status !== 200)
708
+ return built;
709
+ const full = built.body;
710
+ const content = full.content;
711
+ const usage = full.usage ?? { input_tokens: 0, output_tokens: 0 };
712
+ sink({ data: { type: 'message_start', message: { ...full, content: [], stop_reason: null, usage: { input_tokens: usage.input_tokens, output_tokens: 0 } } } });
713
+ sink({ data: { type: 'content_block_start', index: 0, content_block: { type: 'text', text: '' } } });
714
+ for (const piece of chunkText(content[0].type === 'text' ? content[0].text : '')) {
715
+ sink({ data: { type: 'content_block_delta', index: 0, delta: { type: 'text_delta', text: piece } } });
716
+ }
717
+ sink({ data: { type: 'content_block_stop', index: 0 } });
718
+ // ONE message_delta, carrying the ACTUAL token counts (the vendor's own note: the message_start
719
+ // usage is always 0 and should be ignored for metering).
720
+ sink({ data: { type: 'message_delta', delta: { stop_reason: full.stop_reason, stop_sequence: null }, usage: { input_tokens: usage.input_tokens, output_tokens: usage.output_tokens } } });
721
+ sink({ data: { type: 'message_stop' } });
722
+ sink({ done: true });
723
+ return built;
724
+ }
725
+ // ── embeddings + rerank (deterministic pseudo-vectors / scores) ─────────────────────────
726
+ /** The largest vector the twin will ever allocate from a client number. The vendor's resizable
727
+ * embedding models truncate DOWN to the requested `dimensions` (never produce longer vectors
728
+ * than the model's native output); the catalog's largest native dimensionality across the
729
+ * served embedders is the Qwen3 embedding family's 4096. Every array sized from a REQUEST
730
+ * value in this pack is bounded by this constant or by an input length — audited. */
731
+ const MAX_EMBEDDING_DIMENSIONS = 4096;
732
+ function handleEmbeddings(params, root) {
733
+ if (params.model === undefined)
734
+ return { status: 422, body: { detail: [{ loc: ['body', 'model'], msg: 'Field required', type: 'missing' }] } };
735
+ if (params.input === undefined)
736
+ return { status: 422, body: { detail: [{ loc: ['body', 'input'], msg: 'Field required', type: 'missing' }] } };
737
+ if (typeof params.model !== 'string' || !servedModelIds(root).has(params.model))
738
+ return unknownModel(String(params.model));
739
+ const inputs = Array.isArray(params.input) ? params.input.map((v) => (typeof v === 'string' ? v : JSON.stringify(v))) : [String(params.input)];
740
+ if (inputs.some((s) => s.length === 0))
741
+ return invalidRequest("'input' must not be an empty string");
742
+ // `dimensions` is validated BEFORE any allocation: the spec types it `anyOf [integer, null]`
743
+ // (EmbeddingRequest), so pydantic answers a non-integer with int_parsing; the range below is
744
+ // the vendor's own constraint — dimensions must be ≥ 1 (the API reference schema marks
745
+ // minimum 1) and ≤ the model's native dimensionality (the vendor's resizable models return
746
+ // SHORTER vectors — matryoshka truncation — never longer; the catalog's largest is the Qwen3
747
+ // embedding table's 4096). An unvalidated client number flowed straight into the vector
748
+ // allocation: 1e9 tried to allocate a 12 GB array (one-request DoS), -1/2.5 escaped as a 500,
749
+ // 0 answered a fake 200 with an empty vector, 'x' was silently ignored.
750
+ const dimsRaw = params.dimensions;
751
+ if (dimsRaw !== undefined && dimsRaw !== null) {
752
+ if (typeof dimsRaw !== 'number' || Number.isNaN(dimsRaw))
753
+ return validationError(['body', 'dimensions'], 'Input should be a valid integer', 'int_parsing');
754
+ if (!Number.isInteger(dimsRaw))
755
+ return validationError(['body', 'dimensions'], 'Input should be a valid integer', 'int_from_float');
756
+ if (dimsRaw < 1)
757
+ return validationError(['body', 'dimensions'], 'Input should be greater than or equal to 1', 'greater_than_equal');
758
+ if (dimsRaw > MAX_EMBEDDING_DIMENSIONS)
759
+ return validationError(['body', 'dimensions'], `Input should be less than or equal to ${MAX_EMBEDDING_DIMENSIONS}`, 'less_than_equal');
760
+ }
761
+ const dims = typeof dimsRaw === 'number' ? dimsRaw : 768;
762
+ const encoding = params.encoding_format === undefined ? 'float' : params.encoding_format;
763
+ if (encoding !== 'float' && encoding !== 'base64')
764
+ return invalidRequest("'encoding_format' must be one of 'float', 'base64'");
765
+ let promptTokens = 0;
766
+ const data = inputs.map((text, index) => {
767
+ promptTokens += estimateTokens(text);
768
+ const vec = pseudoEmbedding(text, dims);
769
+ return {
770
+ object: 'embedding',
771
+ index,
772
+ embedding: encoding === 'base64' ? Buffer.from(new Float32Array(vec).buffer).toString('base64') : vec,
773
+ };
774
+ });
775
+ return {
776
+ status: 200,
777
+ body: {
778
+ object: 'list',
779
+ data,
780
+ model: params.model,
781
+ usage: { prompt_tokens: promptTokens, total_tokens: promptTokens },
782
+ },
783
+ };
784
+ }
785
+ function handleRerank(params, root) {
786
+ if (params.query === undefined)
787
+ return { status: 422, body: { detail: [{ loc: ['body', 'query'], msg: 'Field required', type: 'missing' }] } };
788
+ if (!Array.isArray(params.documents) || params.documents.length === 0)
789
+ return { status: 422, body: { detail: [{ loc: ['body', 'documents'], msg: 'Field required', type: 'missing' }] } };
790
+ // The vendor's rerank takes a model too (the Qwen3 Reranker family); an unknown id is refused
791
+ // the same way chat/embeddings refuse theirs. A MISSING model is accepted (the twin's own
792
+ // query/document scorer needs no id; the vendor's request schema marks model optional there).
793
+ if (params.model !== undefined && (typeof params.model !== 'string' || !servedModelIds(root).has(params.model)))
794
+ return unknownModel(String(params.model));
795
+ const query = String(params.query);
796
+ const documents = params.documents.map(String);
797
+ const returnDocuments = params.return_documents !== false;
798
+ const scored = documents.map((doc, index) => ({ index, score: stubRelevanceScore(query, doc), doc }));
799
+ // Ordered by relevance score (highest first) — the vendor's own contract for the data array.
800
+ scored.sort((a, b) => b.score - a.score);
801
+ // `top_n` is the spec's integer (RerankRequestBody); a non-integer is refused before it can
802
+ // slice, and a value below 1 would answer an EMPTY result list for a valid request — the
803
+ // vendor's own contract keeps at least one result.
804
+ const topRaw = params.top_n;
805
+ if (topRaw !== undefined && topRaw !== null) {
806
+ if (typeof topRaw !== 'number' || Number.isNaN(topRaw))
807
+ return validationError(['body', 'top_n'], 'Input should be a valid integer', 'int_parsing');
808
+ if (!Number.isInteger(topRaw))
809
+ return validationError(['body', 'top_n'], 'Input should be a valid integer', 'int_from_float');
810
+ if (topRaw < 1)
811
+ return validationError(['body', 'top_n'], 'Input should be greater than or equal to 1', 'greater_than_equal');
812
+ }
813
+ const topN = typeof topRaw === 'number' ? topRaw : documents.length;
814
+ const data = scored.slice(0, Math.max(0, topN)).map((s) => ({
815
+ index: s.index,
816
+ relevance_score: s.score,
817
+ ...(returnDocuments ? { document: s.doc } : {}),
818
+ }));
819
+ const promptTokens = estimateTokens(query) + documents.reduce((sum, d) => sum + estimateTokens(d), 0);
820
+ return {
821
+ status: 200,
822
+ body: {
823
+ object: 'list',
824
+ model: params.model ?? null,
825
+ data,
826
+ usage: { prompt_tokens: promptTokens, total_tokens: promptTokens },
827
+ },
828
+ };
829
+ }
830
+ // ── CONTROL PLANE (Gateway REST API) ───────────────────────────────────────────────────
831
+ // google.rpc-style resources: `name` fields follow accounts/<account>/…, `state`/`status` are
832
+ // vendor enums, list envelopes carry nextPageToken/totalSize, and CREATE ids arrive as QUERY
833
+ // params (deployments/datasets/users) or in the body (datasets carry datasetId in the body too).
834
+ function resourceView(r) {
835
+ // The vendor's control-plane schemas (gatewayDeployment, gatewayDataset, …) carry NO `id`
836
+ // field — identity is the hierarchical `name`. The bare id stays a twin-internal alias and
837
+ // must never appear on the wire (defect: strip() was letting it through on every body).
838
+ const { id: _drop, ...rest } = r;
839
+ return strip(rest);
840
+ }
841
+ function paginate(items, url) {
842
+ const pageSize = Number(url.searchParams.get('pageSize') ?? 0);
843
+ const pageToken = url.searchParams.get('pageToken');
844
+ let start = 0;
845
+ if (pageToken) {
846
+ const n = Number(pageToken);
847
+ start = Number.isFinite(n) && n > 0 ? n : 0;
848
+ }
849
+ const size = Number.isFinite(pageSize) && pageSize > 0 ? pageSize : items.length;
850
+ const page = items.slice(start, start + size);
851
+ const next = start + size < items.length ? String(start + size) : null;
852
+ return { items: page, nextPageToken: next };
853
+ }
854
+ /** The gateway status embedded on resources: OK once READY, else the resource's own state. */
855
+ function okStatus() {
856
+ return { code: 'OK', message: '' };
857
+ }
858
+ async function createControlResource(type, prefix, accountId, params, req, url, buildFields) {
859
+ // The vendor passes create ids where its spec puts them: QUERY params for deployments
860
+ // (deploymentId), users (userId), batchInferenceJobId and supervisedFineTuningJobId; BODY
861
+ // fields for datasets ({dataset, datasetId}) and models ({model, modelId}); the resource's own
862
+ // `name` for secrets. Absent → the vendor mints one; so does the twin.
863
+ const queryId = url.searchParams.get(`${type}Id`) ?? url.searchParams.get(`${type}_id`);
864
+ const wrapped = (params[type] && typeof params[type] === 'object' ? params[type] : params);
865
+ const bodyId = typeof wrapped[`${type}Id`] === 'string' ? wrapped[`${type}Id`]
866
+ : typeof params[`${type}Id`] === 'string' ? params[`${type}Id`]
867
+ // A secret's id is the last segment of the `name` the client supplies in its own body.
868
+ : type === 'secret' && typeof params.name === 'string' ? params.name.split('/').pop()
869
+ : undefined;
870
+ const id = queryId || bodyId || nextId(type, prefix, accountId, req.root);
871
+ // The vendor's own spec marks the id REQUIRED on datasets (CreateDatasetRequest.required:
872
+ // dataset + datasetId) and models (GatewayGatewayCreateModelBody.required: modelId) — a create
873
+ // without one is a 400 at the vendor, never a mint. The dataset body itself is required too
874
+ // (the same required list); a bare {datasetId} is not a create. The other collections' id
875
+ // params are optional (the vendor mints), so the twin keeps minting there.
876
+ if (type === 'dataset' || type === 'model') {
877
+ if (!queryId && !bodyId)
878
+ return gatewayError(400, `${type}Id is required`);
879
+ if (type === 'dataset' && !(params.dataset && typeof params.dataset === 'object')) {
880
+ return gatewayError(400, 'dataset is required');
881
+ }
882
+ }
883
+ // TENANCY: the kernel subject is the ACCOUNT-NAMESPACED id (`{account}/{id}` — the vendor's own
884
+ // name grammar), so the kernel's (type, subject) key is unique per tenant. Writing the bare id
885
+ // let a create under account B with an id account A held DESTROY A's row (the overlay folds by
886
+ // the bare id globally); the duplicate check below is therefore per-account by construction.
887
+ const subject = `${accountId}/${id}`;
888
+ if (getRow(type, id, accountId, req.root) && !isTombstoned(getRow(type, id, accountId, req.root))) {
889
+ return gatewayError(409, `Resource already exists: ${id}`);
890
+ }
891
+ const fields = buildFields(params);
892
+ // The account the create rode under is part of the resource's identity: every later read is
893
+ // scoped to it, so a row created under acct-A is invisible under acct-B's paths (the vendor's
894
+ // own tenancy). The `_` prefix keeps it off the wire (strip() drops it).
895
+ fields._account = accountId;
896
+ // The id is resolved HERE (query param, body field, or the twin's mint) — so the row's `name`
897
+ // is built from it. A caller-supplied buildFields cannot know the minted id; its PLACEHOLDER
898
+ // stand-in must never survive to the wire (the vendor names every row with the real id).
899
+ if (typeof fields.name === 'string')
900
+ fields.name = fields.name.replace('PLACEHOLDER', id);
901
+ await applyTwinWrite(SERVICE, {
902
+ operation: `${type}.create`,
903
+ subjectType: type,
904
+ subjectId: subject,
905
+ fields,
906
+ ...(req.occurredAt ? { occurredAt: req.occurredAt } : {}),
907
+ actor: { kind: 'agent' },
908
+ }, req.root);
909
+ return { status: 200, body: resourceView(getRow(type, id, accountId, req.root) ?? { id: subject, ...fields }) };
910
+ }
911
+ async function deleteControlResource(type, id, accountId, req) {
912
+ const r = getRow(type, id, accountId, req.root);
913
+ if (!r || isTombstoned(r))
914
+ return gatewayError(404, `Not found: ${type}/${id}`);
915
+ await applyTwinWrite(SERVICE, {
916
+ operation: `${type}.delete`, subjectType: type, subjectId: `${accountId}/${id}`, fields: { _deleted: true },
917
+ ...(req.occurredAt ? { occurredAt: req.occurredAt } : {}), actor: { kind: 'agent' },
918
+ }, req.root);
919
+ // The vendor's delete operations answer `{}` (an empty object per its own spec).
920
+ return { status: 200, body: {} };
921
+ }
922
+ const DEPLOYMENT_STATES = ['STATE_UNSPECIFIED', 'CREATING', 'READY', 'DELETING', 'FAILED', 'UPDATING', 'DELETED'];
923
+ const JOB_STATES = ['JOB_STATE_UNSPECIFIED', 'JOB_STATE_CREATING', 'JOB_STATE_RUNNING', 'JOB_STATE_COMPLETED', 'JOB_STATE_FAILED', 'JOB_STATE_CANCELLED', 'JOB_STATE_DELETING', 'JOB_STATE_WRITING_RESULTS', 'JOB_STATE_VALIDATING', 'JOB_STATE_DELETING_CLEANING_UP', 'JOB_STATE_PENDING', 'JOB_STATE_EXPIRED', 'JOB_STATE_RE_QUEUEING', 'JOB_STATE_CREATING_INPUT_DATASET', 'JOB_STATE_IDLE', 'JOB_STATE_CANCELLING', 'JOB_STATE_EARLY_STOPPED', 'JOB_STATE_PAUSED', 'JOB_STATE_DELETED', 'JOB_STATE_ARCHIVED'];
924
+ const USER_STATES = ['STATE_UNSPECIFIED', 'CREATING', 'READY', 'UPDATING', 'DELETING'];
925
+ const USER_ROLES = ['admin', 'user', 'contributor', 'inference-user', 'custom'];
926
+ // ── public entry: cross-cutting protocol (auth) then route ──────────────────────────────
927
+ export async function handleFireworksTwinRequest(req) {
928
+ const method = req.method.toUpperCase();
929
+ if (req.headers !== undefined || req.apiKey !== undefined) {
930
+ const authErr = checkAuth(req);
931
+ if (authErr)
932
+ return authErr;
933
+ }
934
+ return routeFireworks(req, method);
935
+ }
936
+ // ── router ──────────────────────────────────────────────────────────────────────────────
937
+ async function routeFireworks(req, method) {
938
+ const path = (req.path.split('?')[0] ?? '/').replace(/\/+$/, '') || '/';
939
+ const query = req.path.includes('?') ? req.path.slice(req.path.indexOf('?')) : '';
940
+ const params = parseJson(req.body);
941
+ const dec = (s) => decodeURIComponent(s);
942
+ // D3: a read-only twin rejects any mutation with a vendor-shaped error — each plane's OWN
943
+ // envelope (the control plane's google.rpc shape, the inference plane's OpenAI shape).
944
+ if (req.readOnly && method !== 'GET') {
945
+ const message = 'twin is read-only; omit readOnly to accept writes';
946
+ if (path.startsWith(`${FIREWORKS_ACCOUNTS_PREFIX}/`))
947
+ return gatewayError(405, message);
948
+ return { status: 405, body: errBody(message, { code: 405 }) };
949
+ }
950
+ const url = new URL(req.path, 'http://twin');
951
+ // ── INFERENCE PLANE (/inference/v1/…) ──────────────────────────────────────────────────
952
+ if (path === FIREWORKS_INFERENCE_PREFIX || path.startsWith(`${FIREWORKS_INFERENCE_PREFIX}/`)) {
953
+ const seg = path.slice(FIREWORKS_INFERENCE_PREFIX.length).replace(/^\/+/, '').split('/').filter(Boolean);
954
+ // The Anthropic-compat surface has its OWN envelope family (AnthropicErrorResponse), so it
955
+ // routes separately from the OpenAI-compat operations.
956
+ if (seg[0] === 'messages' && seg.length === 1 && method === 'POST') {
957
+ if (params.stream === true) {
958
+ if (!req.sseSink)
959
+ return anthropicError(400, 'invalid_request_error', 'streaming requires an SSE-capable connection');
960
+ return streamAnthropicMessage(params, req.sseSink, req.occurredAt);
961
+ }
962
+ return buildAnthropicMessage(params, req.occurredAt);
963
+ }
964
+ if (seg[0] === 'chat' && seg[1] === 'completions' && seg.length === 2 && method === 'POST') {
965
+ const validated = validateChat(params);
966
+ if ('error' in validated)
967
+ return validated.error;
968
+ const args = validated.args;
969
+ // The vendor refuses an unknown model id BEFORE any generation (404 "Model id not found");
970
+ // an account-owned model (created/pulled through the control plane) is addressable too.
971
+ if (!servedModelIds(req.root).has(args.model))
972
+ return unknownModel(args.model);
973
+ // R15 — the scenario engine decides AND honors a fault here, before any completion exists:
974
+ // a `status` fault is this vendor's own refusal envelope (rate limit, server error), a
975
+ // `slow` has already held the answer, a `drop` never returns. The realizers below only see
976
+ // a content decision.
977
+ let decision;
978
+ if (req.scenarioEngine) {
979
+ const served = await req.scenarioEngine.serve({ model: args.model, messages: args.messages, tools: args.tools, serviceTier: args.serviceTier });
980
+ if (served.kind === 'fault')
981
+ return { status: served.result.status, body: served.result.body, headers: served.result.headers };
982
+ decision = served;
983
+ }
984
+ const result = args.stream && req.sseSink ? streamChat(args, req.sseSink, req.occurredAt, decision) : buildChatCompletion(args, req.occurredAt, decision);
985
+ if ('status' in result)
986
+ return result;
987
+ return { status: 200, body: result };
988
+ }
989
+ if (seg[0] === 'completions' && seg.length === 1 && method === 'POST')
990
+ return handleCompletion(params, req.occurredAt, req.sseSink, req.root);
991
+ if (seg[0] === 'responses' && seg.length === 1 && method === 'POST')
992
+ return createResponse(params, req);
993
+ if (seg[0] === 'responses' && seg.length === 1 && method === 'GET')
994
+ return handleListResponses(req);
995
+ if (seg[0] === 'responses' && seg.length === 2 && method === 'GET')
996
+ return handleGetResponse(dec(seg[1]), req);
997
+ if (seg[0] === 'responses' && seg.length === 2 && method === 'DELETE')
998
+ return handleDeleteResponse(dec(seg[1]), req);
999
+ if (seg[0] === 'embeddings' && seg.length === 1 && method === 'POST')
1000
+ return handleEmbeddings(params, req.root);
1001
+ if (seg[0] === 'rerank' && seg.length === 1 && method === 'POST')
1002
+ return handleRerank(params, req.root);
1003
+ return notFound(`Unknown request URL: ${method} ${path}.`);
1004
+ }
1005
+ // ── CONTROL PLANE (/v1/accounts/{account_id}/…) ────────────────────────────────────────
1006
+ if (path.startsWith(`${FIREWORKS_ACCOUNTS_PREFIX}/`)) {
1007
+ const seg = path.slice(FIREWORKS_ACCOUNTS_PREFIX.length).split('/').filter(Boolean).map(dec);
1008
+ const accountId = seg[0];
1009
+ if (!accountId)
1010
+ return gatewayError(404, `Unknown request URL: ${method} ${path}.`);
1011
+ const rest = seg.slice(1);
1012
+ // The account itself: GET /v1/accounts/{account_id}. The twin holds NO account rows — an
1013
+ // account exists only as the path prefix its resources live under, and the vendor's account
1014
+ // row is one-per-credential vendor state the API cannot create. Fabricating a 200 row for
1015
+ // ANY id (the old behavior) was a fake success; the vendor answers 404 for an account id it
1016
+ // does not know, and that is what the twin answers too. The honest gap is filed as
1017
+ // fireworks.control.get_account / fireworks.account.read (todos).
1018
+ if (rest.length === 0 && method === 'GET') {
1019
+ return gatewayError(404, `Not found: accounts/${accountId}`);
1020
+ }
1021
+ const resource = rest[0];
1022
+ // Verb-suffixed custom methods: `:cancel`, `:resume`, `:promote`, … (the vendor's AIP-158
1023
+ // style). The colon rides EITHER the collection segment (`jobs:cancel` — collection-level
1024
+ // verbs) or the id segment (`conf-job:cancel` — resource-level verbs); the id is the segment
1025
+ // before the verb.
1026
+ const colonAt = resource?.indexOf(':') ?? -1;
1027
+ let verb = colonAt >= 0 ? resource.slice(colonAt + 1) : undefined;
1028
+ const baseResource = colonAt >= 0 ? resource.slice(0, colonAt) : resource;
1029
+ let idSeg = rest[1];
1030
+ if (verb === undefined && rest.length >= 2 && rest[1].includes(':')) {
1031
+ const at = rest[1].indexOf(':');
1032
+ verb = rest[1].slice(at + 1);
1033
+ idSeg = rest[1].slice(0, at);
1034
+ }
1035
+ const TYPE_MAP = {
1036
+ deployments: { plural: 'deployments', states: DEPLOYMENT_STATES, stateKey: 'state' },
1037
+ datasets: { plural: 'datasets', states: ['STATE_UNSPECIFIED', 'UPLOADING', 'READY'], stateKey: 'state' },
1038
+ batchInferenceJobs: { plural: 'batchInferenceJobs', states: JOB_STATES, stateKey: 'state' },
1039
+ supervisedFineTuningJobs: { plural: 'supervisedFineTuningJobs', states: JOB_STATES, stateKey: 'state' },
1040
+ users: { plural: 'users', states: USER_STATES, stateKey: 'state' },
1041
+ models: { plural: 'models', states: DEPLOYMENT_STATES, stateKey: 'state' },
1042
+ secrets: { plural: 'secrets', states: DEPLOYMENT_STATES, stateKey: 'state' },
1043
+ };
1044
+ const meta = baseResource ? TYPE_MAP[baseResource] : undefined;
1045
+ if (!meta)
1046
+ return gatewayError(404, `Unknown request URL: ${method} ${path}.`);
1047
+ // The PATCH whitelist, per collection: exactly the MUTABLE fields the vendor's own update
1048
+ // operations accept (create-only and output-only fields are absent — a PATCH carrying one is
1049
+ // refused below, the way grpc-gateway refuses an unknown field; googleads-twin.ts is the
1050
+ // estate precedent). Sourced from the pinned spec's own update-operation schemas:
1051
+ // Gateway_UpdateDataset (displayName, exampleCount, userUploaded, evaluationResult,
1052
+ // transformed, splitted, evalProtocol, externalUrl, format, sourceJobName), the deployments
1053
+ // PATCH names the deployment fields, users PATCH names the user fields, secrets and models
1054
+ // the gatewaySecret/gatewayModel fields. The job collections carry NO update op in the spec
1055
+ // (delete+get only) — a PATCH there is the vendor's 404 (route not found), never a served
1056
+ // no-op, so the collections are absent from this map entirely.
1057
+ const PATCHABLE = {
1058
+ deployments: ['displayName', 'region', 'replicaCount', 'minReplicaCount', 'maxReplicaCount', 'precision'],
1059
+ users: ['displayName', 'email', 'role', 'permissionPreset'],
1060
+ models: ['displayName', 'contextLength', 'description'],
1061
+ secrets: ['keyName'],
1062
+ datasets: ['displayName', 'exampleCount', 'userUploaded', 'evaluationResult', 'transformed', 'splitted', 'evalProtocol', 'externalUrl', 'format', 'sourceJobName'],
1063
+ };
1064
+ // Rows are stored under the SINGULAR type (the same noun createControlResource writes); the
1065
+ // URL carries the plural collection.
1066
+ const singular = baseResource === 'deployments' ? 'deployment' : baseResource === 'datasets' ? 'dataset' : baseResource === 'batchInferenceJobs' ? 'batchInferenceJob' : baseResource === 'supervisedFineTuningJobs' ? 'supervisedFineTuningJob' : baseResource === 'users' ? 'user' : baseResource === 'models' ? 'model' : 'secret';
1067
+ // LIST: GET /v1/accounts/{id}/<plural>
1068
+ if (rest.length === 1 && method === 'GET') {
1069
+ const all = rows(singular, accountId, req.root).filter((r) => !isTombstoned(r)).map(resourceView);
1070
+ const { items, nextPageToken } = paginate(all, url);
1071
+ return { status: 200, body: { [meta.plural]: items, nextPageToken, totalSize: all.length } };
1072
+ }
1073
+ // CREATE: POST /v1/accounts/{id}/<plural> — and ONLY that. A verb-suffixed collection
1074
+ // (`<plural>:estimateCost`, `apiKeys:<anything>`) is a custom method, not a create: the
1075
+ // vendor answers it with its own 404 (or serves it), never by running the create builder —
1076
+ // letting `POST …/supervisedFineTuningJobs:estimateCost` with a dataset body CREATE a job
1077
+ // would make an estimate mutate state.
1078
+ if (rest.length === 1 && method === 'POST' && !verb) {
1079
+ const build = (p) => {
1080
+ // The vendor's create bodies are the RESOURCE DIRECTLY for deployments,
1081
+ // batchInferenceJobs, supervisedFineTuningJobs, users and secrets (requestBody →
1082
+ // $ref gatewayX). Datasets and models wrap: {dataset, datasetId} / {model, modelId}.
1083
+ const inner = (baseResource === 'datasets' && p.dataset && typeof p.dataset === 'object' ? p.dataset
1084
+ : baseResource === 'models' && p.model && typeof p.model === 'object' ? p.model
1085
+ : p);
1086
+ const at = new Date(req.occurredAt ?? '1970-01-01T00:00:00Z').toISOString().replace(/\.\d{3}Z$/, 'Z');
1087
+ if (baseResource === 'deployments') {
1088
+ if (inner.baseModel === undefined)
1089
+ return { __error: 'baseModel is required' };
1090
+ return {
1091
+ name: `accounts/${accountId}/deployments/PLACEHOLDER`,
1092
+ displayName: inner.displayName ?? '',
1093
+ baseModel: inner.baseModel,
1094
+ state: 'CREATING',
1095
+ status: okStatus(),
1096
+ createTime: at,
1097
+ updateTime: at,
1098
+ ...(inner.region !== undefined ? { region: inner.region } : {}),
1099
+ ...(inner.replicaCount !== undefined ? { replicaCount: inner.replicaCount } : {}),
1100
+ ...(inner.minReplicaCount !== undefined ? { minReplicaCount: inner.minReplicaCount } : {}),
1101
+ ...(inner.maxReplicaCount !== undefined ? { maxReplicaCount: inner.maxReplicaCount } : {}),
1102
+ ...(inner.precision !== undefined ? { precision: inner.precision } : {}),
1103
+ };
1104
+ }
1105
+ if (baseResource === 'datasets') {
1106
+ return {
1107
+ name: `accounts/${accountId}/datasets/PLACEHOLDER`,
1108
+ displayName: inner.displayName ?? '',
1109
+ state: 'UPLOADING',
1110
+ status: okStatus(),
1111
+ createTime: at,
1112
+ updateTime: at,
1113
+ exampleCount: 0,
1114
+ userUploaded: true,
1115
+ ...(inner.format !== undefined ? { format: inner.format } : {}),
1116
+ };
1117
+ }
1118
+ if (baseResource === 'batchInferenceJobs') {
1119
+ // The job's dataset reference must name a dataset that really landed: the vendor
1120
+ // validates the reference at create, and a twin that accepted any string would let a
1121
+ // mistyped id "create" a job over nothing.
1122
+ if (inner.inputDatasetId !== undefined) {
1123
+ const dsId = String(inner.inputDatasetId).split('/').pop() ?? '';
1124
+ const ds = getRow('dataset', dsId, accountId, req.root);
1125
+ if (!ds || isTombstoned(ds))
1126
+ return { __error: `inputDatasetId not found: ${String(inner.inputDatasetId)}` };
1127
+ }
1128
+ return {
1129
+ name: `accounts/${accountId}/batchInferenceJobs/PLACEHOLDER`,
1130
+ displayName: inner.displayName ?? '',
1131
+ state: 'JOB_STATE_CREATING',
1132
+ status: okStatus(),
1133
+ createTime: at,
1134
+ updateTime: at,
1135
+ ...(inner.model !== undefined ? { model: inner.model } : {}),
1136
+ ...(inner.inputDatasetId !== undefined ? { inputDatasetId: inner.inputDatasetId } : {}),
1137
+ ...(inner.outputDatasetId !== undefined ? { outputDatasetId: inner.outputDatasetId } : {}),
1138
+ };
1139
+ }
1140
+ if (baseResource === 'supervisedFineTuningJobs') {
1141
+ if (inner.dataset === undefined)
1142
+ return { __error: 'dataset is required' };
1143
+ const dsId = String(inner.dataset).split('/').pop() ?? '';
1144
+ const ds = getRow('dataset', dsId, accountId, req.root);
1145
+ if (!ds || isTombstoned(ds))
1146
+ return { __error: `dataset not found: ${String(inner.dataset)}` };
1147
+ return {
1148
+ name: `accounts/${accountId}/supervisedFineTuningJobs/PLACEHOLDER`,
1149
+ displayName: inner.displayName ?? '',
1150
+ dataset: inner.dataset,
1151
+ state: 'JOB_STATE_CREATING',
1152
+ status: okStatus(),
1153
+ createTime: at,
1154
+ updateTime: at,
1155
+ ...(inner.baseModel !== undefined ? { baseModel: inner.baseModel } : {}),
1156
+ ...(inner.epochs !== undefined ? { epochs: inner.epochs } : {}),
1157
+ ...(inner.learningRate !== undefined ? { learningRate: inner.learningRate } : {}),
1158
+ };
1159
+ }
1160
+ if (baseResource === 'users') {
1161
+ if (inner.role === undefined)
1162
+ return { __error: 'role is required' };
1163
+ if (!USER_ROLES.includes(inner.role))
1164
+ return { __error: `role must be one of ${USER_ROLES.join(', ')}` };
1165
+ return {
1166
+ name: `accounts/${accountId}/users/PLACEHOLDER`,
1167
+ displayName: inner.displayName ?? '',
1168
+ email: inner.email ?? null,
1169
+ role: inner.role,
1170
+ state: 'CREATING',
1171
+ status: okStatus(),
1172
+ createTime: at,
1173
+ updateTime: at,
1174
+ ...(inner.permissionPreset !== undefined ? { permissionPreset: inner.permissionPreset } : {}),
1175
+ };
1176
+ }
1177
+ if (baseResource === 'secrets') {
1178
+ // The create body IS the gatewaySecret (the spec takes the resource directly), with
1179
+ // `name` and `keyName` required. `name` is the FULL resource name — a bare last
1180
+ // segment is accepted (clients generated from the spec send the short spelling) and
1181
+ // stored canonically; a full name must not be double-prefixed.
1182
+ if (inner.name === undefined || inner.keyName === undefined)
1183
+ return { __error: 'name and keyName are required' };
1184
+ const shortName = String(inner.name).split('/').pop();
1185
+ return {
1186
+ name: `accounts/${accountId}/secrets/${shortName}`,
1187
+ keyName: inner.keyName,
1188
+ // The vendor never returns the secret value on ANY read, including the create
1189
+ // response. The twin keeps it out of the stored row entirely — a secret that can be
1190
+ // read back is a custody lie.
1191
+ createTime: at,
1192
+ updateTime: at,
1193
+ };
1194
+ }
1195
+ if (baseResource === 'models') {
1196
+ return {
1197
+ name: `accounts/${accountId}/models/PLACEHOLDER`,
1198
+ displayName: inner.displayName ?? '',
1199
+ state: 'CREATING',
1200
+ status: okStatus(),
1201
+ createTime: at,
1202
+ updateTime: at,
1203
+ public: false,
1204
+ ...(inner.baseModelDetails !== undefined ? { baseModelDetails: inner.baseModelDetails } : {}),
1205
+ ...(inner.contextLength !== undefined ? { contextLength: inner.contextLength } : {}),
1206
+ ...(inner.description !== undefined ? { description: inner.description } : {}),
1207
+ };
1208
+ }
1209
+ return {};
1210
+ };
1211
+ const fields = build(params);
1212
+ if (fields.__error !== undefined)
1213
+ return gatewayError(400, String(fields.__error));
1214
+ // Replace the PLACEHOLDER name with the real id after the id is known. The singular type
1215
+ // name is what the id params are built from (deploymentId, userId, batchInferenceJobId…).
1216
+ const type = baseResource === 'deployments' ? 'deployment' : baseResource === 'datasets' ? 'dataset' : baseResource === 'batchInferenceJobs' ? 'batchInferenceJob' : baseResource === 'supervisedFineTuningJobs' ? 'supervisedFineTuningJob' : baseResource === 'users' ? 'user' : baseResource === 'models' ? 'model' : 'secret';
1217
+ const prefix = baseResource === 'batchInferenceJobs' ? 'bij' : baseResource === 'supervisedFineTuningJobs' ? 'sft' : type === 'secret' ? 'secret' : `${type}`;
1218
+ return createControlResource(type, prefix, accountId, params, req, url, (p) => build(p));
1219
+ }
1220
+ // SINGLE RESOURCE: GET/PATCH/DELETE /v1/accounts/{id}/<plural>/<resource_id>
1221
+ if (rest.length === 2 && !verb) {
1222
+ const rid = idSeg;
1223
+ // The lookup is ACCOUNT-SCOPED: a row created under another account is invisible here and
1224
+ // answers the same NOT_FOUND an unknown id answers (the vendor's own tenancy — an account
1225
+ // cannot see, patch or delete another account's resource).
1226
+ const r = getRow(singular, rid, accountId, req.root);
1227
+ if (method === 'GET') {
1228
+ if (!r || isTombstoned(r))
1229
+ return gatewayError(404, `Not found: ${baseResource}/${rid}`);
1230
+ return { status: 200, body: resourceView(r) };
1231
+ }
1232
+ if (method === 'PATCH') {
1233
+ // ONE policy for output-only fields across every collection (grpc-gateway's own): the
1234
+ // spec's readOnly fields are REFUSED with the same unknown-field 400 an invented field
1235
+ // gets — "Cannot find field." — never silently dropped and never written. (Silently
1236
+ // ignoring them on deployments while datasets refused them was two policies on one
1237
+ // plane; the refusal is grpc-gateway's own behavior for a client-set readOnly field.)
1238
+ if (!PATCHABLE[baseResource])
1239
+ return gatewayError(404, `Unknown request URL: ${method} ${path}.`);
1240
+ if (!r || isTombstoned(r))
1241
+ return gatewayError(404, `Not found: ${baseResource}/${rid}`);
1242
+ const patchable = { ...params };
1243
+ // The per-collection field whitelist (the same one create uses): an undeclared field is
1244
+ // refused the way grpc-gateway refuses an unknown field, and an output-only field never
1245
+ // passes (googleads-twin.ts is the estate precedent for the refusal shape).
1246
+ const unknown = Object.keys(patchable).filter((k) => !PATCHABLE[baseResource]?.includes(k));
1247
+ if (unknown.length)
1248
+ return gatewayError(400, `Invalid JSON payload received. Unknown name "${unknown[0]}": Cannot find field.`);
1249
+ await applyTwinWrite(SERVICE, {
1250
+ operation: `${singular}.update`, subjectType: singular, subjectId: `${accountId}/${rid}`,
1251
+ fields: patchable,
1252
+ ...(req.occurredAt ? { occurredAt: req.occurredAt } : {}), actor: { kind: 'agent' },
1253
+ }, req.root);
1254
+ return { status: 200, body: resourceView(getRow(singular, rid, accountId, req.root) ?? {}) };
1255
+ }
1256
+ if (method === 'DELETE')
1257
+ return deleteControlResource(singular, rid, accountId, req);
1258
+ }
1259
+ // CUSTOM VERB: POST /v1/accounts/{id}/<plural>/<resource_id>:<verb>
1260
+ if (rest.length === 2 && verb && method === 'POST') {
1261
+ const rid = idSeg;
1262
+ const r = getRow(singular, rid, accountId, req.root);
1263
+ // :undelete addresses a DELETED row by design; every other verb needs a live row.
1264
+ if (!r || (isTombstoned(r) && verb !== 'undelete'))
1265
+ return gatewayError(404, `Not found: ${baseResource}/${rid}`);
1266
+ if (verb === 'cancel' && (baseResource === 'batchInferenceJobs' || baseResource === 'supervisedFineTuningJobs')) {
1267
+ if (r.state === 'JOB_STATE_CANCELLED' || r.state === 'JOB_STATE_CANCELLING') {
1268
+ return gatewayError(400, `Cannot cancel a job in state ${String(r.state)}`);
1269
+ }
1270
+ await applyTwinWrite(SERVICE, {
1271
+ operation: `${singular}.cancel`, subjectType: singular, subjectId: `${accountId}/${rid}`,
1272
+ fields: { state: 'JOB_STATE_CANCELLING' },
1273
+ ...(req.occurredAt ? { occurredAt: req.occurredAt } : {}), actor: { kind: 'agent' },
1274
+ }, req.root);
1275
+ // The vendor's cancel answers `{}`.
1276
+ return { status: 200, body: {} };
1277
+ }
1278
+ if (verb === 'resume' && (baseResource === 'batchInferenceJobs' || baseResource === 'supervisedFineTuningJobs')) {
1279
+ await applyTwinWrite(SERVICE, {
1280
+ operation: `${singular}.resume`, subjectType: singular, subjectId: `${accountId}/${rid}`,
1281
+ fields: { state: 'JOB_STATE_RUNNING' },
1282
+ ...(req.occurredAt ? { occurredAt: req.occurredAt } : {}), actor: { kind: 'agent' },
1283
+ }, req.root);
1284
+ return { status: 200, body: {} };
1285
+ }
1286
+ if (verb === 'undelete' && baseResource === 'deployments') {
1287
+ await applyTwinWrite(SERVICE, {
1288
+ operation: 'deployment.undelete', subjectType: 'deployment', subjectId: `${accountId}/${rid}`,
1289
+ fields: { _deleted: false, state: 'CREATING' },
1290
+ ...(req.occurredAt ? { occurredAt: req.occurredAt } : {}), actor: { kind: 'agent' },
1291
+ }, req.root);
1292
+ return { status: 200, body: resourceView(getRow('deployment', rid, accountId, req.root) ?? {}) };
1293
+ }
1294
+ if (verb === 'scale' && baseResource === 'deployments' && method === 'POST') {
1295
+ const replicaCount = params.replicaCount;
1296
+ if (typeof replicaCount !== 'number')
1297
+ return gatewayError(400, 'replicaCount is required');
1298
+ await applyTwinWrite(SERVICE, {
1299
+ operation: 'deployment.scale', subjectType: 'deployment', subjectId: `${accountId}/${rid}`,
1300
+ fields: { replicaCount },
1301
+ ...(req.occurredAt ? { occurredAt: req.occurredAt } : {}), actor: { kind: 'agent' },
1302
+ }, req.root);
1303
+ return { status: 200, body: resourceView(getRow('deployment', rid, accountId, req.root) ?? {}) };
1304
+ }
1305
+ return gatewayError(404, `Unknown operation: ${verb} on ${baseResource}`);
1306
+ }
1307
+ // PATCH /v1/accounts/{id}/deployments/{deployment_id}:scale — the spec's own spelling
1308
+ // (Gateway_ScaleDeployment, body {replicaCount}); it answers `{}`, not the resource.
1309
+ if (rest.length === 2 && verb === 'scale' && baseResource === 'deployments' && method === 'PATCH') {
1310
+ const rid = idSeg;
1311
+ const r = getRow('deployment', rid, accountId, req.root);
1312
+ if (!r || isTombstoned(r))
1313
+ return gatewayError(404, `Not found: deployments/${rid}`);
1314
+ const replicaCount = params.replicaCount;
1315
+ if (typeof replicaCount !== 'number')
1316
+ return gatewayError(400, 'replicaCount is required');
1317
+ await applyTwinWrite(SERVICE, {
1318
+ operation: 'deployment.scale', subjectType: 'deployment', subjectId: `${accountId}/${rid}`,
1319
+ fields: { replicaCount },
1320
+ ...(req.occurredAt ? { occurredAt: req.occurredAt } : {}), actor: { kind: 'agent' },
1321
+ }, req.root);
1322
+ return { status: 200, body: {} };
1323
+ }
1324
+ // users/{user_id}/apiKeys — nested under users. The verb-suffixed collection spelling
1325
+ // (`apiKeys:delete`) carries its verb on the apiKeys segment itself.
1326
+ if (baseResource === 'users' && rest.length >= 3 && rest[2].split(':')[0] === 'apiKeys') {
1327
+ const userId = rest[1];
1328
+ const user = getRow('user', userId, accountId, req.root);
1329
+ if (!user || user._deleted)
1330
+ return gatewayError(404, `Not found: users/${userId}`);
1331
+ const keySeg = rest.slice(3);
1332
+ if (keySeg.length === 0 && method === 'GET') {
1333
+ // The full key is returned ONCE at creation — the LIST answers it never. A list row
1334
+ // carrying `fw_…` would be a custody lie the single-key GET two branches down doesn't
1335
+ // commit, so the strip happens here too.
1336
+ const keys = rows('apiKey', accountId, req.root).filter((r) => !r._deleted && r._user_id === userId).map(resourceView);
1337
+ for (const k of keys)
1338
+ delete k.key;
1339
+ return { status: 200, body: { apiKeys: keys, nextPageToken: null, totalSize: keys.length } };
1340
+ }
1341
+ // POST .../apiKeys:delete — the vendor's own odd verb-suffixed delete with a {keyId} body
1342
+ // (collection-level spelling, Gateway_DeleteApiKey; answers `{}`).
1343
+ if (keySeg.length === 0 && method === 'POST' && rest[2] === 'apiKeys:delete') {
1344
+ const keyId = typeof params.keyId === 'string' ? params.keyId : undefined;
1345
+ if (!keyId)
1346
+ return gatewayError(400, 'keyId is required');
1347
+ const k = getRow('apiKey', keyId, accountId, req.root);
1348
+ if (!k || k._deleted || k._user_id !== userId)
1349
+ return gatewayError(404, `Not found: apiKeys/${keyId}`);
1350
+ await applyTwinWrite(SERVICE, {
1351
+ operation: 'apiKey.delete', subjectType: 'apiKey', subjectId: `${accountId}/${keyId}`, fields: { _deleted: true },
1352
+ ...(req.occurredAt ? { occurredAt: req.occurredAt } : {}), actor: { kind: 'agent' },
1353
+ }, req.root);
1354
+ return { status: 200, body: {} };
1355
+ }
1356
+ if (keySeg.length === 0 && method === 'POST' && rest[2] === 'apiKeys') {
1357
+ // The verb check is load-bearing: `apiKeys:frobnicate` must 404 below, never mint a key.
1358
+ const inner = (params.apiKey && typeof params.apiKey === 'object' ? params.apiKey : params);
1359
+ const id = nextId('apiKey', 'key', accountId, req.root);
1360
+ const at = new Date(req.occurredAt ?? '1970-01-01T00:00:00Z').toISOString().replace(/\.\d{3}Z$/, 'Z');
1361
+ const key = `fw_${stableSuffix(id + at)}${fnv1a(id).toString(36)}`;
1362
+ await applyTwinWrite(SERVICE, {
1363
+ operation: 'apiKey.create', subjectType: 'apiKey', subjectId: `${accountId}/${id}`,
1364
+ fields: {
1365
+ name: `accounts/${accountId}/users/${userId}/apiKeys/${id}`,
1366
+ displayName: inner.displayName ?? 'default',
1367
+ key, // returned ONCE at creation, never again (the vendor's own contract)
1368
+ prefix: key.slice(0, 6),
1369
+ keyId: id,
1370
+ _user_id: userId,
1371
+ _account: accountId,
1372
+ createTime: at,
1373
+ expireTime: typeof inner.expireTime === 'string' ? inner.expireTime : null,
1374
+ },
1375
+ ...(req.occurredAt ? { occurredAt: req.occurredAt } : {}), actor: { kind: 'agent' },
1376
+ }, req.root);
1377
+ return { status: 200, body: resourceView(getRow('apiKey', id, accountId, req.root) ?? {}) };
1378
+ }
1379
+ if (keySeg.length === 1 && method === 'GET') {
1380
+ const k = getRow('apiKey', keySeg[0], accountId, req.root);
1381
+ if (!k || k._deleted || k._user_id !== userId)
1382
+ return gatewayError(404, `Not found: apiKeys/${keySeg[0]}`);
1383
+ const view = resourceView(k);
1384
+ delete view.key; // "only available upon creation and not stored thereafter"
1385
+ return { status: 200, body: view };
1386
+ }
1387
+ if (keySeg.length === 1 && method === 'PATCH') {
1388
+ const k = getRow('apiKey', keySeg[0], accountId, req.root);
1389
+ if (!k || k._deleted || k._user_id !== userId)
1390
+ return gatewayError(404, `Not found: apiKeys/${keySeg[0]}`);
1391
+ const patchable = { ...params };
1392
+ for (const f of ['key', 'keyId', 'prefix', 'secure', 'email', 'createTime', 'lastUsed', 'isFirepass'])
1393
+ delete patchable[f];
1394
+ const unknown = Object.keys(patchable).filter((f) => !['displayName', 'expireTime'].includes(f));
1395
+ if (unknown.length)
1396
+ return gatewayError(400, `Invalid JSON payload received. Unknown name "${unknown[0]}": Cannot find field.`);
1397
+ await applyTwinWrite(SERVICE, {
1398
+ operation: 'apiKey.update', subjectType: 'apiKey', subjectId: `${accountId}/${keySeg[0]}`,
1399
+ fields: patchable,
1400
+ ...(req.occurredAt ? { occurredAt: req.occurredAt } : {}), actor: { kind: 'agent' },
1401
+ }, req.root);
1402
+ const view = resourceView(getRow('apiKey', keySeg[0], accountId, req.root) ?? {});
1403
+ delete view.key;
1404
+ return { status: 200, body: view };
1405
+ }
1406
+ // POST .../apiKeys:delete — the vendor's own odd verb-suffixed delete with a {keyId} body.
1407
+ // The key must belong to the addressed user: the id segment is under that user's own path,
1408
+ // and deleting another user's key through it would skip the scoping the collection-level
1409
+ // spelling enforces one branch up.
1410
+ if (keySeg.length === 1 && keySeg[0].endsWith(':delete') && method === 'POST') {
1411
+ const keyId = typeof params.keyId === 'string' ? params.keyId : keySeg[0].slice(0, -':delete'.length);
1412
+ const k = getRow('apiKey', keyId, accountId, req.root);
1413
+ if (!k || k._deleted || k._user_id !== userId)
1414
+ return gatewayError(404, `Not found: apiKeys/${keyId}`);
1415
+ await applyTwinWrite(SERVICE, {
1416
+ operation: 'apiKey.delete', subjectType: 'apiKey', subjectId: `${accountId}/${keyId}`, fields: { _deleted: true },
1417
+ ...(req.occurredAt ? { occurredAt: req.occurredAt } : {}), actor: { kind: 'agent' },
1418
+ }, req.root);
1419
+ return { status: 200, body: {} };
1420
+ }
1421
+ }
1422
+ return gatewayError(404, `Unknown request URL: ${method} ${path}.`);
1423
+ }
1424
+ // Not on either plane — the twin serves nothing else.
1425
+ return notFound(`Unknown request URL: ${method} ${path}.`);
1426
+ }