@volter/twin-fireworks 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (39) hide show
  1. package/LICENSE +202 -0
  2. package/README.md +184 -0
  3. package/dist/src/cli.d.ts +2 -0
  4. package/dist/src/cli.js +28 -0
  5. package/dist/src/fireworks-budget.d.ts +54 -0
  6. package/dist/src/fireworks-budget.js +146 -0
  7. package/dist/src/fireworks-capabilities.d.ts +4 -0
  8. package/dist/src/fireworks-capabilities.js +1205 -0
  9. package/dist/src/fireworks-conformance.d.ts +14 -0
  10. package/dist/src/fireworks-conformance.js +514 -0
  11. package/dist/src/fireworks-connector.d.ts +168 -0
  12. package/dist/src/fireworks-connector.js +641 -0
  13. package/dist/src/fireworks-models.d.ts +11 -0
  14. package/dist/src/fireworks-models.js +53 -0
  15. package/dist/src/fireworks-scenario.d.ts +55 -0
  16. package/dist/src/fireworks-scenario.js +171 -0
  17. package/dist/src/fireworks-server.d.ts +16 -0
  18. package/dist/src/fireworks-server.js +144 -0
  19. package/dist/src/fireworks-stub.d.ts +26 -0
  20. package/dist/src/fireworks-stub.js +78 -0
  21. package/dist/src/fireworks-twin.d.ts +51 -0
  22. package/dist/src/fireworks-twin.js +1426 -0
  23. package/dist/src/fireworks-types.d.ts +212 -0
  24. package/dist/src/fireworks-types.js +4 -0
  25. package/dist/src/index.d.ts +9 -0
  26. package/dist/src/index.js +105 -0
  27. package/package.json +52 -0
  28. package/src/cli.ts +27 -0
  29. package/src/fireworks-budget.ts +172 -0
  30. package/src/fireworks-capabilities.ts +1229 -0
  31. package/src/fireworks-conformance.ts +542 -0
  32. package/src/fireworks-connector.ts +700 -0
  33. package/src/fireworks-models.ts +63 -0
  34. package/src/fireworks-scenario.ts +191 -0
  35. package/src/fireworks-server.ts +153 -0
  36. package/src/fireworks-stub.ts +83 -0
  37. package/src/fireworks-twin.ts +1427 -0
  38. package/src/fireworks-types.ts +165 -0
  39. package/src/index.ts +134 -0
@@ -0,0 +1,1427 @@
1
+ // Fireworks twin REQUEST HANDLER — the canonical Fireworks API surface for the twin.
2
+ // Contract: handleFireworksTwinRequest({method, path, body}) -> {status, body}. It is the
3
+ // faithful Fireworks API the official clients (the `fireworks` Python SDK, `@ai-sdk/fireworks`,
4
+ // plain OpenAI clients pointed at api.fireworks.ai/inference/v1) talk to UNMODIFIED.
5
+ //
6
+ // Fireworks serves TWO planes on ONE host, and the twin serves both under the paths the vendor
7
+ // actually publishes:
8
+ // • INFERENCE — `/inference/v1/…` (OpenAI-compatible: chat/completions, completions, responses,
9
+ // embeddings, rerank) plus the Anthropic-compatible `/inference/v1/messages`. The twin cannot
10
+ // run a model, so generative output is a DETERMINISTIC STUB (fireworks-stub.ts), clearly
11
+ // labeled — never pretending to be real inference. The wire around it is faithful, INCLUDING
12
+ // Fireworks' documented OpenAI differences:
13
+ // - streaming usage arrives in the final chunk BY DEFAULT (OpenAI makes it opt-in);
14
+ // - `context_length_exceeded_behavior` defaults to `truncate`, not `error`;
15
+ // - `service_tier` accepts only 'priority' — every other value is treated as 'default'
16
+ // (documented; NOT an error);
17
+ // - validation failures answer the FastAPI `422 HTTPValidationError` envelope the vendor's
18
+ // own spec declares on these operations.
19
+ // • CONTROL — the Gateway REST API under `/v1/accounts/{account_id}/…`: deployments, datasets,
20
+ // batch-inference and supervised-fine-tuning jobs, users + their apiKeys, secrets, models.
21
+ // Stateful over the kernel action log; google.rpc-style errors (gatewayStatus {code,message});
22
+ // list envelopes `{<plural>, nextPageToken, totalSize}`; create ids passed as QUERY params
23
+ // (deployments/datasets/users) or inside the body (datasets), per the vendor's own spec.
24
+ //
25
+ // State lives in the kernel action log (D1): all writes are local actions, reads are the
26
+ // projection. No real Fireworks API is ever called from this path (D4). Streaming uses an
27
+ // INJECTED sink — no real sockets / setTimeout (D5 verify is offline + deterministic).
28
+ import { applyTwinWrite, projectResources } from '@volter/world-core';
29
+ import { FIREWORKS_MODEL_IDS } from './fireworks-models.ts';
30
+ import {
31
+ contentToText,
32
+ countPromptTokens,
33
+ estimateTokens,
34
+ fnv1a,
35
+ lastUserText,
36
+ pseudoEmbedding,
37
+ stubAssistantText,
38
+ stubReasoningText,
39
+ stubRelevanceScore,
40
+ } from './fireworks-stub.ts';
41
+ import type { FireworksScenarioEngine, FireworksScenarioRespond, ScriptedResult } from './fireworks-scenario.ts';
42
+ import { realizeFireworksRespond } from './fireworks-scenario.ts';
43
+ import type { ScenarioDecision } from '@volter/world-core';
44
+ import type {
45
+ FireworksAnthropicContentBlock,
46
+ FireworksAnthropicErrorType,
47
+ FireworksAnthropicMessage,
48
+ FireworksChatCompletion,
49
+ FireworksChatMessage,
50
+ FireworksChatRequest,
51
+ FireworksCompletion,
52
+ FireworksGatewayCode,
53
+ FireworksMessageParam,
54
+ FireworksResponseObject,
55
+ } from './fireworks-types.ts';
56
+
57
+ const SERVICE = 'fireworks';
58
+
59
+ /** The inference base path. The vendor's inference server root is `https://api.fireworks.ai/inference`
60
+ * and every OpenAI-compat operation hangs off `/v1/…` under it. */
61
+ export const FIREWORKS_INFERENCE_PREFIX = '/inference/v1';
62
+ /** The control-plane base path: everything Gateway REST hangs off `/v1/accounts/{account_id}/…`. */
63
+ export const FIREWORKS_ACCOUNTS_PREFIX = '/v1/accounts';
64
+
65
+ export type FireworksRequest = {
66
+ method: string;
67
+ path: string;
68
+ body?: string;
69
+ occurredAt?: string;
70
+ root?: string;
71
+ readOnly?: boolean;
72
+ /** The credential the caller presents (the SDKs' bearer `Authorization` header). When a request
73
+ * carries an auth SURFACE (this field set, or `headers` present), the twin holds it to the real
74
+ * vendor rule: a credential is required → 401 on missing/invalid. In-process trusted calls
75
+ * (capability verify, connector) omit BOTH and are not auth-gated. */
76
+ apiKey?: string;
77
+ /** Lower-cased request headers the HTTP server passes through so the handler can model auth. */
78
+ headers?: Record<string, string>;
79
+ /** When set on a streaming POST, chunks are written here (no sockets). */
80
+ sseSink?: SseSink;
81
+ /** Scenario scripting (fireworks-scenario.ts): when set, chat completions consult the engine
82
+ * first — a matching handler scripts the answer, a miss answers the labeled stub with a
83
+ * pointer to the miss. Twin-only scaffolding, never vendor surface. */
84
+ scenarioEngine?: FireworksScenarioEngine;
85
+ };
86
+
87
+ export type SseEvent = { data?: unknown; done?: boolean };
88
+ export type SseSink = (event: SseEvent) => void;
89
+
90
+ /** The handler response. `headers` (when present) are response headers the HTTP server should set. */
91
+ export type FireworksResponseEnvelope = { status: number; body: unknown; headers?: Record<string, string> };
92
+
93
+ // ── vendor-shaped errors ──────────────────────────────────────────────────────────────
94
+ // TWO error envelopes, one per plane — mixing them would be a wire lie:
95
+ // • the inference plane's OpenAI-compat surface answers `{ error: { message, … } }` (the
96
+ // OpenAI shape the clients parse); the Anthropic-compat /v1/messages answers
97
+ // `{ type: 'error', error: { type, message }, request_id }` (the Anthropic shape);
98
+ // • the control plane answers google.rpc-style `{ code: <gatewayCode>, message }` — the shape
99
+ // the vendor's own spec embeds on every resource (`gatewayStatus`) and its gatewayCode enum.
100
+ function errBody(message: string, extra: Record<string, unknown> = {}) {
101
+ return { error: { message, ...extra } };
102
+ }
103
+ function invalidRequest(message: string, extra: Record<string, unknown> = {}): FireworksResponseEnvelope {
104
+ return { status: 400, body: errBody(message, { code: 400, ...extra }) };
105
+ }
106
+ function notFound(message: string): FireworksResponseEnvelope {
107
+ return { status: 404, body: errBody(message, { code: 404 }) };
108
+ }
109
+ function authError(message: string): FireworksResponseEnvelope {
110
+ return { status: 401, body: errBody(message, { code: 401 }) };
111
+ }
112
+ /** The FastAPI validation envelope the vendor's own spec declares on the inference operations
113
+ * (`422 HTTPValidationError` → `detail: [{loc, msg, type}]`). */
114
+ function validationError(loc: (string | number)[], msg: string, type: string): FireworksResponseEnvelope {
115
+ return { status: 422, body: { detail: [{ loc, msg, type }] } };
116
+ }
117
+ /** The google.rpc-style control-plane error. `code` is a gatewayCode string, mapped from the
118
+ * HTTP status the way the vendor's own status embedding implies (NOT_FOUND → 'NOT_FOUND', …). */
119
+ function gatewayError(status: number, message: string): FireworksResponseEnvelope {
120
+ const code: FireworksGatewayCode =
121
+ status === 400 ? 'INVALID_ARGUMENT' :
122
+ status === 401 ? 'UNAUTHENTICATED' :
123
+ status === 403 ? 'PERMISSION_DENIED' :
124
+ status === 404 ? 'NOT_FOUND' :
125
+ status === 409 ? 'ALREADY_EXISTS' :
126
+ 'UNKNOWN';
127
+ return { status, body: { code, message } };
128
+ }
129
+
130
+ // ── modeled authentication (401) ────────────────────────────────────────────────────────
131
+ // Real Fireworks requires a bearer credential on every request (docs.fireworks.ai/api-reference/
132
+ // introduction: "All requests … must include an Authorization header with a valid Bearer token")
133
+ // and answers 401 Unauthorized when it is missing or invalid (the vendor's own error-code table).
134
+ // The twin can't validate against real keys, so it models the CHECKABLE failures: a missing
135
+ // credential, and a reserved sentinel for the invalid-key path. Any other non-empty key is
136
+ // accepted. Trusted in-process calls carry NEITHER `headers` nor `apiKey` and are NOT auth-gated;
137
+ // the official clients always send a key → they pass.
138
+ function checkAuth(req: FireworksRequest): FireworksResponseEnvelope | null {
139
+ const auth = req.headers?.['authorization'];
140
+ const bearer = typeof auth === 'string' && auth.toLowerCase().startsWith('bearer ') ? auth.slice(7).trim() : '';
141
+ const key = (req.apiKey ?? '').trim() || bearer;
142
+ if (!key || key === 'fw_invalid' || key === 'invalid') {
143
+ // Each plane answers 401 in its OWN envelope — a control-plane 401 carrying the OpenAI shape
144
+ // would mix the envelopes the header above calls a wire lie.
145
+ if (req.path.startsWith(`${FIREWORKS_INFERENCE_PREFIX}/messages`)) {
146
+ return anthropicError(401, 'authentication_error', 'Unauthorized: missing or invalid API key');
147
+ }
148
+ if (req.path.startsWith(`${FIREWORKS_ACCOUNTS_PREFIX}/`)) {
149
+ return gatewayError(401, 'Unauthorized: missing or invalid API key');
150
+ }
151
+ return authError('Unauthorized: missing or invalid API key');
152
+ }
153
+ return null;
154
+ }
155
+
156
+ function nowEpoch(occurredAt?: string): number {
157
+ return Math.floor((occurredAt ? Date.parse(occurredAt) : 0) / 1000);
158
+ }
159
+
160
+ // ── kernel helpers ──────────────────────────────────────────────────────────────────────
161
+ /**
162
+ * Rows of `type`, SCOPED TO ONE ACCOUNT. Every control-plane resource is stored with the
163
+ * `accountId` it was created (or pulled) under, and every read goes through this — a resource
164
+ * addressed under a different account's path is invisible here, which is what makes the
165
+ * cross-account GET/PATCH/DELETE answer the vendor's own NOT_FOUND. `accountId === undefined`
166
+ * means the caller reads a plane with NO account segment (the inference plane's stored
167
+ * responses) — those rows are not account-scoped and the filter is skipped for them.
168
+ */
169
+ /** A row is gone when the TWIN deleted it (`_deleted`, the twin's own marker) or when the
170
+ * CONNECTOR observed it vanish from the vendor (`deleted`, the kernel's tombstone field).
171
+ * Reading only the first served a resource the pull had correctly tombstoned (round-four review). */
172
+ export function isTombstoned(r: Record<string, unknown>): boolean {
173
+ return r._deleted === true || r.deleted === true;
174
+ }
175
+ function rows(type: string, accountId: string | undefined, root?: string): Array<Record<string, unknown>> {
176
+ return projectResources(SERVICE, root).filter((r) => r.type === type && (accountId === undefined || r._account === accountId));
177
+ }
178
+ /**
179
+ * Mint the next local id for `type`. Derived from the ID SET ALREADY IN STATE (a scan of the
180
+ * projection), never a row count — a count-mint silently clobbers a pulled vendor id sitting in a
181
+ * gap above the count (ADDING_A_TWIN.md §5). The `_twin_` infix namespaces LOCAL mints, so a
182
+ * pulled Fireworks id can never be matched by this regex and therefore can never be re-minted;
183
+ * the scan includes TOMBSTONED rows, so the counter RATCHETS across delete→recreate. The scan is
184
+ * PER ACCOUNT (rows() filters by `_account`) and matches the id SUFFIX after the account
185
+ * namespace, so two tenants each mint their own `dep_twin_1` without colliding.
186
+ */
187
+ function nextId(type: string, prefix: string, accountId: string | undefined, root?: string): string {
188
+ let max = 0;
189
+ for (const r of rows(type, accountId, root)) {
190
+ const local = accountId !== undefined && typeof r.id === 'string' && r.id.startsWith(`${accountId}/`)
191
+ ? r.id.slice(accountId.length + 1)
192
+ : String(r.id);
193
+ const m = new RegExp(`^${prefix}_twin_(\\d+)$`).exec(local);
194
+ if (m) max = Math.max(max, Number(m[1]));
195
+ }
196
+ return `${prefix}_twin_${max + 1}`;
197
+ }
198
+ /**
199
+ * Resolve one row by its BARE (wire) id inside one account. The kernel subject is the
200
+ * ACCOUNT-NAMESPACED id `${accountId}/${id}` — matching the vendor's own name grammar
201
+ * `accounts/{account}/{collection}/{id}` — so (type, subject) is unique per tenant and account
202
+ * B's create of the same id can never land on account A's row. Pulled rows carry the same
203
+ * namespaced id (the connector's mappers emit it), so created and pulled rows share one id space.
204
+ */
205
+ function getRow(type: string, id: string, accountId: string | undefined, root?: string): Record<string, unknown> | undefined {
206
+ const full = accountId === undefined ? id : `${accountId}/${id}`;
207
+ return rows(type, accountId, root).find((r) => r.id === full);
208
+ }
209
+ /** Strip the kernel's housekeeping fields and the twin's private underscore-prefixed fields. */
210
+ function strip(r: Record<string, unknown>): Record<string, unknown> {
211
+ const out: Record<string, unknown> = {};
212
+ for (const [k, v] of Object.entries(r)) {
213
+ if (k === 'type' || k === 'updatedAt' || k === 'deleted' || k.startsWith('_')) continue;
214
+ out[k] = v;
215
+ }
216
+ return out;
217
+ }
218
+
219
+ // ── request parsing ─────────────────────────────────────────────────────────────────────
220
+ function parseJson(body?: string): Record<string, unknown> {
221
+ if (!body || !body.trim()) return {};
222
+ try {
223
+ const v = JSON.parse(body);
224
+ return v && typeof v === 'object' ? (v as Record<string, unknown>) : {};
225
+ } catch {
226
+ return {};
227
+ }
228
+ }
229
+
230
+ /** A deterministic id suffix from the request (so ids are stable + assertable). */
231
+ function stableSuffix(seedText: string): string {
232
+ return fnv1a(seedText).toString(36);
233
+ }
234
+
235
+ // ── model catalog (the 404 the vendor answers for an unknown model id) ──────────────────
236
+ /**
237
+ * The model ids the INFERENCE plane accepts: the static catalog (fireworks-models.ts, sourced)
238
+ * PLUS every model row the control plane holds (a create/pull under /v1/accounts/.../models —
239
+ * an account-owned model is addressable for inference exactly as the vendor allows, including
240
+ * the fine-tunes). A pulled/created row of the same id as a static entry just re-states it.
241
+ * Real Fireworks answers 404 "Model id not found" for an id outside this set — the twin must
242
+ * refuse the same way, never answer 200 for any string.
243
+ */
244
+ function servedModelIds(root?: string): Set<string> {
245
+ const ids = new Set(FIREWORKS_MODEL_IDS);
246
+ for (const r of projectResources(SERVICE, root)) {
247
+ if (r.type !== 'model' || isTombstoned(r)) continue;
248
+ // A control-plane-created model is addressed ONLY by its FULL resource name on the
249
+ // inference plane (`accounts/{account}/models/{id}`) — the `name` the create handler
250
+ // minted. The bare suffix is NOT an alias: the vendor's own inference API takes full
251
+ // resource names for account-owned models (the static catalog's short aliases are
252
+ // documented serverless serving paths, a different thing), and accepting the bare suffix
253
+ // leaked one account's model into EVERY account's inference plane — a model created only
254
+ // under SECRETACCT answered 200 as the bare id while its qualified name 404'd under
255
+ // another account (inverted tenancy). The kernel subject stays the account-namespaced id
256
+ // (`{account}/{id}`); only the wire `name` joins the catalog.
257
+ if (typeof r.name === 'string') ids.add(r.name);
258
+ }
259
+ return ids;
260
+ }
261
+
262
+ /** The vendor's own refusal for an unknown model id (its error table: 404 covers "the model
263
+ * doesn't exist, the model is not deployed, or you don't have permission to access it"). */
264
+ function unknownModel(model: string): FireworksResponseEnvelope {
265
+ return notFound(`Model id not found: ${model}`);
266
+ }
267
+
268
+ // ── chat completions: validate the request the way Fireworks does ───────────────────────
269
+ const SERVICE_TIERS = ['auto', 'default', 'flex', 'priority'] as const;
270
+ const CONTEXT_BEHAVIORS = ['error', 'truncate'] as const;
271
+
272
+ /**
273
+ * The inference sampling parameters' documented ranges, from the pinned spec's own field
274
+ * descriptions (temperature "0 to 2", n "between 1 and 128", top_k "between 0 and 100",
275
+ * top_p "Required range: `0 <= x <= 1`" — CompletionRequest's own wording; ChatCompletionRequest
276
+ * carries the same nucleus-sampling semantics, frequency/presence_penalty "between -2 and 2")
277
+ * plus the OpenAI-compat types (role enum, stream boolean). The vendor's server (FastAPI)
278
+ * answers each violation with the 422 HTTPValidationError envelope and the pydantic error
279
+ * `type` — the twin answers the same, per parameter and constraint. `loc` matches where the
280
+ * field rides in the request body.
281
+ */
282
+ const SAMPLING_RULES: Array<{ field: string; kind: 'number'; min?: number; max?: number; integer?: boolean } | { field: string; kind: 'enum'; values: readonly string[] } | { field: string; kind: 'bool' }> = [
283
+ { field: 'temperature', kind: 'number', min: 0, max: 2 },
284
+ { field: 'top_p', kind: 'number', min: 0, max: 1 },
285
+ { field: 'n', kind: 'number', min: 1, max: 128, integer: true },
286
+ { field: 'top_k', kind: 'number', min: 0, max: 100, integer: true },
287
+ { field: 'frequency_penalty', kind: 'number', min: -2, max: 2 },
288
+ { field: 'presence_penalty', kind: 'number', min: -2, max: 2 },
289
+ ];
290
+
291
+ /** Validate the sampling/typing rules; returns the FastAPI 422 envelope on violation. Shared by
292
+ * the chat and legacy-completions doors so the two surfaces cannot drift. The message ROLE enum
293
+ * is checked per message by the chat door's own loop (the role rides inside `messages`). */
294
+ function validateSampling(params: Record<string, unknown>): FireworksResponseEnvelope | null {
295
+ for (const rule of SAMPLING_RULES) {
296
+ const v = params[rule.field];
297
+ if (v === undefined || v === null) continue;
298
+ if (rule.kind === 'number') {
299
+ // pydantic v2's kinds: a JSON string/bool where a number belongs is int_parsing (for an
300
+ // integer field) or float_parsing (for a float field) — int_type/float_type are
301
+ // python-typed-argument errors that never fire on a JSON body; a float where an integer
302
+ // belongs is int_from_float.
303
+ if (typeof v !== 'number' || Number.isNaN(v)) {
304
+ return validationError(['body', rule.field], `Input should be a valid ${rule.integer ? 'integer' : 'number'}`, rule.integer ? 'int_parsing' : 'float_parsing');
305
+ }
306
+ if (rule.integer && !Number.isInteger(v)) return validationError(['body', rule.field], 'Input should be a valid integer', 'int_from_float');
307
+ if (rule.min !== undefined && v < rule.min) return validationError(['body', rule.field], `Input should be greater than or equal to ${rule.min}`, 'greater_than_equal');
308
+ if (rule.max !== undefined && v > rule.max) return validationError(['body', rule.field], `Input should be less than or equal to ${rule.max}`, 'less_than_equal');
309
+ }
310
+ }
311
+ if (params.stream !== undefined && params.stream !== null && typeof params.stream !== 'boolean') {
312
+ return validationError(['body', 'stream'], 'Input should be a valid boolean', 'bool_type');
313
+ }
314
+ return null;
315
+ }
316
+
317
+ /** The OpenAI-compat message role enum (the spec's ChatCompletionRequestMessage.role). */
318
+ const MESSAGE_ROLES = ['system', 'user', 'assistant', 'tool', 'function', 'developer'] as const;
319
+
320
+ type ChatArgs = {
321
+ model: string;
322
+ messages: FireworksMessageParam[];
323
+ promptText: string;
324
+ tools?: unknown;
325
+ maxTokens?: number;
326
+ stop?: string[];
327
+ stream: boolean;
328
+ includeUsage: boolean;
329
+ serviceTier: string;
330
+ contextBehavior: 'error' | 'truncate';
331
+ responseFormat?: { type: string; schema?: unknown };
332
+ reasoningEffort?: string | number | boolean;
333
+ };
334
+
335
+ /**
336
+ * Fireworks' documented OpenAI differences are the fidelity surface here:
337
+ * • `max_tokens`/`max_completion_tokens` are MUTUALLY EXCLUSIVE (the spec's own note: "Alias for
338
+ * max_tokens. Cannot be specified together with max_tokens.") — OpenAI allows both;
339
+ * • `service_tier` is a closed enum whose values are all ACCEPTED but only 'priority' is
340
+ * honored — the others are treated as 'default', never an error;
341
+ * • `context_length_exceeded_behavior` defaults to 'truncate' (OpenAI's is 'error').
342
+ * The twin has no real context window, so the truncate path is modeled as the documented
343
+ * SEMANTIC (max_tokens lowered to fit) whenever a stub context is exceeded.
344
+ */
345
+ function validateChat(params: Record<string, unknown>): { args: ChatArgs } | { error: FireworksResponseEnvelope } {
346
+ if (params.model === undefined) return { error: validationError(['body', 'model'], 'Field required', 'missing') };
347
+ if (typeof params.model !== 'string' || !params.model) return { error: validationError(['body', 'model'], 'Input should be a valid string', 'string_type') };
348
+ if (!Array.isArray(params.messages)) return { error: validationError(['body', 'messages'], 'Field required', 'missing') };
349
+ if (params.messages.length === 0) return { error: invalidRequest("'messages' must not be empty") };
350
+ const messages = params.messages as FireworksMessageParam[];
351
+ for (const [i, m] of messages.entries()) {
352
+ if (!m || typeof m !== 'object' || typeof m.role !== 'string') {
353
+ return { error: validationError(['body', 'messages'], 'Input should be a valid dictionary', 'model_attributes_type') };
354
+ }
355
+ if (!MESSAGE_ROLES.includes(m.role as (typeof MESSAGE_ROLES)[number])) {
356
+ // FastAPI's loc carries the message INDEX (`body.messages.<i>.role`), not just the field.
357
+ return { error: validationError(['body', 'messages', i, 'role'], `Input should be one of ${MESSAGE_ROLES.map((r) => `'${r}'`).join(', ')}`, 'enum') };
358
+ }
359
+ }
360
+ // The sampling/typing table (temperature/n/top_k/penalties/stream) — the same rules the legacy
361
+ // door applies, so the two inference surfaces cannot drift.
362
+ const sampling = validateSampling(params);
363
+ if (sampling) return { error: sampling };
364
+ if (params.max_tokens !== undefined && params.max_tokens !== null && params.max_completion_tokens !== undefined && params.max_completion_tokens !== null) {
365
+ return { error: invalidRequest("'max_tokens' and 'max_completion_tokens' cannot both be specified — 'max_completion_tokens' is an alias for 'max_tokens'") };
366
+ }
367
+ const maxRaw = params.max_completion_tokens ?? params.max_tokens;
368
+ let maxTokens: number | undefined;
369
+ if (maxRaw !== undefined && maxRaw !== null) {
370
+ // pydantic parses the JSON value BEFORE any range check: a string is int_parsing, a float
371
+ // int_from_float, and only a parsed integer then hits the ≥1 bound.
372
+ maxTokens = Number(maxRaw);
373
+ if (typeof maxRaw !== 'number' || Number.isNaN(maxRaw)) return { error: validationError(['body', 'max_tokens'], 'Input should be a valid integer', 'int_parsing') };
374
+ if (!Number.isInteger(maxTokens)) return { error: validationError(['body', 'max_tokens'], 'Input should be a valid integer', 'int_from_float') };
375
+ if (maxTokens < 1) return { error: validationError(['body', 'max_tokens'], 'Input should be a valid integer greater than or equal to 1', 'greater_than_equal') };
376
+ }
377
+ let stop: string[] | undefined;
378
+ if (params.stop !== undefined && params.stop !== null) {
379
+ if (typeof params.stop === 'string') stop = [params.stop];
380
+ else if (Array.isArray(params.stop)) stop = params.stop as string[];
381
+ else return { error: validationError(['body', 'stop'], 'Input should be a valid string or array of strings', 'string_type') };
382
+ }
383
+ let serviceTier = 'default';
384
+ if (params.service_tier !== undefined && params.service_tier !== null) {
385
+ if (typeof params.service_tier !== 'string' || !SERVICE_TIERS.includes(params.service_tier as (typeof SERVICE_TIERS)[number])) {
386
+ return { error: validationError(['body', 'service_tier'], `Input should be one of ${SERVICE_TIERS.map((t) => `'${t}'`).join(', ')}`, 'enum') };
387
+ }
388
+ // The vendor's own semantics: "Only 'priority' is supported, while all other values will be
389
+ // treated as 'default' tier." Every enum value is ACCEPTED; 'auto'/'flex' do NOT error.
390
+ serviceTier = params.service_tier === 'priority' ? 'priority' : 'default';
391
+ }
392
+ let contextBehavior: 'error' | 'truncate' = 'truncate';
393
+ if (params.context_length_exceeded_behavior !== undefined && params.context_length_exceeded_behavior !== null) {
394
+ if (typeof params.context_length_exceeded_behavior !== 'string' || !CONTEXT_BEHAVIORS.includes(params.context_length_exceeded_behavior as 'error' | 'truncate')) {
395
+ return { error: validationError(['body', 'context_length_exceeded_behavior'], "Input should be one of 'error', 'truncate'", 'enum') };
396
+ }
397
+ contextBehavior = params.context_length_exceeded_behavior as 'error' | 'truncate';
398
+ }
399
+ let responseFormat: { type: string; schema?: unknown } | undefined;
400
+ const rf = params.response_format as { type?: unknown; json_schema?: unknown } | undefined;
401
+ if (rf && typeof rf === 'object') {
402
+ if (rf.type === 'json_object') responseFormat = { type: 'json_object' };
403
+ else if (rf.type === 'json_schema') responseFormat = { type: 'json_schema', schema: rf.json_schema };
404
+ else if (rf.type !== undefined && rf.type !== 'text') return { error: invalidRequest("'response_format.type' must be one of 'text', 'json_object', 'json_schema'") };
405
+ }
406
+ const stream = params.stream === true;
407
+ // Fireworks streams usage BY DEFAULT (the documented OpenAI difference); `include_usage: false`
408
+ // is the opt-OUT — the inverse of OpenAI's opt-IN.
409
+ const streamOptions = params.stream_options as { include_usage?: boolean } | undefined;
410
+ const includeUsage = streamOptions?.include_usage !== false;
411
+ return {
412
+ args: {
413
+ model: params.model,
414
+ messages,
415
+ promptText: messages.map((m) => contentToText(m.content)).join('\n'),
416
+ ...(params.tools !== undefined ? { tools: params.tools } : {}),
417
+ ...(maxTokens !== undefined ? { maxTokens } : {}),
418
+ ...(stop !== undefined ? { stop } : {}),
419
+ stream,
420
+ includeUsage,
421
+ serviceTier,
422
+ contextBehavior,
423
+ ...(responseFormat !== undefined ? { responseFormat } : {}),
424
+ ...(params.reasoning_effort !== undefined && params.reasoning_effort !== null ? { reasoningEffort: params.reasoning_effort as string | number | boolean } : {}),
425
+ },
426
+ };
427
+ }
428
+
429
+ /** Build ONE deterministic stub chat completion. */
430
+ function buildChatCompletion(args: ChatArgs, occurredAt?: string, decision?: ScenarioDecision): FireworksChatCompletion | FireworksResponseEnvelope {
431
+ const promptTokens = countPromptTokens(args.messages);
432
+ // Scenario scripting first — the DECISION was made (and any fault honored) by the request
433
+ // handler through the engine's serve(); only a content decision reaches here. A matching
434
+ // handler scripts the answer; a miss answers the labeled stub with a pointer naming the miss
435
+ // and the features seen.
436
+ let scripted: ScriptedResult | null = null;
437
+ let missTeach = '';
438
+ if (decision) {
439
+ if (decision.kind === 'handler') {
440
+ scripted = realizeFireworksRespond(decision.respond as FireworksScenarioRespond);
441
+ } else {
442
+ missTeach = `\n[twin-scenario miss — no handler matched. Author one in the world dir's handlers/fireworks.json (GET /twin explains; GET /twin/scenario lists handlers + misses). Features seen: ${JSON.stringify(decision.miss.features)}]`;
443
+ }
444
+ }
445
+ if (scripted) {
446
+ const message: FireworksChatMessage = { role: 'assistant', content: scripted.text ?? '', ...(scripted.toolCalls.length ? { tool_calls: scripted.toolCalls } : {}) };
447
+ if (scripted.reasoning !== null) message.reasoning_content = scripted.reasoning;
448
+ const completionTokens = estimateTokens(JSON.stringify(scripted.toolCalls.length ? scripted.toolCalls : scripted.text ?? ''));
449
+ return {
450
+ id: `chatcmpl-twin-${stableSuffix(JSON.stringify(args.messages) + args.model)}`,
451
+ object: 'chat.completion',
452
+ created: nowEpoch(occurredAt),
453
+ model: args.model,
454
+ choices: [{ index: 0, message, finish_reason: scripted.finishReason, logprobs: null }],
455
+ usage: { prompt_tokens: promptTokens, total_tokens: promptTokens + completionTokens, completion_tokens: completionTokens },
456
+ };
457
+ }
458
+ let text = args.responseFormat?.type === 'json_object' || args.responseFormat?.type === 'json_schema'
459
+ ? `${stubAssistantText(args.model, args.promptText)}\n{"twin_stub":true}`
460
+ : stubAssistantText(args.model, args.promptText);
461
+ if (missTeach) text += missTeach;
462
+ let finish = 'stop';
463
+ // Truncate at the EARLIEST-occurring stop sequence across the whole `stop` list.
464
+ let stopAt = -1;
465
+ for (const s of args.stop ?? []) {
466
+ if (!s) continue;
467
+ const i = text.indexOf(s);
468
+ if (i >= 0 && (stopAt < 0 || i < stopAt)) stopAt = i;
469
+ }
470
+ if (stopAt >= 0) { text = text.slice(0, stopAt); finish = 'stop'; }
471
+ // THE VENDOR DIFFERENCE: exceeding the (stub) context truncates max_tokens by default instead
472
+ // of erroring. The stub context is generous and fixed; the behavior is the fidelity surface.
473
+ const stubContextTokens = 8192;
474
+ if (promptTokens + (args.maxTokens ?? 0) > stubContextTokens) {
475
+ if (args.contextBehavior === 'error') {
476
+ return invalidRequest(`This model's maximum context length is ${stubContextTokens} tokens. However, you requested ${(args.maxTokens ?? 0) + promptTokens} tokens in the messages, which exceeds the model's context limit.`);
477
+ }
478
+ // truncate: max_tokens is lowered to fit.
479
+ if (args.maxTokens !== undefined) args.maxTokens = Math.max(1, stubContextTokens - promptTokens);
480
+ }
481
+ if (args.maxTokens !== undefined && estimateTokens(text) > args.maxTokens) {
482
+ text = text.slice(0, args.maxTokens * 4);
483
+ finish = 'length';
484
+ }
485
+ const message: FireworksChatMessage = { role: 'assistant', content: text };
486
+ // reasoning_effort (any truthy non-'none'/'false' value) surfaces Fireworks' separate
487
+ // `reasoning_content` field — the field the vendor's reasoning models answer with.
488
+ const effortOn = args.reasoningEffort !== undefined && args.reasoningEffort !== 'none' && args.reasoningEffort !== false;
489
+ if (effortOn) message.reasoning_content = stubReasoningText(args.model, args.promptText);
490
+ const completionTokens = estimateTokens(text) + estimateTokens(message.reasoning_content ?? '');
491
+ const usage = { prompt_tokens: promptTokens, total_tokens: promptTokens + completionTokens, completion_tokens: completionTokens };
492
+ const id = `chatcmpl-twin-${stableSuffix(JSON.stringify(args.messages) + args.model)}`;
493
+ return {
494
+ id,
495
+ object: 'chat.completion',
496
+ created: nowEpoch(occurredAt),
497
+ model: args.model,
498
+ choices: [{ index: 0, message, finish_reason: finish, logprobs: null }],
499
+ // usage is carried unconditionally (the vendor's streaming behavior implies it is always
500
+ // computed; the schema marks it nullable only for the echo/logprobs edge paths).
501
+ usage,
502
+ };
503
+ }
504
+
505
+ /** Split text into deterministic streaming chunks (≤ ~20 chars each), preserving order. */
506
+ function chunkText(text: string): string[] {
507
+ if (!text) return [];
508
+ const out: string[] = [];
509
+ for (let i = 0; i < text.length; i += 20) out.push(text.slice(i, i + 20));
510
+ return out;
511
+ }
512
+
513
+ /**
514
+ * Emit the vendor-faithful Fireworks streaming sequence into the injected sink (NO sockets, NO
515
+ * setTimeout). THE VENDOR DIFFERENCE (docs.fireworks.ai/tools-sdks/openai-compatibility): "For
516
+ * streaming responses, the `usage` field is returned in the very last chunk on the response (i.e.
517
+ * the one having `finish_reason` set)" — BY DEFAULT, no `stream_options.include_usage` needed
518
+ * (OpenAI makes it opt-in; Fireworks makes it opt-OUT via `include_usage: false`).
519
+ */
520
+ function streamChat(args: ChatArgs, sink: SseSink, occurredAt?: string, decision?: ScenarioDecision): FireworksChatCompletion | FireworksResponseEnvelope {
521
+ const built = buildChatCompletion(args, occurredAt, decision);
522
+ if ('status' in built) return built;
523
+ const full = built;
524
+ const base = { id: full.id, object: 'chat.completion.chunk' as const, created: full.created, model: full.model };
525
+ sink({ data: { ...base, choices: [{ index: 0, delta: { role: 'assistant', content: '' }, finish_reason: null }], usage: null } });
526
+ if (full.choices[0]!.message.reasoning_content) {
527
+ sink({ data: { ...base, choices: [{ index: 0, delta: { reasoning_content: full.choices[0]!.message.reasoning_content }, finish_reason: null }], usage: null } });
528
+ }
529
+ for (const piece of chunkText(full.choices[0]!.message.content ?? '')) {
530
+ sink({ data: { ...base, choices: [{ index: 0, delta: { content: piece }, finish_reason: null }], usage: null } });
531
+ }
532
+ // The final chunk carries finish_reason AND — by default — the usage.
533
+ sink({
534
+ data: {
535
+ ...base,
536
+ choices: [{ index: 0, delta: {}, finish_reason: full.choices[0]!.finish_reason }],
537
+ ...(args.includeUsage ? { usage: full.usage } : { usage: null }),
538
+ },
539
+ });
540
+ sink({ done: true });
541
+ return full;
542
+ }
543
+
544
+ // ── legacy completions (/v1/completions) ────────────────────────────────────────────────
545
+ /** Stream a legacy completion the OpenAI-compatible way: chunks carry the delta in
546
+ * `choices[].text` (`object: 'text_completion'` throughout — the legacy stream has no separate
547
+ * chunk object), ending with the finish chunk that — the Fireworks default — carries `usage`. */
548
+ function streamCompletion(full: FireworksCompletion, sink: SseSink, includeUsage: boolean): FireworksCompletion {
549
+ const base = { id: full.id, object: 'text_completion' as const, created: full.created, model: full.model };
550
+ for (const choice of full.choices) {
551
+ sink({ data: { ...base, choices: [{ index: choice.index, text: choice.text, finish_reason: null, logprobs: choice.logprobs }], usage: null } });
552
+ }
553
+ sink({
554
+ data: {
555
+ ...base,
556
+ choices: full.choices.map((c) => ({ index: c.index, text: '', finish_reason: c.finish_reason, logprobs: c.logprobs })),
557
+ ...(includeUsage ? { usage: full.usage } : { usage: null }),
558
+ },
559
+ });
560
+ sink({ done: true });
561
+ return full;
562
+ }
563
+
564
+ function handleCompletion(params: Record<string, unknown>, occurredAt?: string, sink?: SseSink, root?: string): FireworksResponseEnvelope {
565
+ if (params.model === undefined) return { status: 422, body: { detail: [{ loc: ['body', 'model'], msg: 'Field required', type: 'missing' }] } };
566
+ if (params.prompt === undefined) return { status: 422, body: { detail: [{ loc: ['body', 'prompt'], msg: 'Field required', type: 'missing' }] } };
567
+ if (typeof params.model !== 'string' || !params.model) return { status: 422, body: { detail: [{ loc: ['body', 'model'], msg: 'Input should be a valid string', type: 'string_type' }] } };
568
+ if (!servedModelIds(root).has(params.model)) return unknownModel(params.model);
569
+ // The same sampling/typing table the chat door applies.
570
+ const sampling = validateSampling(params);
571
+ if (sampling) return sampling;
572
+ const prompts = Array.isArray(params.prompt) ? (params.prompt as unknown[]).map(String) : [String(params.prompt)];
573
+ const maxRaw = params.max_completion_tokens ?? params.max_tokens;
574
+ if (params.max_tokens !== undefined && params.max_tokens !== null && params.max_completion_tokens !== undefined && params.max_completion_tokens !== null) {
575
+ return invalidRequest("'max_tokens' and 'max_completion_tokens' cannot both be specified — 'max_completion_tokens' is an alias for 'max_tokens'");
576
+ }
577
+ let maxTokens: number | undefined;
578
+ if (maxRaw !== undefined && maxRaw !== null) {
579
+ // The same pydantic parse order the chat door applies (string → int_parsing, float →
580
+ // int_from_float, then the ≥1 bound).
581
+ maxTokens = Number(maxRaw);
582
+ if (typeof maxRaw !== 'number' || Number.isNaN(maxRaw)) return { status: 422, body: { detail: [{ loc: ['body', 'max_tokens'], msg: 'Input should be a valid integer', type: 'int_parsing' }] } };
583
+ if (!Number.isInteger(maxTokens)) return { status: 422, body: { detail: [{ loc: ['body', 'max_tokens'], msg: 'Input should be a valid integer', type: 'int_from_float' }] } };
584
+ if (maxTokens < 1) return { status: 422, body: { detail: [{ loc: ['body', 'max_tokens'], msg: 'Input should be a valid integer greater than or equal to 1', type: 'greater_than_equal' }] } };
585
+ }
586
+ const model = String(params.model);
587
+ const promptTokens = prompts.reduce((sum, p) => sum + estimateTokens(p), 0);
588
+ const choices = prompts.map((p, i) => {
589
+ let text = stubAssistantText(model, p);
590
+ let finish = 'stop';
591
+ if (maxTokens !== undefined && estimateTokens(text) > maxTokens) {
592
+ text = text.slice(0, maxTokens * 4);
593
+ finish = 'length';
594
+ }
595
+ return { index: i, text, finish_reason: finish, logprobs: null };
596
+ });
597
+ const completionTokens = choices.reduce((sum, c) => sum + estimateTokens(c.text), 0);
598
+ const body: FireworksCompletion = {
599
+ id: `cmpl-twin-${stableSuffix(JSON.stringify(prompts) + model)}`,
600
+ object: 'text_completion',
601
+ created: nowEpoch(occurredAt),
602
+ model,
603
+ choices,
604
+ // The spec marks `usage` REQUIRED on Completion (nullable only on chat).
605
+ usage: { prompt_tokens: promptTokens, total_tokens: promptTokens + completionTokens, completion_tokens: completionTokens },
606
+ };
607
+ // The server hands a sink only for `stream:true`; ignoring it would answer a 200 event stream
608
+ // with zero frames — a fake success on the wire.
609
+ const streamOptions = params.stream_options as { include_usage?: unknown } | undefined;
610
+ if (sink) return { status: 200, body: streamCompletion(body, sink, streamOptions?.include_usage !== false) };
611
+ return { status: 200, body };
612
+ }
613
+
614
+ // ── Responses API (/v1/responses — stateful CRUD over the kernel log) ───────────────────
615
+ /**
616
+ * Fireworks' Responses API stores responses server-side (`store` defaults true; `store:false`
617
+ * answers with a NULL id per the vendor's own schema: "Will be None if store=False"). The twin
618
+ * mirrors that: stored responses live in the kernel action log and are retrievable/deletable;
619
+ * a store=false response is answered id-less and NOT stored.
620
+ */
621
+ function responseView(r: Record<string, unknown>): Record<string, unknown> {
622
+ return { id: r.id, ...strip(r) };
623
+ }
624
+
625
+ async function createResponse(params: Record<string, unknown>, req: FireworksRequest): Promise<FireworksResponseEnvelope> {
626
+ if (params.model === undefined) return { status: 422, body: { detail: [{ loc: ['body', 'model'], msg: 'Field required', type: 'missing' }] } };
627
+ if (params.input === undefined) return { status: 422, body: { detail: [{ loc: ['body', 'input'], msg: 'Field required', type: 'missing' }] } };
628
+ if (typeof params.model !== 'string' || !params.model) return { status: 422, body: { detail: [{ loc: ['body', 'model'], msg: 'Input should be a valid string', type: 'string_type' }] } };
629
+ const store = params.store !== false;
630
+ const inputText = typeof params.input === 'string' ? params.input : JSON.stringify(params.input);
631
+ const promptTokens = estimateTokens(inputText);
632
+ const text = stubAssistantText(String(params.model), inputText);
633
+ const completionTokens = estimateTokens(text);
634
+ const outputItem = {
635
+ type: 'message' as const,
636
+ id: `msg_twin_${stableSuffix(text)}`,
637
+ role: 'assistant' as const,
638
+ status: 'completed' as const,
639
+ content: [{ type: 'output_text' as const, text }],
640
+ };
641
+ const base: Record<string, unknown> = {
642
+ object: 'response',
643
+ created_at: nowEpoch(req.occurredAt),
644
+ status: 'completed',
645
+ model: params.model,
646
+ output: [outputItem],
647
+ usage: { prompt_tokens: promptTokens, total_tokens: promptTokens + completionTokens, completion_tokens: completionTokens },
648
+ ...(params.instructions !== undefined ? { instructions: params.instructions } : {}),
649
+ ...(params.metadata !== undefined ? { metadata: params.metadata } : {}),
650
+ ...(params.previous_response_id !== undefined ? { previous_response_id: params.previous_response_id } : {}),
651
+ ...(params.temperature !== undefined ? { temperature: params.temperature } : {}),
652
+ ...(params.max_output_tokens !== undefined ? { max_output_tokens: params.max_output_tokens } : {}),
653
+ store,
654
+ };
655
+ if (!store) {
656
+ // The vendor's own contract: id is null when store=false, and nothing is retrievable later.
657
+ return { status: 200, body: { ...base, id: null } };
658
+ }
659
+ const id = `resp_twin_${stableSuffix(inputText + String(params.model))}`;
660
+ await applyTwinWrite(SERVICE, {
661
+ operation: 'response.create',
662
+ subjectType: 'response',
663
+ subjectId: id,
664
+ fields: base,
665
+ ...(req.occurredAt ? { occurredAt: req.occurredAt } : {}),
666
+ actor: { kind: 'agent' },
667
+ }, req.root);
668
+ return { status: 200, body: responseView(getRow('response', id, undefined, req.root) ?? { ...base, id }) };
669
+ }
670
+
671
+ function handleListResponses(req: FireworksRequest): FireworksResponseEnvelope {
672
+ const url = new URL(req.path, 'http://twin');
673
+ const limit = Number(url.searchParams.get('limit') ?? 20);
674
+ const data = rows('response', undefined, req.root).filter((r) => !isTombstoned(r)).map(responseView);
675
+ const page = data.slice(0, Number.isFinite(limit) && limit > 0 ? limit : 20);
676
+ return { status: 200, body: { object: 'list', data: page, has_more: data.length > page.length, first_id: page[0]?.id ?? null, last_id: page[page.length - 1]?.id ?? null } };
677
+ }
678
+
679
+ function handleGetResponse(id: string, req: FireworksRequest): FireworksResponseEnvelope {
680
+ const r = getRow('response', id, undefined, req.root);
681
+ if (!r || isTombstoned(r)) return notFound(`No response found with id '${id}'.`);
682
+ return { status: 200, body: responseView(r) };
683
+ }
684
+
685
+ async function handleDeleteResponse(id: string, req: FireworksRequest): Promise<FireworksResponseEnvelope> {
686
+ const r = getRow('response', id, undefined, req.root);
687
+ if (!r || isTombstoned(r)) return notFound(`No response found with id '${id}'.`);
688
+ await applyTwinWrite(SERVICE, {
689
+ operation: 'response.delete', subjectType: 'response', subjectId: id, fields: { _deleted: true },
690
+ ...(req.occurredAt ? { occurredAt: req.occurredAt } : {}), actor: { kind: 'agent' },
691
+ }, req.root);
692
+ return { status: 200, body: { id, object: 'response', deleted: true } };
693
+ }
694
+
695
+ // ── Anthropic-compatible /v1/messages ───────────────────────────────────────────────────
696
+ /** The Anthropic error envelope — a DIFFERENT shape from the OpenAI-compat plane's. */
697
+ function anthropicError(status: number, type: FireworksAnthropicErrorType, message: string): FireworksResponseEnvelope {
698
+ return { status, body: { type: 'error', error: { type, message }, request_id: null } };
699
+ }
700
+
701
+ function validateAnthropicMessages(params: Record<string, unknown>): { ok: true } | { error: FireworksResponseEnvelope } {
702
+ if (params.model === undefined) return { error: anthropicError(400, 'invalid_request_error', 'model: Field required') };
703
+ if (!Array.isArray(params.messages) || params.messages.length === 0) {
704
+ return { error: anthropicError(400, 'invalid_request_error', 'messages: Field required') };
705
+ }
706
+ for (const m of params.messages as Array<Record<string, unknown>>) {
707
+ if (!m || typeof m !== 'object' || (m.role !== 'user' && m.role !== 'assistant')) {
708
+ return { error: anthropicError(400, 'invalid_request_error', "messages: each message must have role 'user' or 'assistant'") };
709
+ }
710
+ }
711
+ return { ok: true };
712
+ }
713
+
714
+ function buildAnthropicMessage(params: Record<string, unknown>, occurredAt?: string): FireworksResponseEnvelope {
715
+ const bad = validateAnthropicMessages(params);
716
+ if ('error' in bad) return bad.error;
717
+ const messages = params.messages as Array<{ role: string; content: unknown }>;
718
+ const promptText = messages.map((m) => contentToText(m.content)).join('\n');
719
+ const systemText = typeof params.system === 'string' ? params.system : Array.isArray(params.system) ? params.system.map((b) => contentToText((b as { text?: unknown }).text)).join('\n') : '';
720
+ const inputTokens = estimateTokens(promptText + systemText);
721
+ // max_tokens is OPTIONAL on Fireworks (required on Anthropic — the documented difference).
722
+ const maxTokens = typeof params.max_tokens === 'number' ? params.max_tokens : undefined;
723
+ let text = stubAssistantText(String(params.model), promptText.trim() || systemText);
724
+ let stopReason: FireworksAnthropicMessage['stop_reason'] = 'end_turn';
725
+ if (maxTokens !== undefined && estimateTokens(text) > maxTokens) {
726
+ text = text.slice(0, maxTokens * 4);
727
+ stopReason = 'max_tokens';
728
+ }
729
+ const content: FireworksAnthropicContentBlock[] = [{ type: 'text', text, citations: null }];
730
+ const outputTokens = estimateTokens(text);
731
+ const body: FireworksAnthropicMessage = {
732
+ id: `msg_twin_${stableSuffix(promptText + String(params.model))}`,
733
+ type: 'message',
734
+ role: 'assistant',
735
+ content,
736
+ model: String(params.model),
737
+ stop_reason: stopReason,
738
+ stop_sequence: null,
739
+ // Usage is included in BOTH streaming and non-streaming responses (the documented difference
740
+ // from Anthropic, where streaming omits it until the final delta).
741
+ usage: { input_tokens: inputTokens, output_tokens: outputTokens },
742
+ };
743
+ return { status: 200, body };
744
+ }
745
+
746
+ /** The Anthropic SSE sequence: message_start → content_block_start → content_block_delta* →
747
+ * content_block_stop → message_delta (carrying the ACTUAL usage — the one message_delta per
748
+ * stream) → message_stop. */
749
+ function streamAnthropicMessage(params: Record<string, unknown>, sink: SseSink, occurredAt?: string): FireworksResponseEnvelope {
750
+ const built = buildAnthropicMessage(params, occurredAt);
751
+ if (built.status !== 200) return built;
752
+ const full = built.body as FireworksAnthropicMessage;
753
+ const content = full.content;
754
+ const usage = full.usage ?? { input_tokens: 0, output_tokens: 0 };
755
+ sink({ data: { type: 'message_start', message: { ...full, content: [], stop_reason: null, usage: { input_tokens: usage.input_tokens, output_tokens: 0 } } } });
756
+ sink({ data: { type: 'content_block_start', index: 0, content_block: { type: 'text', text: '' } } });
757
+ for (const piece of chunkText(content[0]!.type === 'text' ? content[0]!.text : '')) {
758
+ sink({ data: { type: 'content_block_delta', index: 0, delta: { type: 'text_delta', text: piece } } });
759
+ }
760
+ sink({ data: { type: 'content_block_stop', index: 0 } });
761
+ // ONE message_delta, carrying the ACTUAL token counts (the vendor's own note: the message_start
762
+ // usage is always 0 and should be ignored for metering).
763
+ sink({ data: { type: 'message_delta', delta: { stop_reason: full.stop_reason, stop_sequence: null }, usage: { input_tokens: usage.input_tokens, output_tokens: usage.output_tokens } } });
764
+ sink({ data: { type: 'message_stop' } });
765
+ sink({ done: true });
766
+ return built;
767
+ }
768
+
769
+ // ── embeddings + rerank (deterministic pseudo-vectors / scores) ─────────────────────────
770
+ /** The largest vector the twin will ever allocate from a client number. The vendor's resizable
771
+ * embedding models truncate DOWN to the requested `dimensions` (never produce longer vectors
772
+ * than the model's native output); the catalog's largest native dimensionality across the
773
+ * served embedders is the Qwen3 embedding family's 4096. Every array sized from a REQUEST
774
+ * value in this pack is bounded by this constant or by an input length — audited. */
775
+ const MAX_EMBEDDING_DIMENSIONS = 4096;
776
+
777
+ function handleEmbeddings(params: Record<string, unknown>, root?: string): FireworksResponseEnvelope {
778
+ if (params.model === undefined) return { status: 422, body: { detail: [{ loc: ['body', 'model'], msg: 'Field required', type: 'missing' }] } };
779
+ if (params.input === undefined) return { status: 422, body: { detail: [{ loc: ['body', 'input'], msg: 'Field required', type: 'missing' }] } };
780
+ if (typeof params.model !== 'string' || !servedModelIds(root).has(params.model)) return unknownModel(String(params.model));
781
+ const inputs = Array.isArray(params.input) ? (params.input as unknown[]).map((v) => (typeof v === 'string' ? v : JSON.stringify(v))) : [String(params.input)];
782
+ if (inputs.some((s) => s.length === 0)) return invalidRequest("'input' must not be an empty string");
783
+ // `dimensions` is validated BEFORE any allocation: the spec types it `anyOf [integer, null]`
784
+ // (EmbeddingRequest), so pydantic answers a non-integer with int_parsing; the range below is
785
+ // the vendor's own constraint — dimensions must be ≥ 1 (the API reference schema marks
786
+ // minimum 1) and ≤ the model's native dimensionality (the vendor's resizable models return
787
+ // SHORTER vectors — matryoshka truncation — never longer; the catalog's largest is the Qwen3
788
+ // embedding table's 4096). An unvalidated client number flowed straight into the vector
789
+ // allocation: 1e9 tried to allocate a 12 GB array (one-request DoS), -1/2.5 escaped as a 500,
790
+ // 0 answered a fake 200 with an empty vector, 'x' was silently ignored.
791
+ const dimsRaw = params.dimensions;
792
+ if (dimsRaw !== undefined && dimsRaw !== null) {
793
+ if (typeof dimsRaw !== 'number' || Number.isNaN(dimsRaw)) return validationError(['body', 'dimensions'], 'Input should be a valid integer', 'int_parsing');
794
+ if (!Number.isInteger(dimsRaw)) return validationError(['body', 'dimensions'], 'Input should be a valid integer', 'int_from_float');
795
+ if (dimsRaw < 1) return validationError(['body', 'dimensions'], 'Input should be greater than or equal to 1', 'greater_than_equal');
796
+ if (dimsRaw > MAX_EMBEDDING_DIMENSIONS) return validationError(['body', 'dimensions'], `Input should be less than or equal to ${MAX_EMBEDDING_DIMENSIONS}`, 'less_than_equal');
797
+ }
798
+ const dims = typeof dimsRaw === 'number' ? dimsRaw : 768;
799
+ const encoding = params.encoding_format === undefined ? 'float' : params.encoding_format;
800
+ if (encoding !== 'float' && encoding !== 'base64') return invalidRequest("'encoding_format' must be one of 'float', 'base64'");
801
+ let promptTokens = 0;
802
+ const data = inputs.map((text, index) => {
803
+ promptTokens += estimateTokens(text);
804
+ const vec = pseudoEmbedding(text, dims);
805
+ return {
806
+ object: 'embedding' as const,
807
+ index,
808
+ embedding: encoding === 'base64' ? Buffer.from(new Float32Array(vec).buffer).toString('base64') : vec,
809
+ };
810
+ });
811
+ return {
812
+ status: 200,
813
+ body: {
814
+ object: 'list',
815
+ data,
816
+ model: params.model,
817
+ usage: { prompt_tokens: promptTokens, total_tokens: promptTokens },
818
+ },
819
+ };
820
+ }
821
+
822
+ function handleRerank(params: Record<string, unknown>, root?: string): FireworksResponseEnvelope {
823
+ if (params.query === undefined) return { status: 422, body: { detail: [{ loc: ['body', 'query'], msg: 'Field required', type: 'missing' }] } };
824
+ if (!Array.isArray(params.documents) || params.documents.length === 0) return { status: 422, body: { detail: [{ loc: ['body', 'documents'], msg: 'Field required', type: 'missing' }] } };
825
+ // The vendor's rerank takes a model too (the Qwen3 Reranker family); an unknown id is refused
826
+ // the same way chat/embeddings refuse theirs. A MISSING model is accepted (the twin's own
827
+ // query/document scorer needs no id; the vendor's request schema marks model optional there).
828
+ if (params.model !== undefined && (typeof params.model !== 'string' || !servedModelIds(root).has(params.model))) return unknownModel(String(params.model));
829
+ const query = String(params.query);
830
+ const documents = (params.documents as unknown[]).map(String);
831
+ const returnDocuments = params.return_documents !== false;
832
+ const scored = documents.map((doc, index) => ({ index, score: stubRelevanceScore(query, doc), doc }));
833
+ // Ordered by relevance score (highest first) — the vendor's own contract for the data array.
834
+ scored.sort((a, b) => b.score - a.score);
835
+ // `top_n` is the spec's integer (RerankRequestBody); a non-integer is refused before it can
836
+ // slice, and a value below 1 would answer an EMPTY result list for a valid request — the
837
+ // vendor's own contract keeps at least one result.
838
+ const topRaw = params.top_n;
839
+ if (topRaw !== undefined && topRaw !== null) {
840
+ if (typeof topRaw !== 'number' || Number.isNaN(topRaw)) return validationError(['body', 'top_n'], 'Input should be a valid integer', 'int_parsing');
841
+ if (!Number.isInteger(topRaw)) return validationError(['body', 'top_n'], 'Input should be a valid integer', 'int_from_float');
842
+ if (topRaw < 1) return validationError(['body', 'top_n'], 'Input should be greater than or equal to 1', 'greater_than_equal');
843
+ }
844
+ const topN = typeof topRaw === 'number' ? topRaw : documents.length;
845
+ const data = scored.slice(0, Math.max(0, topN)).map((s) => ({
846
+ index: s.index,
847
+ relevance_score: s.score,
848
+ ...(returnDocuments ? { document: s.doc } : {}),
849
+ }));
850
+ const promptTokens = estimateTokens(query) + documents.reduce((sum, d) => sum + estimateTokens(d), 0);
851
+ return {
852
+ status: 200,
853
+ body: {
854
+ object: 'list',
855
+ model: params.model ?? null,
856
+ data,
857
+ usage: { prompt_tokens: promptTokens, total_tokens: promptTokens },
858
+ },
859
+ };
860
+ }
861
+
862
+ // ── CONTROL PLANE (Gateway REST API) ───────────────────────────────────────────────────
863
+ // google.rpc-style resources: `name` fields follow accounts/<account>/…, `state`/`status` are
864
+ // vendor enums, list envelopes carry nextPageToken/totalSize, and CREATE ids arrive as QUERY
865
+ // params (deployments/datasets/users) or in the body (datasets carry datasetId in the body too).
866
+ function resourceView(r: Record<string, unknown>): Record<string, unknown> {
867
+ // The vendor's control-plane schemas (gatewayDeployment, gatewayDataset, …) carry NO `id`
868
+ // field — identity is the hierarchical `name`. The bare id stays a twin-internal alias and
869
+ // must never appear on the wire (defect: strip() was letting it through on every body).
870
+ const { id: _drop, ...rest } = r;
871
+ return strip(rest);
872
+ }
873
+
874
+ function paginate(items: Array<Record<string, unknown>>, url: URL): { items: Array<Record<string, unknown>>; nextPageToken: string | null } {
875
+ const pageSize = Number(url.searchParams.get('pageSize') ?? 0);
876
+ const pageToken = url.searchParams.get('pageToken');
877
+ let start = 0;
878
+ if (pageToken) {
879
+ const n = Number(pageToken);
880
+ start = Number.isFinite(n) && n > 0 ? n : 0;
881
+ }
882
+ const size = Number.isFinite(pageSize) && pageSize > 0 ? pageSize : items.length;
883
+ const page = items.slice(start, start + size);
884
+ const next = start + size < items.length ? String(start + size) : null;
885
+ return { items: page, nextPageToken: next };
886
+ }
887
+
888
+ /** The gateway status embedded on resources: OK once READY, else the resource's own state. */
889
+ function okStatus(): Record<string, unknown> {
890
+ return { code: 'OK', message: '' };
891
+ }
892
+
893
+ async function createControlResource(type: string, prefix: string, accountId: string, params: Record<string, unknown>, req: FireworksRequest, url: URL, buildFields: (params: Record<string, unknown>) => Record<string, unknown>): Promise<FireworksResponseEnvelope> {
894
+ // The vendor passes create ids where its spec puts them: QUERY params for deployments
895
+ // (deploymentId), users (userId), batchInferenceJobId and supervisedFineTuningJobId; BODY
896
+ // fields for datasets ({dataset, datasetId}) and models ({model, modelId}); the resource's own
897
+ // `name` for secrets. Absent → the vendor mints one; so does the twin.
898
+ const queryId = url.searchParams.get(`${type}Id`) ?? url.searchParams.get(`${type}_id`);
899
+ const wrapped = (params[type] && typeof params[type] === 'object' ? params[type] : params) as Record<string, unknown>;
900
+ const bodyId = typeof wrapped[`${type}Id`] === 'string' ? (wrapped[`${type}Id`] as string)
901
+ : typeof params[`${type}Id`] === 'string' ? (params[`${type}Id`] as string)
902
+ // A secret's id is the last segment of the `name` the client supplies in its own body.
903
+ : type === 'secret' && typeof params.name === 'string' ? (params.name.split('/').pop() as string)
904
+ : undefined;
905
+ const id = queryId || bodyId || nextId(type, prefix, accountId, req.root);
906
+ // The vendor's own spec marks the id REQUIRED on datasets (CreateDatasetRequest.required:
907
+ // dataset + datasetId) and models (GatewayGatewayCreateModelBody.required: modelId) — a create
908
+ // without one is a 400 at the vendor, never a mint. The dataset body itself is required too
909
+ // (the same required list); a bare {datasetId} is not a create. The other collections' id
910
+ // params are optional (the vendor mints), so the twin keeps minting there.
911
+ if (type === 'dataset' || type === 'model') {
912
+ if (!queryId && !bodyId) return gatewayError(400, `${type}Id is required`);
913
+ if (type === 'dataset' && !(params.dataset && typeof params.dataset === 'object')) {
914
+ return gatewayError(400, 'dataset is required');
915
+ }
916
+ }
917
+ // TENANCY: the kernel subject is the ACCOUNT-NAMESPACED id (`{account}/{id}` — the vendor's own
918
+ // name grammar), so the kernel's (type, subject) key is unique per tenant. Writing the bare id
919
+ // let a create under account B with an id account A held DESTROY A's row (the overlay folds by
920
+ // the bare id globally); the duplicate check below is therefore per-account by construction.
921
+ const subject = `${accountId}/${id}`;
922
+ if (getRow(type, id, accountId, req.root) && !isTombstoned(getRow(type, id, accountId, req.root)!)) {
923
+ return gatewayError(409, `Resource already exists: ${id}`);
924
+ }
925
+ const fields = buildFields(params);
926
+ // The account the create rode under is part of the resource's identity: every later read is
927
+ // scoped to it, so a row created under acct-A is invisible under acct-B's paths (the vendor's
928
+ // own tenancy). The `_` prefix keeps it off the wire (strip() drops it).
929
+ fields._account = accountId;
930
+ // The id is resolved HERE (query param, body field, or the twin's mint) — so the row's `name`
931
+ // is built from it. A caller-supplied buildFields cannot know the minted id; its PLACEHOLDER
932
+ // stand-in must never survive to the wire (the vendor names every row with the real id).
933
+ if (typeof fields.name === 'string') fields.name = fields.name.replace('PLACEHOLDER', id);
934
+ await applyTwinWrite(SERVICE, {
935
+ operation: `${type}.create`,
936
+ subjectType: type,
937
+ subjectId: subject,
938
+ fields,
939
+ ...(req.occurredAt ? { occurredAt: req.occurredAt } : {}),
940
+ actor: { kind: 'agent' },
941
+ }, req.root);
942
+ return { status: 200, body: resourceView(getRow(type, id, accountId, req.root) ?? { id: subject, ...fields }) };
943
+ }
944
+
945
+ async function deleteControlResource(type: string, id: string, accountId: string, req: FireworksRequest): Promise<FireworksResponseEnvelope> {
946
+ const r = getRow(type, id, accountId, req.root);
947
+ if (!r || isTombstoned(r)) return gatewayError(404, `Not found: ${type}/${id}`);
948
+ await applyTwinWrite(SERVICE, {
949
+ operation: `${type}.delete`, subjectType: type, subjectId: `${accountId}/${id}`, fields: { _deleted: true },
950
+ ...(req.occurredAt ? { occurredAt: req.occurredAt } : {}), actor: { kind: 'agent' },
951
+ }, req.root);
952
+ // The vendor's delete operations answer `{}` (an empty object per its own spec).
953
+ return { status: 200, body: {} };
954
+ }
955
+
956
+ const DEPLOYMENT_STATES = ['STATE_UNSPECIFIED', 'CREATING', 'READY', 'DELETING', 'FAILED', 'UPDATING', 'DELETED'] as const;
957
+ const JOB_STATES = ['JOB_STATE_UNSPECIFIED', 'JOB_STATE_CREATING', 'JOB_STATE_RUNNING', 'JOB_STATE_COMPLETED', 'JOB_STATE_FAILED', 'JOB_STATE_CANCELLED', 'JOB_STATE_DELETING', 'JOB_STATE_WRITING_RESULTS', 'JOB_STATE_VALIDATING', 'JOB_STATE_DELETING_CLEANING_UP', 'JOB_STATE_PENDING', 'JOB_STATE_EXPIRED', 'JOB_STATE_RE_QUEUEING', 'JOB_STATE_CREATING_INPUT_DATASET', 'JOB_STATE_IDLE', 'JOB_STATE_CANCELLING', 'JOB_STATE_EARLY_STOPPED', 'JOB_STATE_PAUSED', 'JOB_STATE_DELETED', 'JOB_STATE_ARCHIVED'] as const;
958
+ const USER_STATES = ['STATE_UNSPECIFIED', 'CREATING', 'READY', 'UPDATING', 'DELETING'] as const;
959
+ const USER_ROLES = ['admin', 'user', 'contributor', 'inference-user', 'custom'] as const;
960
+
961
+ // ── public entry: cross-cutting protocol (auth) then route ──────────────────────────────
962
+ export async function handleFireworksTwinRequest(req: FireworksRequest): Promise<FireworksResponseEnvelope> {
963
+ const method = req.method.toUpperCase();
964
+ if (req.headers !== undefined || req.apiKey !== undefined) {
965
+ const authErr = checkAuth(req);
966
+ if (authErr) return authErr;
967
+ }
968
+ return routeFireworks(req, method);
969
+ }
970
+
971
+ // ── router ──────────────────────────────────────────────────────────────────────────────
972
+ async function routeFireworks(req: FireworksRequest, method: string): Promise<FireworksResponseEnvelope> {
973
+ const path = (req.path.split('?')[0] ?? '/').replace(/\/+$/, '') || '/';
974
+ const query = req.path.includes('?') ? req.path.slice(req.path.indexOf('?')) : '';
975
+ const params = parseJson(req.body);
976
+ const dec = (s: string) => decodeURIComponent(s);
977
+
978
+ // D3: a read-only twin rejects any mutation with a vendor-shaped error — each plane's OWN
979
+ // envelope (the control plane's google.rpc shape, the inference plane's OpenAI shape).
980
+ if (req.readOnly && method !== 'GET') {
981
+ const message = 'twin is read-only; omit readOnly to accept writes';
982
+ if (path.startsWith(`${FIREWORKS_ACCOUNTS_PREFIX}/`)) return gatewayError(405, message);
983
+ return { status: 405, body: errBody(message, { code: 405 }) };
984
+ }
985
+
986
+ const url = new URL(req.path, 'http://twin');
987
+
988
+ // ── INFERENCE PLANE (/inference/v1/…) ──────────────────────────────────────────────────
989
+ if (path === FIREWORKS_INFERENCE_PREFIX || path.startsWith(`${FIREWORKS_INFERENCE_PREFIX}/`)) {
990
+ const seg = path.slice(FIREWORKS_INFERENCE_PREFIX.length).replace(/^\/+/, '').split('/').filter(Boolean);
991
+ // The Anthropic-compat surface has its OWN envelope family (AnthropicErrorResponse), so it
992
+ // routes separately from the OpenAI-compat operations.
993
+ if (seg[0] === 'messages' && seg.length === 1 && method === 'POST') {
994
+ if (params.stream === true) {
995
+ if (!req.sseSink) return anthropicError(400, 'invalid_request_error', 'streaming requires an SSE-capable connection');
996
+ return streamAnthropicMessage(params, req.sseSink, req.occurredAt);
997
+ }
998
+ return buildAnthropicMessage(params, req.occurredAt);
999
+ }
1000
+ if (seg[0] === 'chat' && seg[1] === 'completions' && seg.length === 2 && method === 'POST') {
1001
+ const validated = validateChat(params);
1002
+ if ('error' in validated) return validated.error;
1003
+ const args = validated.args;
1004
+ // The vendor refuses an unknown model id BEFORE any generation (404 "Model id not found");
1005
+ // an account-owned model (created/pulled through the control plane) is addressable too.
1006
+ if (!servedModelIds(req.root).has(args.model)) return unknownModel(args.model);
1007
+ // R15 — the scenario engine decides AND honors a fault here, before any completion exists:
1008
+ // a `status` fault is this vendor's own refusal envelope (rate limit, server error), a
1009
+ // `slow` has already held the answer, a `drop` never returns. The realizers below only see
1010
+ // a content decision.
1011
+ let decision: ScenarioDecision | undefined;
1012
+ if (req.scenarioEngine) {
1013
+ const served = await req.scenarioEngine.serve({ model: args.model, messages: args.messages, tools: args.tools, serviceTier: args.serviceTier });
1014
+ if (served.kind === 'fault') return { status: served.result.status, body: served.result.body, headers: served.result.headers };
1015
+ decision = served;
1016
+ }
1017
+ const result = args.stream && req.sseSink ? streamChat(args, req.sseSink, req.occurredAt, decision) : buildChatCompletion(args, req.occurredAt, decision);
1018
+ if ('status' in result) return result;
1019
+ return { status: 200, body: result };
1020
+ }
1021
+ if (seg[0] === 'completions' && seg.length === 1 && method === 'POST') return handleCompletion(params, req.occurredAt, req.sseSink, req.root);
1022
+ if (seg[0] === 'responses' && seg.length === 1 && method === 'POST') return createResponse(params, req);
1023
+ if (seg[0] === 'responses' && seg.length === 1 && method === 'GET') return handleListResponses(req);
1024
+ if (seg[0] === 'responses' && seg.length === 2 && method === 'GET') return handleGetResponse(dec(seg[1]!), req);
1025
+ if (seg[0] === 'responses' && seg.length === 2 && method === 'DELETE') return handleDeleteResponse(dec(seg[1]!), req);
1026
+ if (seg[0] === 'embeddings' && seg.length === 1 && method === 'POST') return handleEmbeddings(params, req.root);
1027
+ if (seg[0] === 'rerank' && seg.length === 1 && method === 'POST') return handleRerank(params, req.root);
1028
+ return notFound(`Unknown request URL: ${method} ${path}.`);
1029
+ }
1030
+
1031
+ // ── CONTROL PLANE (/v1/accounts/{account_id}/…) ────────────────────────────────────────
1032
+ if (path.startsWith(`${FIREWORKS_ACCOUNTS_PREFIX}/`)) {
1033
+ const seg = path.slice(FIREWORKS_ACCOUNTS_PREFIX.length).split('/').filter(Boolean).map(dec);
1034
+ const accountId = seg[0];
1035
+ if (!accountId) return gatewayError(404, `Unknown request URL: ${method} ${path}.`);
1036
+ const rest = seg.slice(1);
1037
+ // The account itself: GET /v1/accounts/{account_id}. The twin holds NO account rows — an
1038
+ // account exists only as the path prefix its resources live under, and the vendor's account
1039
+ // row is one-per-credential vendor state the API cannot create. Fabricating a 200 row for
1040
+ // ANY id (the old behavior) was a fake success; the vendor answers 404 for an account id it
1041
+ // does not know, and that is what the twin answers too. The honest gap is filed as
1042
+ // fireworks.control.get_account / fireworks.account.read (todos).
1043
+ if (rest.length === 0 && method === 'GET') {
1044
+ return gatewayError(404, `Not found: accounts/${accountId}`);
1045
+ }
1046
+ const resource = rest[0];
1047
+ // Verb-suffixed custom methods: `:cancel`, `:resume`, `:promote`, … (the vendor's AIP-158
1048
+ // style). The colon rides EITHER the collection segment (`jobs:cancel` — collection-level
1049
+ // verbs) or the id segment (`conf-job:cancel` — resource-level verbs); the id is the segment
1050
+ // before the verb.
1051
+ const colonAt = resource?.indexOf(':') ?? -1;
1052
+ let verb = colonAt >= 0 ? resource!.slice(colonAt + 1) : undefined;
1053
+ const baseResource = colonAt >= 0 ? resource!.slice(0, colonAt) : resource;
1054
+ let idSeg = rest[1];
1055
+ if (verb === undefined && rest.length >= 2 && rest[1]!.includes(':')) {
1056
+ const at = rest[1]!.indexOf(':');
1057
+ verb = rest[1]!.slice(at + 1);
1058
+ idSeg = rest[1]!.slice(0, at);
1059
+ }
1060
+
1061
+ const TYPE_MAP: Record<string, { plural: string; states: readonly string[]; stateKey: string }> = {
1062
+ deployments: { plural: 'deployments', states: DEPLOYMENT_STATES, stateKey: 'state' },
1063
+ datasets: { plural: 'datasets', states: ['STATE_UNSPECIFIED', 'UPLOADING', 'READY'], stateKey: 'state' },
1064
+ batchInferenceJobs: { plural: 'batchInferenceJobs', states: JOB_STATES, stateKey: 'state' },
1065
+ supervisedFineTuningJobs: { plural: 'supervisedFineTuningJobs', states: JOB_STATES, stateKey: 'state' },
1066
+ users: { plural: 'users', states: USER_STATES, stateKey: 'state' },
1067
+ models: { plural: 'models', states: DEPLOYMENT_STATES, stateKey: 'state' },
1068
+ secrets: { plural: 'secrets', states: DEPLOYMENT_STATES, stateKey: 'state' },
1069
+ };
1070
+ const meta = baseResource ? TYPE_MAP[baseResource] : undefined;
1071
+ if (!meta) return gatewayError(404, `Unknown request URL: ${method} ${path}.`);
1072
+ // The PATCH whitelist, per collection: exactly the MUTABLE fields the vendor's own update
1073
+ // operations accept (create-only and output-only fields are absent — a PATCH carrying one is
1074
+ // refused below, the way grpc-gateway refuses an unknown field; googleads-twin.ts is the
1075
+ // estate precedent). Sourced from the pinned spec's own update-operation schemas:
1076
+ // Gateway_UpdateDataset (displayName, exampleCount, userUploaded, evaluationResult,
1077
+ // transformed, splitted, evalProtocol, externalUrl, format, sourceJobName), the deployments
1078
+ // PATCH names the deployment fields, users PATCH names the user fields, secrets and models
1079
+ // the gatewaySecret/gatewayModel fields. The job collections carry NO update op in the spec
1080
+ // (delete+get only) — a PATCH there is the vendor's 404 (route not found), never a served
1081
+ // no-op, so the collections are absent from this map entirely.
1082
+ const PATCHABLE: Record<string, string[]> = {
1083
+ deployments: ['displayName', 'region', 'replicaCount', 'minReplicaCount', 'maxReplicaCount', 'precision'],
1084
+ users: ['displayName', 'email', 'role', 'permissionPreset'],
1085
+ models: ['displayName', 'contextLength', 'description'],
1086
+ secrets: ['keyName'],
1087
+ datasets: ['displayName', 'exampleCount', 'userUploaded', 'evaluationResult', 'transformed', 'splitted', 'evalProtocol', 'externalUrl', 'format', 'sourceJobName'],
1088
+ };
1089
+ // Rows are stored under the SINGULAR type (the same noun createControlResource writes); the
1090
+ // URL carries the plural collection.
1091
+ const singular = baseResource === 'deployments' ? 'deployment' : baseResource === 'datasets' ? 'dataset' : baseResource === 'batchInferenceJobs' ? 'batchInferenceJob' : baseResource === 'supervisedFineTuningJobs' ? 'supervisedFineTuningJob' : baseResource === 'users' ? 'user' : baseResource === 'models' ? 'model' : 'secret';
1092
+
1093
+ // LIST: GET /v1/accounts/{id}/<plural>
1094
+ if (rest.length === 1 && method === 'GET') {
1095
+ const all = rows(singular, accountId, req.root).filter((r) => !isTombstoned(r)).map(resourceView);
1096
+ const { items, nextPageToken } = paginate(all, url);
1097
+ return { status: 200, body: { [meta.plural]: items, nextPageToken, totalSize: all.length } };
1098
+ }
1099
+ // CREATE: POST /v1/accounts/{id}/<plural> — and ONLY that. A verb-suffixed collection
1100
+ // (`<plural>:estimateCost`, `apiKeys:<anything>`) is a custom method, not a create: the
1101
+ // vendor answers it with its own 404 (or serves it), never by running the create builder —
1102
+ // letting `POST …/supervisedFineTuningJobs:estimateCost` with a dataset body CREATE a job
1103
+ // would make an estimate mutate state.
1104
+ if (rest.length === 1 && method === 'POST' && !verb) {
1105
+ const build = (p: Record<string, unknown>): Record<string, unknown> => {
1106
+ // The vendor's create bodies are the RESOURCE DIRECTLY for deployments,
1107
+ // batchInferenceJobs, supervisedFineTuningJobs, users and secrets (requestBody →
1108
+ // $ref gatewayX). Datasets and models wrap: {dataset, datasetId} / {model, modelId}.
1109
+ const inner = (baseResource === 'datasets' && p.dataset && typeof p.dataset === 'object' ? p.dataset
1110
+ : baseResource === 'models' && p.model && typeof p.model === 'object' ? p.model
1111
+ : p) as Record<string, unknown>;
1112
+ const at = new Date(req.occurredAt ?? '1970-01-01T00:00:00Z').toISOString().replace(/\.\d{3}Z$/, 'Z');
1113
+ if (baseResource === 'deployments') {
1114
+ if (inner.baseModel === undefined) return { __error: 'baseModel is required' } as Record<string, unknown>;
1115
+ return {
1116
+ name: `accounts/${accountId}/deployments/PLACEHOLDER`,
1117
+ displayName: inner.displayName ?? '',
1118
+ baseModel: inner.baseModel,
1119
+ state: 'CREATING',
1120
+ status: okStatus(),
1121
+ createTime: at,
1122
+ updateTime: at,
1123
+ ...(inner.region !== undefined ? { region: inner.region } : {}),
1124
+ ...(inner.replicaCount !== undefined ? { replicaCount: inner.replicaCount } : {}),
1125
+ ...(inner.minReplicaCount !== undefined ? { minReplicaCount: inner.minReplicaCount } : {}),
1126
+ ...(inner.maxReplicaCount !== undefined ? { maxReplicaCount: inner.maxReplicaCount } : {}),
1127
+ ...(inner.precision !== undefined ? { precision: inner.precision } : {}),
1128
+ };
1129
+ }
1130
+ if (baseResource === 'datasets') {
1131
+ return {
1132
+ name: `accounts/${accountId}/datasets/PLACEHOLDER`,
1133
+ displayName: inner.displayName ?? '',
1134
+ state: 'UPLOADING',
1135
+ status: okStatus(),
1136
+ createTime: at,
1137
+ updateTime: at,
1138
+ exampleCount: 0,
1139
+ userUploaded: true,
1140
+ ...(inner.format !== undefined ? { format: inner.format } : {}),
1141
+ };
1142
+ }
1143
+ if (baseResource === 'batchInferenceJobs') {
1144
+ // The job's dataset reference must name a dataset that really landed: the vendor
1145
+ // validates the reference at create, and a twin that accepted any string would let a
1146
+ // mistyped id "create" a job over nothing.
1147
+ if (inner.inputDatasetId !== undefined) {
1148
+ const dsId = String(inner.inputDatasetId).split('/').pop() ?? '';
1149
+ const ds = getRow('dataset', dsId, accountId, req.root);
1150
+ if (!ds || isTombstoned(ds)) return { __error: `inputDatasetId not found: ${String(inner.inputDatasetId)}` } as Record<string, unknown>;
1151
+ }
1152
+ return {
1153
+ name: `accounts/${accountId}/batchInferenceJobs/PLACEHOLDER`,
1154
+ displayName: inner.displayName ?? '',
1155
+ state: 'JOB_STATE_CREATING',
1156
+ status: okStatus(),
1157
+ createTime: at,
1158
+ updateTime: at,
1159
+ ...(inner.model !== undefined ? { model: inner.model } : {}),
1160
+ ...(inner.inputDatasetId !== undefined ? { inputDatasetId: inner.inputDatasetId } : {}),
1161
+ ...(inner.outputDatasetId !== undefined ? { outputDatasetId: inner.outputDatasetId } : {}),
1162
+ };
1163
+ }
1164
+ if (baseResource === 'supervisedFineTuningJobs') {
1165
+ if (inner.dataset === undefined) return { __error: 'dataset is required' } as Record<string, unknown>;
1166
+ const dsId = String(inner.dataset).split('/').pop() ?? '';
1167
+ const ds = getRow('dataset', dsId, accountId, req.root);
1168
+ if (!ds || isTombstoned(ds)) return { __error: `dataset not found: ${String(inner.dataset)}` } as Record<string, unknown>;
1169
+ return {
1170
+ name: `accounts/${accountId}/supervisedFineTuningJobs/PLACEHOLDER`,
1171
+ displayName: inner.displayName ?? '',
1172
+ dataset: inner.dataset,
1173
+ state: 'JOB_STATE_CREATING',
1174
+ status: okStatus(),
1175
+ createTime: at,
1176
+ updateTime: at,
1177
+ ...(inner.baseModel !== undefined ? { baseModel: inner.baseModel } : {}),
1178
+ ...(inner.epochs !== undefined ? { epochs: inner.epochs } : {}),
1179
+ ...(inner.learningRate !== undefined ? { learningRate: inner.learningRate } : {}),
1180
+ };
1181
+ }
1182
+ if (baseResource === 'users') {
1183
+ if (inner.role === undefined) return { __error: 'role is required' } as Record<string, unknown>;
1184
+ if (!USER_ROLES.includes(inner.role as (typeof USER_ROLES)[number])) return { __error: `role must be one of ${USER_ROLES.join(', ')}` } as Record<string, unknown>;
1185
+ return {
1186
+ name: `accounts/${accountId}/users/PLACEHOLDER`,
1187
+ displayName: inner.displayName ?? '',
1188
+ email: inner.email ?? null,
1189
+ role: inner.role,
1190
+ state: 'CREATING',
1191
+ status: okStatus(),
1192
+ createTime: at,
1193
+ updateTime: at,
1194
+ ...(inner.permissionPreset !== undefined ? { permissionPreset: inner.permissionPreset } : {}),
1195
+ };
1196
+ }
1197
+ if (baseResource === 'secrets') {
1198
+ // The create body IS the gatewaySecret (the spec takes the resource directly), with
1199
+ // `name` and `keyName` required. `name` is the FULL resource name — a bare last
1200
+ // segment is accepted (clients generated from the spec send the short spelling) and
1201
+ // stored canonically; a full name must not be double-prefixed.
1202
+ if (inner.name === undefined || inner.keyName === undefined) return { __error: 'name and keyName are required' } as Record<string, unknown>;
1203
+ const shortName = String(inner.name).split('/').pop() as string;
1204
+ return {
1205
+ name: `accounts/${accountId}/secrets/${shortName}`,
1206
+ keyName: inner.keyName,
1207
+ // The vendor never returns the secret value on ANY read, including the create
1208
+ // response. The twin keeps it out of the stored row entirely — a secret that can be
1209
+ // read back is a custody lie.
1210
+ createTime: at,
1211
+ updateTime: at,
1212
+ };
1213
+ }
1214
+ if (baseResource === 'models') {
1215
+ return {
1216
+ name: `accounts/${accountId}/models/PLACEHOLDER`,
1217
+ displayName: inner.displayName ?? '',
1218
+ state: 'CREATING',
1219
+ status: okStatus(),
1220
+ createTime: at,
1221
+ updateTime: at,
1222
+ public: false,
1223
+ ...(inner.baseModelDetails !== undefined ? { baseModelDetails: inner.baseModelDetails } : {}),
1224
+ ...(inner.contextLength !== undefined ? { contextLength: inner.contextLength } : {}),
1225
+ ...(inner.description !== undefined ? { description: inner.description } : {}),
1226
+ };
1227
+ }
1228
+ return {};
1229
+ };
1230
+ const fields = build(params);
1231
+ if (fields.__error !== undefined) return gatewayError(400, String(fields.__error));
1232
+ // Replace the PLACEHOLDER name with the real id after the id is known. The singular type
1233
+ // name is what the id params are built from (deploymentId, userId, batchInferenceJobId…).
1234
+ const type = baseResource === 'deployments' ? 'deployment' : baseResource === 'datasets' ? 'dataset' : baseResource === 'batchInferenceJobs' ? 'batchInferenceJob' : baseResource === 'supervisedFineTuningJobs' ? 'supervisedFineTuningJob' : baseResource === 'users' ? 'user' : baseResource === 'models' ? 'model' : 'secret';
1235
+ const prefix = baseResource === 'batchInferenceJobs' ? 'bij' : baseResource === 'supervisedFineTuningJobs' ? 'sft' : type === 'secret' ? 'secret' : `${type}`;
1236
+ return createControlResource(type, prefix, accountId, params, req, url, (p) => build(p));
1237
+ }
1238
+ // SINGLE RESOURCE: GET/PATCH/DELETE /v1/accounts/{id}/<plural>/<resource_id>
1239
+ if (rest.length === 2 && !verb) {
1240
+ const rid = idSeg!;
1241
+ // The lookup is ACCOUNT-SCOPED: a row created under another account is invisible here and
1242
+ // answers the same NOT_FOUND an unknown id answers (the vendor's own tenancy — an account
1243
+ // cannot see, patch or delete another account's resource).
1244
+ const r = getRow(singular, rid, accountId, req.root);
1245
+ if (method === 'GET') {
1246
+ if (!r || isTombstoned(r)) return gatewayError(404, `Not found: ${baseResource}/${rid}`);
1247
+ return { status: 200, body: resourceView(r) };
1248
+ }
1249
+ if (method === 'PATCH') {
1250
+ // ONE policy for output-only fields across every collection (grpc-gateway's own): the
1251
+ // spec's readOnly fields are REFUSED with the same unknown-field 400 an invented field
1252
+ // gets — "Cannot find field." — never silently dropped and never written. (Silently
1253
+ // ignoring them on deployments while datasets refused them was two policies on one
1254
+ // plane; the refusal is grpc-gateway's own behavior for a client-set readOnly field.)
1255
+ if (!PATCHABLE[baseResource]) return gatewayError(404, `Unknown request URL: ${method} ${path}.`);
1256
+ if (!r || isTombstoned(r)) return gatewayError(404, `Not found: ${baseResource}/${rid}`);
1257
+ const patchable = { ...(params as Record<string, unknown>) };
1258
+ // The per-collection field whitelist (the same one create uses): an undeclared field is
1259
+ // refused the way grpc-gateway refuses an unknown field, and an output-only field never
1260
+ // passes (googleads-twin.ts is the estate precedent for the refusal shape).
1261
+ const unknown = Object.keys(patchable).filter((k) => !PATCHABLE[baseResource]?.includes(k));
1262
+ if (unknown.length) return gatewayError(400, `Invalid JSON payload received. Unknown name "${unknown[0]}": Cannot find field.`);
1263
+ await applyTwinWrite(SERVICE, {
1264
+ operation: `${singular}.update`, subjectType: singular, subjectId: `${accountId}/${rid}`,
1265
+ fields: patchable,
1266
+ ...(req.occurredAt ? { occurredAt: req.occurredAt } : {}), actor: { kind: 'agent' },
1267
+ }, req.root);
1268
+ return { status: 200, body: resourceView(getRow(singular, rid, accountId, req.root) ?? {}) };
1269
+ }
1270
+ if (method === 'DELETE') return deleteControlResource(singular, rid, accountId, req);
1271
+ }
1272
+ // CUSTOM VERB: POST /v1/accounts/{id}/<plural>/<resource_id>:<verb>
1273
+ if (rest.length === 2 && verb && method === 'POST') {
1274
+ const rid = idSeg!;
1275
+ const r = getRow(singular, rid, accountId, req.root);
1276
+ // :undelete addresses a DELETED row by design; every other verb needs a live row.
1277
+ if (!r || (isTombstoned(r) && verb !== 'undelete')) return gatewayError(404, `Not found: ${baseResource}/${rid}`);
1278
+ if (verb === 'cancel' && (baseResource === 'batchInferenceJobs' || baseResource === 'supervisedFineTuningJobs')) {
1279
+ if (r.state === 'JOB_STATE_CANCELLED' || r.state === 'JOB_STATE_CANCELLING') {
1280
+ return gatewayError(400, `Cannot cancel a job in state ${String(r.state)}`);
1281
+ }
1282
+ await applyTwinWrite(SERVICE, {
1283
+ operation: `${singular}.cancel`, subjectType: singular, subjectId: `${accountId}/${rid}`,
1284
+ fields: { state: 'JOB_STATE_CANCELLING' },
1285
+ ...(req.occurredAt ? { occurredAt: req.occurredAt } : {}), actor: { kind: 'agent' },
1286
+ }, req.root);
1287
+ // The vendor's cancel answers `{}`.
1288
+ return { status: 200, body: {} };
1289
+ }
1290
+ if (verb === 'resume' && (baseResource === 'batchInferenceJobs' || baseResource === 'supervisedFineTuningJobs')) {
1291
+ await applyTwinWrite(SERVICE, {
1292
+ operation: `${singular}.resume`, subjectType: singular, subjectId: `${accountId}/${rid}`,
1293
+ fields: { state: 'JOB_STATE_RUNNING' },
1294
+ ...(req.occurredAt ? { occurredAt: req.occurredAt } : {}), actor: { kind: 'agent' },
1295
+ }, req.root);
1296
+ return { status: 200, body: {} };
1297
+ }
1298
+ if (verb === 'undelete' && baseResource === 'deployments') {
1299
+ await applyTwinWrite(SERVICE, {
1300
+ operation: 'deployment.undelete', subjectType: 'deployment', subjectId: `${accountId}/${rid}`,
1301
+ fields: { _deleted: false, state: 'CREATING' },
1302
+ ...(req.occurredAt ? { occurredAt: req.occurredAt } : {}), actor: { kind: 'agent' },
1303
+ }, req.root);
1304
+ return { status: 200, body: resourceView(getRow('deployment', rid, accountId, req.root) ?? {}) };
1305
+ }
1306
+ if (verb === 'scale' && baseResource === 'deployments' && method === 'POST') {
1307
+ const replicaCount = (params as Record<string, unknown>).replicaCount;
1308
+ if (typeof replicaCount !== 'number') return gatewayError(400, 'replicaCount is required');
1309
+ await applyTwinWrite(SERVICE, {
1310
+ operation: 'deployment.scale', subjectType: 'deployment', subjectId: `${accountId}/${rid}`,
1311
+ fields: { replicaCount },
1312
+ ...(req.occurredAt ? { occurredAt: req.occurredAt } : {}), actor: { kind: 'agent' },
1313
+ }, req.root);
1314
+ return { status: 200, body: resourceView(getRow('deployment', rid, accountId, req.root) ?? {}) };
1315
+ }
1316
+ return gatewayError(404, `Unknown operation: ${verb} on ${baseResource}`);
1317
+ }
1318
+ // PATCH /v1/accounts/{id}/deployments/{deployment_id}:scale — the spec's own spelling
1319
+ // (Gateway_ScaleDeployment, body {replicaCount}); it answers `{}`, not the resource.
1320
+ if (rest.length === 2 && verb === 'scale' && baseResource === 'deployments' && method === 'PATCH') {
1321
+ const rid = idSeg!;
1322
+ const r = getRow('deployment', rid, accountId, req.root);
1323
+ if (!r || isTombstoned(r)) return gatewayError(404, `Not found: deployments/${rid}`);
1324
+ const replicaCount = (params as Record<string, unknown>).replicaCount;
1325
+ if (typeof replicaCount !== 'number') return gatewayError(400, 'replicaCount is required');
1326
+ await applyTwinWrite(SERVICE, {
1327
+ operation: 'deployment.scale', subjectType: 'deployment', subjectId: `${accountId}/${rid}`,
1328
+ fields: { replicaCount },
1329
+ ...(req.occurredAt ? { occurredAt: req.occurredAt } : {}), actor: { kind: 'agent' },
1330
+ }, req.root);
1331
+ return { status: 200, body: {} };
1332
+ }
1333
+ // users/{user_id}/apiKeys — nested under users. The verb-suffixed collection spelling
1334
+ // (`apiKeys:delete`) carries its verb on the apiKeys segment itself.
1335
+ if (baseResource === 'users' && rest.length >= 3 && rest[2]!.split(':')[0] === 'apiKeys') {
1336
+ const userId = rest[1]!;
1337
+ const user = getRow('user', userId, accountId, req.root);
1338
+ if (!user || user._deleted) return gatewayError(404, `Not found: users/${userId}`);
1339
+ const keySeg = rest.slice(3);
1340
+ if (keySeg.length === 0 && method === 'GET') {
1341
+ // The full key is returned ONCE at creation — the LIST answers it never. A list row
1342
+ // carrying `fw_…` would be a custody lie the single-key GET two branches down doesn't
1343
+ // commit, so the strip happens here too.
1344
+ const keys = rows('apiKey', accountId, req.root).filter((r) => !r._deleted && r._user_id === userId).map(resourceView);
1345
+ for (const k of keys) delete k.key;
1346
+ return { status: 200, body: { apiKeys: keys, nextPageToken: null, totalSize: keys.length } };
1347
+ }
1348
+ // POST .../apiKeys:delete — the vendor's own odd verb-suffixed delete with a {keyId} body
1349
+ // (collection-level spelling, Gateway_DeleteApiKey; answers `{}`).
1350
+ if (keySeg.length === 0 && method === 'POST' && rest[2] === 'apiKeys:delete') {
1351
+ const keyId = typeof params.keyId === 'string' ? params.keyId : undefined;
1352
+ if (!keyId) return gatewayError(400, 'keyId is required');
1353
+ const k = getRow('apiKey', keyId, accountId, req.root);
1354
+ if (!k || k._deleted || k._user_id !== userId) return gatewayError(404, `Not found: apiKeys/${keyId}`);
1355
+ await applyTwinWrite(SERVICE, {
1356
+ operation: 'apiKey.delete', subjectType: 'apiKey', subjectId: `${accountId}/${keyId}`, fields: { _deleted: true },
1357
+ ...(req.occurredAt ? { occurredAt: req.occurredAt } : {}), actor: { kind: 'agent' },
1358
+ }, req.root);
1359
+ return { status: 200, body: {} };
1360
+ }
1361
+ if (keySeg.length === 0 && method === 'POST' && rest[2] === 'apiKeys') {
1362
+ // The verb check is load-bearing: `apiKeys:frobnicate` must 404 below, never mint a key.
1363
+ const inner = (params.apiKey && typeof params.apiKey === 'object' ? params.apiKey : params) as Record<string, unknown>;
1364
+ const id = nextId('apiKey', 'key', accountId, req.root);
1365
+ const at = new Date(req.occurredAt ?? '1970-01-01T00:00:00Z').toISOString().replace(/\.\d{3}Z$/, 'Z');
1366
+ const key = `fw_${stableSuffix(id + at)}${fnv1a(id).toString(36)}`;
1367
+ await applyTwinWrite(SERVICE, {
1368
+ operation: 'apiKey.create', subjectType: 'apiKey', subjectId: `${accountId}/${id}`,
1369
+ fields: {
1370
+ name: `accounts/${accountId}/users/${userId}/apiKeys/${id}`,
1371
+ displayName: inner.displayName ?? 'default',
1372
+ key, // returned ONCE at creation, never again (the vendor's own contract)
1373
+ prefix: key.slice(0, 6),
1374
+ keyId: id,
1375
+ _user_id: userId,
1376
+ _account: accountId,
1377
+ createTime: at,
1378
+ expireTime: typeof inner.expireTime === 'string' ? inner.expireTime : null,
1379
+ },
1380
+ ...(req.occurredAt ? { occurredAt: req.occurredAt } : {}), actor: { kind: 'agent' },
1381
+ }, req.root);
1382
+ return { status: 200, body: resourceView(getRow('apiKey', id, accountId, req.root) ?? {}) };
1383
+ }
1384
+ if (keySeg.length === 1 && method === 'GET') {
1385
+ const k = getRow('apiKey', keySeg[0]!, accountId, req.root);
1386
+ if (!k || k._deleted || k._user_id !== userId) return gatewayError(404, `Not found: apiKeys/${keySeg[0]}`);
1387
+ const view = resourceView(k);
1388
+ delete view.key; // "only available upon creation and not stored thereafter"
1389
+ return { status: 200, body: view };
1390
+ }
1391
+ if (keySeg.length === 1 && method === 'PATCH') {
1392
+ const k = getRow('apiKey', keySeg[0]!, accountId, req.root);
1393
+ if (!k || k._deleted || k._user_id !== userId) return gatewayError(404, `Not found: apiKeys/${keySeg[0]}`);
1394
+ const patchable = { ...(params as Record<string, unknown>) };
1395
+ for (const f of ['key', 'keyId', 'prefix', 'secure', 'email', 'createTime', 'lastUsed', 'isFirepass']) delete patchable[f];
1396
+ const unknown = Object.keys(patchable).filter((f) => !['displayName', 'expireTime'].includes(f));
1397
+ if (unknown.length) return gatewayError(400, `Invalid JSON payload received. Unknown name "${unknown[0]}": Cannot find field.`);
1398
+ await applyTwinWrite(SERVICE, {
1399
+ operation: 'apiKey.update', subjectType: 'apiKey', subjectId: `${accountId}/${keySeg[0]!}`,
1400
+ fields: patchable,
1401
+ ...(req.occurredAt ? { occurredAt: req.occurredAt } : {}), actor: { kind: 'agent' },
1402
+ }, req.root);
1403
+ const view = resourceView(getRow('apiKey', keySeg[0]!, accountId, req.root) ?? {});
1404
+ delete view.key;
1405
+ return { status: 200, body: view };
1406
+ }
1407
+ // POST .../apiKeys:delete — the vendor's own odd verb-suffixed delete with a {keyId} body.
1408
+ // The key must belong to the addressed user: the id segment is under that user's own path,
1409
+ // and deleting another user's key through it would skip the scoping the collection-level
1410
+ // spelling enforces one branch up.
1411
+ if (keySeg.length === 1 && keySeg[0]!.endsWith(':delete') && method === 'POST') {
1412
+ const keyId = typeof params.keyId === 'string' ? params.keyId : keySeg[0]!.slice(0, -':delete'.length);
1413
+ const k = getRow('apiKey', keyId, accountId, req.root);
1414
+ if (!k || k._deleted || k._user_id !== userId) return gatewayError(404, `Not found: apiKeys/${keyId}`);
1415
+ await applyTwinWrite(SERVICE, {
1416
+ operation: 'apiKey.delete', subjectType: 'apiKey', subjectId: `${accountId}/${keyId}`, fields: { _deleted: true },
1417
+ ...(req.occurredAt ? { occurredAt: req.occurredAt } : {}), actor: { kind: 'agent' },
1418
+ }, req.root);
1419
+ return { status: 200, body: {} };
1420
+ }
1421
+ }
1422
+ return gatewayError(404, `Unknown request URL: ${method} ${path}.`);
1423
+ }
1424
+
1425
+ // Not on either plane — the twin serves nothing else.
1426
+ return notFound(`Unknown request URL: ${method} ${path}.`);
1427
+ }