@volter/twin-fireworks 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +202 -0
- package/README.md +184 -0
- package/dist/src/cli.d.ts +2 -0
- package/dist/src/cli.js +28 -0
- package/dist/src/fireworks-budget.d.ts +54 -0
- package/dist/src/fireworks-budget.js +146 -0
- package/dist/src/fireworks-capabilities.d.ts +4 -0
- package/dist/src/fireworks-capabilities.js +1205 -0
- package/dist/src/fireworks-conformance.d.ts +14 -0
- package/dist/src/fireworks-conformance.js +514 -0
- package/dist/src/fireworks-connector.d.ts +168 -0
- package/dist/src/fireworks-connector.js +641 -0
- package/dist/src/fireworks-models.d.ts +11 -0
- package/dist/src/fireworks-models.js +53 -0
- package/dist/src/fireworks-scenario.d.ts +55 -0
- package/dist/src/fireworks-scenario.js +171 -0
- package/dist/src/fireworks-server.d.ts +16 -0
- package/dist/src/fireworks-server.js +144 -0
- package/dist/src/fireworks-stub.d.ts +26 -0
- package/dist/src/fireworks-stub.js +78 -0
- package/dist/src/fireworks-twin.d.ts +51 -0
- package/dist/src/fireworks-twin.js +1426 -0
- package/dist/src/fireworks-types.d.ts +212 -0
- package/dist/src/fireworks-types.js +4 -0
- package/dist/src/index.d.ts +9 -0
- package/dist/src/index.js +105 -0
- package/package.json +52 -0
- package/src/cli.ts +27 -0
- package/src/fireworks-budget.ts +172 -0
- package/src/fireworks-capabilities.ts +1229 -0
- package/src/fireworks-conformance.ts +542 -0
- package/src/fireworks-connector.ts +700 -0
- package/src/fireworks-models.ts +63 -0
- package/src/fireworks-scenario.ts +191 -0
- package/src/fireworks-server.ts +153 -0
- package/src/fireworks-stub.ts +83 -0
- package/src/fireworks-twin.ts +1427 -0
- package/src/fireworks-types.ts +165 -0
- package/src/index.ts +134 -0
|
@@ -0,0 +1,1427 @@
|
|
|
1
|
+
// Fireworks twin REQUEST HANDLER — the canonical Fireworks API surface for the twin.
|
|
2
|
+
// Contract: handleFireworksTwinRequest({method, path, body}) -> {status, body}. It is the
|
|
3
|
+
// faithful Fireworks API the official clients (the `fireworks` Python SDK, `@ai-sdk/fireworks`,
|
|
4
|
+
// plain OpenAI clients pointed at api.fireworks.ai/inference/v1) talk to UNMODIFIED.
|
|
5
|
+
//
|
|
6
|
+
// Fireworks serves TWO planes on ONE host, and the twin serves both under the paths the vendor
|
|
7
|
+
// actually publishes:
|
|
8
|
+
// • INFERENCE — `/inference/v1/…` (OpenAI-compatible: chat/completions, completions, responses,
|
|
9
|
+
// embeddings, rerank) plus the Anthropic-compatible `/inference/v1/messages`. The twin cannot
|
|
10
|
+
// run a model, so generative output is a DETERMINISTIC STUB (fireworks-stub.ts), clearly
|
|
11
|
+
// labeled — never pretending to be real inference. The wire around it is faithful, INCLUDING
|
|
12
|
+
// Fireworks' documented OpenAI differences:
|
|
13
|
+
// - streaming usage arrives in the final chunk BY DEFAULT (OpenAI makes it opt-in);
|
|
14
|
+
// - `context_length_exceeded_behavior` defaults to `truncate`, not `error`;
|
|
15
|
+
// - `service_tier` accepts only 'priority' — every other value is treated as 'default'
|
|
16
|
+
// (documented; NOT an error);
|
|
17
|
+
// - validation failures answer the FastAPI `422 HTTPValidationError` envelope the vendor's
|
|
18
|
+
// own spec declares on these operations.
|
|
19
|
+
// • CONTROL — the Gateway REST API under `/v1/accounts/{account_id}/…`: deployments, datasets,
|
|
20
|
+
// batch-inference and supervised-fine-tuning jobs, users + their apiKeys, secrets, models.
|
|
21
|
+
// Stateful over the kernel action log; google.rpc-style errors (gatewayStatus {code,message});
|
|
22
|
+
// list envelopes `{<plural>, nextPageToken, totalSize}`; create ids passed as QUERY params
|
|
23
|
+
// (deployments/datasets/users) or inside the body (datasets), per the vendor's own spec.
|
|
24
|
+
//
|
|
25
|
+
// State lives in the kernel action log (D1): all writes are local actions, reads are the
|
|
26
|
+
// projection. No real Fireworks API is ever called from this path (D4). Streaming uses an
|
|
27
|
+
// INJECTED sink — no real sockets / setTimeout (D5 verify is offline + deterministic).
|
|
28
|
+
import { applyTwinWrite, projectResources } from '@volter/world-core';
|
|
29
|
+
import { FIREWORKS_MODEL_IDS } from './fireworks-models.ts';
|
|
30
|
+
import {
|
|
31
|
+
contentToText,
|
|
32
|
+
countPromptTokens,
|
|
33
|
+
estimateTokens,
|
|
34
|
+
fnv1a,
|
|
35
|
+
lastUserText,
|
|
36
|
+
pseudoEmbedding,
|
|
37
|
+
stubAssistantText,
|
|
38
|
+
stubReasoningText,
|
|
39
|
+
stubRelevanceScore,
|
|
40
|
+
} from './fireworks-stub.ts';
|
|
41
|
+
import type { FireworksScenarioEngine, FireworksScenarioRespond, ScriptedResult } from './fireworks-scenario.ts';
|
|
42
|
+
import { realizeFireworksRespond } from './fireworks-scenario.ts';
|
|
43
|
+
import type { ScenarioDecision } from '@volter/world-core';
|
|
44
|
+
import type {
|
|
45
|
+
FireworksAnthropicContentBlock,
|
|
46
|
+
FireworksAnthropicErrorType,
|
|
47
|
+
FireworksAnthropicMessage,
|
|
48
|
+
FireworksChatCompletion,
|
|
49
|
+
FireworksChatMessage,
|
|
50
|
+
FireworksChatRequest,
|
|
51
|
+
FireworksCompletion,
|
|
52
|
+
FireworksGatewayCode,
|
|
53
|
+
FireworksMessageParam,
|
|
54
|
+
FireworksResponseObject,
|
|
55
|
+
} from './fireworks-types.ts';
|
|
56
|
+
|
|
57
|
+
const SERVICE = 'fireworks';
|
|
58
|
+
|
|
59
|
+
/** The inference base path. The vendor's inference server root is `https://api.fireworks.ai/inference`
|
|
60
|
+
* and every OpenAI-compat operation hangs off `/v1/…` under it. */
|
|
61
|
+
export const FIREWORKS_INFERENCE_PREFIX = '/inference/v1';
|
|
62
|
+
/** The control-plane base path: everything Gateway REST hangs off `/v1/accounts/{account_id}/…`. */
|
|
63
|
+
export const FIREWORKS_ACCOUNTS_PREFIX = '/v1/accounts';
|
|
64
|
+
|
|
65
|
+
export type FireworksRequest = {
|
|
66
|
+
method: string;
|
|
67
|
+
path: string;
|
|
68
|
+
body?: string;
|
|
69
|
+
occurredAt?: string;
|
|
70
|
+
root?: string;
|
|
71
|
+
readOnly?: boolean;
|
|
72
|
+
/** The credential the caller presents (the SDKs' bearer `Authorization` header). When a request
|
|
73
|
+
* carries an auth SURFACE (this field set, or `headers` present), the twin holds it to the real
|
|
74
|
+
* vendor rule: a credential is required → 401 on missing/invalid. In-process trusted calls
|
|
75
|
+
* (capability verify, connector) omit BOTH and are not auth-gated. */
|
|
76
|
+
apiKey?: string;
|
|
77
|
+
/** Lower-cased request headers the HTTP server passes through so the handler can model auth. */
|
|
78
|
+
headers?: Record<string, string>;
|
|
79
|
+
/** When set on a streaming POST, chunks are written here (no sockets). */
|
|
80
|
+
sseSink?: SseSink;
|
|
81
|
+
/** Scenario scripting (fireworks-scenario.ts): when set, chat completions consult the engine
|
|
82
|
+
* first — a matching handler scripts the answer, a miss answers the labeled stub with a
|
|
83
|
+
* pointer to the miss. Twin-only scaffolding, never vendor surface. */
|
|
84
|
+
scenarioEngine?: FireworksScenarioEngine;
|
|
85
|
+
};
|
|
86
|
+
|
|
87
|
+
export type SseEvent = { data?: unknown; done?: boolean };
|
|
88
|
+
export type SseSink = (event: SseEvent) => void;
|
|
89
|
+
|
|
90
|
+
/** The handler response. `headers` (when present) are response headers the HTTP server should set. */
|
|
91
|
+
export type FireworksResponseEnvelope = { status: number; body: unknown; headers?: Record<string, string> };
|
|
92
|
+
|
|
93
|
+
// ── vendor-shaped errors ──────────────────────────────────────────────────────────────
|
|
94
|
+
// TWO error envelopes, one per plane — mixing them would be a wire lie:
|
|
95
|
+
// • the inference plane's OpenAI-compat surface answers `{ error: { message, … } }` (the
|
|
96
|
+
// OpenAI shape the clients parse); the Anthropic-compat /v1/messages answers
|
|
97
|
+
// `{ type: 'error', error: { type, message }, request_id }` (the Anthropic shape);
|
|
98
|
+
// • the control plane answers google.rpc-style `{ code: <gatewayCode>, message }` — the shape
|
|
99
|
+
// the vendor's own spec embeds on every resource (`gatewayStatus`) and its gatewayCode enum.
|
|
100
|
+
function errBody(message: string, extra: Record<string, unknown> = {}) {
|
|
101
|
+
return { error: { message, ...extra } };
|
|
102
|
+
}
|
|
103
|
+
function invalidRequest(message: string, extra: Record<string, unknown> = {}): FireworksResponseEnvelope {
|
|
104
|
+
return { status: 400, body: errBody(message, { code: 400, ...extra }) };
|
|
105
|
+
}
|
|
106
|
+
function notFound(message: string): FireworksResponseEnvelope {
|
|
107
|
+
return { status: 404, body: errBody(message, { code: 404 }) };
|
|
108
|
+
}
|
|
109
|
+
function authError(message: string): FireworksResponseEnvelope {
|
|
110
|
+
return { status: 401, body: errBody(message, { code: 401 }) };
|
|
111
|
+
}
|
|
112
|
+
/** The FastAPI validation envelope the vendor's own spec declares on the inference operations
|
|
113
|
+
* (`422 HTTPValidationError` → `detail: [{loc, msg, type}]`). */
|
|
114
|
+
function validationError(loc: (string | number)[], msg: string, type: string): FireworksResponseEnvelope {
|
|
115
|
+
return { status: 422, body: { detail: [{ loc, msg, type }] } };
|
|
116
|
+
}
|
|
117
|
+
/** The google.rpc-style control-plane error. `code` is a gatewayCode string, mapped from the
|
|
118
|
+
* HTTP status the way the vendor's own status embedding implies (NOT_FOUND → 'NOT_FOUND', …). */
|
|
119
|
+
function gatewayError(status: number, message: string): FireworksResponseEnvelope {
|
|
120
|
+
const code: FireworksGatewayCode =
|
|
121
|
+
status === 400 ? 'INVALID_ARGUMENT' :
|
|
122
|
+
status === 401 ? 'UNAUTHENTICATED' :
|
|
123
|
+
status === 403 ? 'PERMISSION_DENIED' :
|
|
124
|
+
status === 404 ? 'NOT_FOUND' :
|
|
125
|
+
status === 409 ? 'ALREADY_EXISTS' :
|
|
126
|
+
'UNKNOWN';
|
|
127
|
+
return { status, body: { code, message } };
|
|
128
|
+
}
|
|
129
|
+
|
|
130
|
+
// ── modeled authentication (401) ────────────────────────────────────────────────────────
|
|
131
|
+
// Real Fireworks requires a bearer credential on every request (docs.fireworks.ai/api-reference/
|
|
132
|
+
// introduction: "All requests … must include an Authorization header with a valid Bearer token")
|
|
133
|
+
// and answers 401 Unauthorized when it is missing or invalid (the vendor's own error-code table).
|
|
134
|
+
// The twin can't validate against real keys, so it models the CHECKABLE failures: a missing
|
|
135
|
+
// credential, and a reserved sentinel for the invalid-key path. Any other non-empty key is
|
|
136
|
+
// accepted. Trusted in-process calls carry NEITHER `headers` nor `apiKey` and are NOT auth-gated;
|
|
137
|
+
// the official clients always send a key → they pass.
|
|
138
|
+
function checkAuth(req: FireworksRequest): FireworksResponseEnvelope | null {
|
|
139
|
+
const auth = req.headers?.['authorization'];
|
|
140
|
+
const bearer = typeof auth === 'string' && auth.toLowerCase().startsWith('bearer ') ? auth.slice(7).trim() : '';
|
|
141
|
+
const key = (req.apiKey ?? '').trim() || bearer;
|
|
142
|
+
if (!key || key === 'fw_invalid' || key === 'invalid') {
|
|
143
|
+
// Each plane answers 401 in its OWN envelope — a control-plane 401 carrying the OpenAI shape
|
|
144
|
+
// would mix the envelopes the header above calls a wire lie.
|
|
145
|
+
if (req.path.startsWith(`${FIREWORKS_INFERENCE_PREFIX}/messages`)) {
|
|
146
|
+
return anthropicError(401, 'authentication_error', 'Unauthorized: missing or invalid API key');
|
|
147
|
+
}
|
|
148
|
+
if (req.path.startsWith(`${FIREWORKS_ACCOUNTS_PREFIX}/`)) {
|
|
149
|
+
return gatewayError(401, 'Unauthorized: missing or invalid API key');
|
|
150
|
+
}
|
|
151
|
+
return authError('Unauthorized: missing or invalid API key');
|
|
152
|
+
}
|
|
153
|
+
return null;
|
|
154
|
+
}
|
|
155
|
+
|
|
156
|
+
function nowEpoch(occurredAt?: string): number {
|
|
157
|
+
return Math.floor((occurredAt ? Date.parse(occurredAt) : 0) / 1000);
|
|
158
|
+
}
|
|
159
|
+
|
|
160
|
+
// ── kernel helpers ──────────────────────────────────────────────────────────────────────
|
|
161
|
+
/**
|
|
162
|
+
* Rows of `type`, SCOPED TO ONE ACCOUNT. Every control-plane resource is stored with the
|
|
163
|
+
* `accountId` it was created (or pulled) under, and every read goes through this — a resource
|
|
164
|
+
* addressed under a different account's path is invisible here, which is what makes the
|
|
165
|
+
* cross-account GET/PATCH/DELETE answer the vendor's own NOT_FOUND. `accountId === undefined`
|
|
166
|
+
* means the caller reads a plane with NO account segment (the inference plane's stored
|
|
167
|
+
* responses) — those rows are not account-scoped and the filter is skipped for them.
|
|
168
|
+
*/
|
|
169
|
+
/** A row is gone when the TWIN deleted it (`_deleted`, the twin's own marker) or when the
|
|
170
|
+
* CONNECTOR observed it vanish from the vendor (`deleted`, the kernel's tombstone field).
|
|
171
|
+
* Reading only the first served a resource the pull had correctly tombstoned (round-four review). */
|
|
172
|
+
export function isTombstoned(r: Record<string, unknown>): boolean {
|
|
173
|
+
return r._deleted === true || r.deleted === true;
|
|
174
|
+
}
|
|
175
|
+
function rows(type: string, accountId: string | undefined, root?: string): Array<Record<string, unknown>> {
|
|
176
|
+
return projectResources(SERVICE, root).filter((r) => r.type === type && (accountId === undefined || r._account === accountId));
|
|
177
|
+
}
|
|
178
|
+
/**
|
|
179
|
+
* Mint the next local id for `type`. Derived from the ID SET ALREADY IN STATE (a scan of the
|
|
180
|
+
* projection), never a row count — a count-mint silently clobbers a pulled vendor id sitting in a
|
|
181
|
+
* gap above the count (ADDING_A_TWIN.md §5). The `_twin_` infix namespaces LOCAL mints, so a
|
|
182
|
+
* pulled Fireworks id can never be matched by this regex and therefore can never be re-minted;
|
|
183
|
+
* the scan includes TOMBSTONED rows, so the counter RATCHETS across delete→recreate. The scan is
|
|
184
|
+
* PER ACCOUNT (rows() filters by `_account`) and matches the id SUFFIX after the account
|
|
185
|
+
* namespace, so two tenants each mint their own `dep_twin_1` without colliding.
|
|
186
|
+
*/
|
|
187
|
+
function nextId(type: string, prefix: string, accountId: string | undefined, root?: string): string {
|
|
188
|
+
let max = 0;
|
|
189
|
+
for (const r of rows(type, accountId, root)) {
|
|
190
|
+
const local = accountId !== undefined && typeof r.id === 'string' && r.id.startsWith(`${accountId}/`)
|
|
191
|
+
? r.id.slice(accountId.length + 1)
|
|
192
|
+
: String(r.id);
|
|
193
|
+
const m = new RegExp(`^${prefix}_twin_(\\d+)$`).exec(local);
|
|
194
|
+
if (m) max = Math.max(max, Number(m[1]));
|
|
195
|
+
}
|
|
196
|
+
return `${prefix}_twin_${max + 1}`;
|
|
197
|
+
}
|
|
198
|
+
/**
|
|
199
|
+
* Resolve one row by its BARE (wire) id inside one account. The kernel subject is the
|
|
200
|
+
* ACCOUNT-NAMESPACED id `${accountId}/${id}` — matching the vendor's own name grammar
|
|
201
|
+
* `accounts/{account}/{collection}/{id}` — so (type, subject) is unique per tenant and account
|
|
202
|
+
* B's create of the same id can never land on account A's row. Pulled rows carry the same
|
|
203
|
+
* namespaced id (the connector's mappers emit it), so created and pulled rows share one id space.
|
|
204
|
+
*/
|
|
205
|
+
function getRow(type: string, id: string, accountId: string | undefined, root?: string): Record<string, unknown> | undefined {
|
|
206
|
+
const full = accountId === undefined ? id : `${accountId}/${id}`;
|
|
207
|
+
return rows(type, accountId, root).find((r) => r.id === full);
|
|
208
|
+
}
|
|
209
|
+
/** Strip the kernel's housekeeping fields and the twin's private underscore-prefixed fields. */
|
|
210
|
+
function strip(r: Record<string, unknown>): Record<string, unknown> {
|
|
211
|
+
const out: Record<string, unknown> = {};
|
|
212
|
+
for (const [k, v] of Object.entries(r)) {
|
|
213
|
+
if (k === 'type' || k === 'updatedAt' || k === 'deleted' || k.startsWith('_')) continue;
|
|
214
|
+
out[k] = v;
|
|
215
|
+
}
|
|
216
|
+
return out;
|
|
217
|
+
}
|
|
218
|
+
|
|
219
|
+
// ── request parsing ─────────────────────────────────────────────────────────────────────
|
|
220
|
+
function parseJson(body?: string): Record<string, unknown> {
|
|
221
|
+
if (!body || !body.trim()) return {};
|
|
222
|
+
try {
|
|
223
|
+
const v = JSON.parse(body);
|
|
224
|
+
return v && typeof v === 'object' ? (v as Record<string, unknown>) : {};
|
|
225
|
+
} catch {
|
|
226
|
+
return {};
|
|
227
|
+
}
|
|
228
|
+
}
|
|
229
|
+
|
|
230
|
+
/** A deterministic id suffix from the request (so ids are stable + assertable). */
|
|
231
|
+
function stableSuffix(seedText: string): string {
|
|
232
|
+
return fnv1a(seedText).toString(36);
|
|
233
|
+
}
|
|
234
|
+
|
|
235
|
+
// ── model catalog (the 404 the vendor answers for an unknown model id) ──────────────────
|
|
236
|
+
/**
|
|
237
|
+
* The model ids the INFERENCE plane accepts: the static catalog (fireworks-models.ts, sourced)
|
|
238
|
+
* PLUS every model row the control plane holds (a create/pull under /v1/accounts/.../models —
|
|
239
|
+
* an account-owned model is addressable for inference exactly as the vendor allows, including
|
|
240
|
+
* the fine-tunes). A pulled/created row of the same id as a static entry just re-states it.
|
|
241
|
+
* Real Fireworks answers 404 "Model id not found" for an id outside this set — the twin must
|
|
242
|
+
* refuse the same way, never answer 200 for any string.
|
|
243
|
+
*/
|
|
244
|
+
function servedModelIds(root?: string): Set<string> {
|
|
245
|
+
const ids = new Set(FIREWORKS_MODEL_IDS);
|
|
246
|
+
for (const r of projectResources(SERVICE, root)) {
|
|
247
|
+
if (r.type !== 'model' || isTombstoned(r)) continue;
|
|
248
|
+
// A control-plane-created model is addressed ONLY by its FULL resource name on the
|
|
249
|
+
// inference plane (`accounts/{account}/models/{id}`) — the `name` the create handler
|
|
250
|
+
// minted. The bare suffix is NOT an alias: the vendor's own inference API takes full
|
|
251
|
+
// resource names for account-owned models (the static catalog's short aliases are
|
|
252
|
+
// documented serverless serving paths, a different thing), and accepting the bare suffix
|
|
253
|
+
// leaked one account's model into EVERY account's inference plane — a model created only
|
|
254
|
+
// under SECRETACCT answered 200 as the bare id while its qualified name 404'd under
|
|
255
|
+
// another account (inverted tenancy). The kernel subject stays the account-namespaced id
|
|
256
|
+
// (`{account}/{id}`); only the wire `name` joins the catalog.
|
|
257
|
+
if (typeof r.name === 'string') ids.add(r.name);
|
|
258
|
+
}
|
|
259
|
+
return ids;
|
|
260
|
+
}
|
|
261
|
+
|
|
262
|
+
/** The vendor's own refusal for an unknown model id (its error table: 404 covers "the model
|
|
263
|
+
* doesn't exist, the model is not deployed, or you don't have permission to access it"). */
|
|
264
|
+
function unknownModel(model: string): FireworksResponseEnvelope {
|
|
265
|
+
return notFound(`Model id not found: ${model}`);
|
|
266
|
+
}
|
|
267
|
+
|
|
268
|
+
// ── chat completions: validate the request the way Fireworks does ───────────────────────
|
|
269
|
+
const SERVICE_TIERS = ['auto', 'default', 'flex', 'priority'] as const;
|
|
270
|
+
const CONTEXT_BEHAVIORS = ['error', 'truncate'] as const;
|
|
271
|
+
|
|
272
|
+
/**
|
|
273
|
+
* The inference sampling parameters' documented ranges, from the pinned spec's own field
|
|
274
|
+
* descriptions (temperature "0 to 2", n "between 1 and 128", top_k "between 0 and 100",
|
|
275
|
+
* top_p "Required range: `0 <= x <= 1`" — CompletionRequest's own wording; ChatCompletionRequest
|
|
276
|
+
* carries the same nucleus-sampling semantics, frequency/presence_penalty "between -2 and 2")
|
|
277
|
+
* plus the OpenAI-compat types (role enum, stream boolean). The vendor's server (FastAPI)
|
|
278
|
+
* answers each violation with the 422 HTTPValidationError envelope and the pydantic error
|
|
279
|
+
* `type` — the twin answers the same, per parameter and constraint. `loc` matches where the
|
|
280
|
+
* field rides in the request body.
|
|
281
|
+
*/
|
|
282
|
+
const SAMPLING_RULES: Array<{ field: string; kind: 'number'; min?: number; max?: number; integer?: boolean } | { field: string; kind: 'enum'; values: readonly string[] } | { field: string; kind: 'bool' }> = [
|
|
283
|
+
{ field: 'temperature', kind: 'number', min: 0, max: 2 },
|
|
284
|
+
{ field: 'top_p', kind: 'number', min: 0, max: 1 },
|
|
285
|
+
{ field: 'n', kind: 'number', min: 1, max: 128, integer: true },
|
|
286
|
+
{ field: 'top_k', kind: 'number', min: 0, max: 100, integer: true },
|
|
287
|
+
{ field: 'frequency_penalty', kind: 'number', min: -2, max: 2 },
|
|
288
|
+
{ field: 'presence_penalty', kind: 'number', min: -2, max: 2 },
|
|
289
|
+
];
|
|
290
|
+
|
|
291
|
+
/** Validate the sampling/typing rules; returns the FastAPI 422 envelope on violation. Shared by
|
|
292
|
+
* the chat and legacy-completions doors so the two surfaces cannot drift. The message ROLE enum
|
|
293
|
+
* is checked per message by the chat door's own loop (the role rides inside `messages`). */
|
|
294
|
+
function validateSampling(params: Record<string, unknown>): FireworksResponseEnvelope | null {
|
|
295
|
+
for (const rule of SAMPLING_RULES) {
|
|
296
|
+
const v = params[rule.field];
|
|
297
|
+
if (v === undefined || v === null) continue;
|
|
298
|
+
if (rule.kind === 'number') {
|
|
299
|
+
// pydantic v2's kinds: a JSON string/bool where a number belongs is int_parsing (for an
|
|
300
|
+
// integer field) or float_parsing (for a float field) — int_type/float_type are
|
|
301
|
+
// python-typed-argument errors that never fire on a JSON body; a float where an integer
|
|
302
|
+
// belongs is int_from_float.
|
|
303
|
+
if (typeof v !== 'number' || Number.isNaN(v)) {
|
|
304
|
+
return validationError(['body', rule.field], `Input should be a valid ${rule.integer ? 'integer' : 'number'}`, rule.integer ? 'int_parsing' : 'float_parsing');
|
|
305
|
+
}
|
|
306
|
+
if (rule.integer && !Number.isInteger(v)) return validationError(['body', rule.field], 'Input should be a valid integer', 'int_from_float');
|
|
307
|
+
if (rule.min !== undefined && v < rule.min) return validationError(['body', rule.field], `Input should be greater than or equal to ${rule.min}`, 'greater_than_equal');
|
|
308
|
+
if (rule.max !== undefined && v > rule.max) return validationError(['body', rule.field], `Input should be less than or equal to ${rule.max}`, 'less_than_equal');
|
|
309
|
+
}
|
|
310
|
+
}
|
|
311
|
+
if (params.stream !== undefined && params.stream !== null && typeof params.stream !== 'boolean') {
|
|
312
|
+
return validationError(['body', 'stream'], 'Input should be a valid boolean', 'bool_type');
|
|
313
|
+
}
|
|
314
|
+
return null;
|
|
315
|
+
}
|
|
316
|
+
|
|
317
|
+
/** The OpenAI-compat message role enum (the spec's ChatCompletionRequestMessage.role). */
|
|
318
|
+
const MESSAGE_ROLES = ['system', 'user', 'assistant', 'tool', 'function', 'developer'] as const;
|
|
319
|
+
|
|
320
|
+
type ChatArgs = {
|
|
321
|
+
model: string;
|
|
322
|
+
messages: FireworksMessageParam[];
|
|
323
|
+
promptText: string;
|
|
324
|
+
tools?: unknown;
|
|
325
|
+
maxTokens?: number;
|
|
326
|
+
stop?: string[];
|
|
327
|
+
stream: boolean;
|
|
328
|
+
includeUsage: boolean;
|
|
329
|
+
serviceTier: string;
|
|
330
|
+
contextBehavior: 'error' | 'truncate';
|
|
331
|
+
responseFormat?: { type: string; schema?: unknown };
|
|
332
|
+
reasoningEffort?: string | number | boolean;
|
|
333
|
+
};
|
|
334
|
+
|
|
335
|
+
/**
|
|
336
|
+
* Fireworks' documented OpenAI differences are the fidelity surface here:
|
|
337
|
+
* • `max_tokens`/`max_completion_tokens` are MUTUALLY EXCLUSIVE (the spec's own note: "Alias for
|
|
338
|
+
* max_tokens. Cannot be specified together with max_tokens.") — OpenAI allows both;
|
|
339
|
+
* • `service_tier` is a closed enum whose values are all ACCEPTED but only 'priority' is
|
|
340
|
+
* honored — the others are treated as 'default', never an error;
|
|
341
|
+
* • `context_length_exceeded_behavior` defaults to 'truncate' (OpenAI's is 'error').
|
|
342
|
+
* The twin has no real context window, so the truncate path is modeled as the documented
|
|
343
|
+
* SEMANTIC (max_tokens lowered to fit) whenever a stub context is exceeded.
|
|
344
|
+
*/
|
|
345
|
+
function validateChat(params: Record<string, unknown>): { args: ChatArgs } | { error: FireworksResponseEnvelope } {
|
|
346
|
+
if (params.model === undefined) return { error: validationError(['body', 'model'], 'Field required', 'missing') };
|
|
347
|
+
if (typeof params.model !== 'string' || !params.model) return { error: validationError(['body', 'model'], 'Input should be a valid string', 'string_type') };
|
|
348
|
+
if (!Array.isArray(params.messages)) return { error: validationError(['body', 'messages'], 'Field required', 'missing') };
|
|
349
|
+
if (params.messages.length === 0) return { error: invalidRequest("'messages' must not be empty") };
|
|
350
|
+
const messages = params.messages as FireworksMessageParam[];
|
|
351
|
+
for (const [i, m] of messages.entries()) {
|
|
352
|
+
if (!m || typeof m !== 'object' || typeof m.role !== 'string') {
|
|
353
|
+
return { error: validationError(['body', 'messages'], 'Input should be a valid dictionary', 'model_attributes_type') };
|
|
354
|
+
}
|
|
355
|
+
if (!MESSAGE_ROLES.includes(m.role as (typeof MESSAGE_ROLES)[number])) {
|
|
356
|
+
// FastAPI's loc carries the message INDEX (`body.messages.<i>.role`), not just the field.
|
|
357
|
+
return { error: validationError(['body', 'messages', i, 'role'], `Input should be one of ${MESSAGE_ROLES.map((r) => `'${r}'`).join(', ')}`, 'enum') };
|
|
358
|
+
}
|
|
359
|
+
}
|
|
360
|
+
// The sampling/typing table (temperature/n/top_k/penalties/stream) — the same rules the legacy
|
|
361
|
+
// door applies, so the two inference surfaces cannot drift.
|
|
362
|
+
const sampling = validateSampling(params);
|
|
363
|
+
if (sampling) return { error: sampling };
|
|
364
|
+
if (params.max_tokens !== undefined && params.max_tokens !== null && params.max_completion_tokens !== undefined && params.max_completion_tokens !== null) {
|
|
365
|
+
return { error: invalidRequest("'max_tokens' and 'max_completion_tokens' cannot both be specified — 'max_completion_tokens' is an alias for 'max_tokens'") };
|
|
366
|
+
}
|
|
367
|
+
const maxRaw = params.max_completion_tokens ?? params.max_tokens;
|
|
368
|
+
let maxTokens: number | undefined;
|
|
369
|
+
if (maxRaw !== undefined && maxRaw !== null) {
|
|
370
|
+
// pydantic parses the JSON value BEFORE any range check: a string is int_parsing, a float
|
|
371
|
+
// int_from_float, and only a parsed integer then hits the ≥1 bound.
|
|
372
|
+
maxTokens = Number(maxRaw);
|
|
373
|
+
if (typeof maxRaw !== 'number' || Number.isNaN(maxRaw)) return { error: validationError(['body', 'max_tokens'], 'Input should be a valid integer', 'int_parsing') };
|
|
374
|
+
if (!Number.isInteger(maxTokens)) return { error: validationError(['body', 'max_tokens'], 'Input should be a valid integer', 'int_from_float') };
|
|
375
|
+
if (maxTokens < 1) return { error: validationError(['body', 'max_tokens'], 'Input should be a valid integer greater than or equal to 1', 'greater_than_equal') };
|
|
376
|
+
}
|
|
377
|
+
let stop: string[] | undefined;
|
|
378
|
+
if (params.stop !== undefined && params.stop !== null) {
|
|
379
|
+
if (typeof params.stop === 'string') stop = [params.stop];
|
|
380
|
+
else if (Array.isArray(params.stop)) stop = params.stop as string[];
|
|
381
|
+
else return { error: validationError(['body', 'stop'], 'Input should be a valid string or array of strings', 'string_type') };
|
|
382
|
+
}
|
|
383
|
+
let serviceTier = 'default';
|
|
384
|
+
if (params.service_tier !== undefined && params.service_tier !== null) {
|
|
385
|
+
if (typeof params.service_tier !== 'string' || !SERVICE_TIERS.includes(params.service_tier as (typeof SERVICE_TIERS)[number])) {
|
|
386
|
+
return { error: validationError(['body', 'service_tier'], `Input should be one of ${SERVICE_TIERS.map((t) => `'${t}'`).join(', ')}`, 'enum') };
|
|
387
|
+
}
|
|
388
|
+
// The vendor's own semantics: "Only 'priority' is supported, while all other values will be
|
|
389
|
+
// treated as 'default' tier." Every enum value is ACCEPTED; 'auto'/'flex' do NOT error.
|
|
390
|
+
serviceTier = params.service_tier === 'priority' ? 'priority' : 'default';
|
|
391
|
+
}
|
|
392
|
+
let contextBehavior: 'error' | 'truncate' = 'truncate';
|
|
393
|
+
if (params.context_length_exceeded_behavior !== undefined && params.context_length_exceeded_behavior !== null) {
|
|
394
|
+
if (typeof params.context_length_exceeded_behavior !== 'string' || !CONTEXT_BEHAVIORS.includes(params.context_length_exceeded_behavior as 'error' | 'truncate')) {
|
|
395
|
+
return { error: validationError(['body', 'context_length_exceeded_behavior'], "Input should be one of 'error', 'truncate'", 'enum') };
|
|
396
|
+
}
|
|
397
|
+
contextBehavior = params.context_length_exceeded_behavior as 'error' | 'truncate';
|
|
398
|
+
}
|
|
399
|
+
let responseFormat: { type: string; schema?: unknown } | undefined;
|
|
400
|
+
const rf = params.response_format as { type?: unknown; json_schema?: unknown } | undefined;
|
|
401
|
+
if (rf && typeof rf === 'object') {
|
|
402
|
+
if (rf.type === 'json_object') responseFormat = { type: 'json_object' };
|
|
403
|
+
else if (rf.type === 'json_schema') responseFormat = { type: 'json_schema', schema: rf.json_schema };
|
|
404
|
+
else if (rf.type !== undefined && rf.type !== 'text') return { error: invalidRequest("'response_format.type' must be one of 'text', 'json_object', 'json_schema'") };
|
|
405
|
+
}
|
|
406
|
+
const stream = params.stream === true;
|
|
407
|
+
// Fireworks streams usage BY DEFAULT (the documented OpenAI difference); `include_usage: false`
|
|
408
|
+
// is the opt-OUT — the inverse of OpenAI's opt-IN.
|
|
409
|
+
const streamOptions = params.stream_options as { include_usage?: boolean } | undefined;
|
|
410
|
+
const includeUsage = streamOptions?.include_usage !== false;
|
|
411
|
+
return {
|
|
412
|
+
args: {
|
|
413
|
+
model: params.model,
|
|
414
|
+
messages,
|
|
415
|
+
promptText: messages.map((m) => contentToText(m.content)).join('\n'),
|
|
416
|
+
...(params.tools !== undefined ? { tools: params.tools } : {}),
|
|
417
|
+
...(maxTokens !== undefined ? { maxTokens } : {}),
|
|
418
|
+
...(stop !== undefined ? { stop } : {}),
|
|
419
|
+
stream,
|
|
420
|
+
includeUsage,
|
|
421
|
+
serviceTier,
|
|
422
|
+
contextBehavior,
|
|
423
|
+
...(responseFormat !== undefined ? { responseFormat } : {}),
|
|
424
|
+
...(params.reasoning_effort !== undefined && params.reasoning_effort !== null ? { reasoningEffort: params.reasoning_effort as string | number | boolean } : {}),
|
|
425
|
+
},
|
|
426
|
+
};
|
|
427
|
+
}
|
|
428
|
+
|
|
429
|
+
/** Build ONE deterministic stub chat completion. */
|
|
430
|
+
function buildChatCompletion(args: ChatArgs, occurredAt?: string, decision?: ScenarioDecision): FireworksChatCompletion | FireworksResponseEnvelope {
|
|
431
|
+
const promptTokens = countPromptTokens(args.messages);
|
|
432
|
+
// Scenario scripting first — the DECISION was made (and any fault honored) by the request
|
|
433
|
+
// handler through the engine's serve(); only a content decision reaches here. A matching
|
|
434
|
+
// handler scripts the answer; a miss answers the labeled stub with a pointer naming the miss
|
|
435
|
+
// and the features seen.
|
|
436
|
+
let scripted: ScriptedResult | null = null;
|
|
437
|
+
let missTeach = '';
|
|
438
|
+
if (decision) {
|
|
439
|
+
if (decision.kind === 'handler') {
|
|
440
|
+
scripted = realizeFireworksRespond(decision.respond as FireworksScenarioRespond);
|
|
441
|
+
} else {
|
|
442
|
+
missTeach = `\n[twin-scenario miss — no handler matched. Author one in the world dir's handlers/fireworks.json (GET /twin explains; GET /twin/scenario lists handlers + misses). Features seen: ${JSON.stringify(decision.miss.features)}]`;
|
|
443
|
+
}
|
|
444
|
+
}
|
|
445
|
+
if (scripted) {
|
|
446
|
+
const message: FireworksChatMessage = { role: 'assistant', content: scripted.text ?? '', ...(scripted.toolCalls.length ? { tool_calls: scripted.toolCalls } : {}) };
|
|
447
|
+
if (scripted.reasoning !== null) message.reasoning_content = scripted.reasoning;
|
|
448
|
+
const completionTokens = estimateTokens(JSON.stringify(scripted.toolCalls.length ? scripted.toolCalls : scripted.text ?? ''));
|
|
449
|
+
return {
|
|
450
|
+
id: `chatcmpl-twin-${stableSuffix(JSON.stringify(args.messages) + args.model)}`,
|
|
451
|
+
object: 'chat.completion',
|
|
452
|
+
created: nowEpoch(occurredAt),
|
|
453
|
+
model: args.model,
|
|
454
|
+
choices: [{ index: 0, message, finish_reason: scripted.finishReason, logprobs: null }],
|
|
455
|
+
usage: { prompt_tokens: promptTokens, total_tokens: promptTokens + completionTokens, completion_tokens: completionTokens },
|
|
456
|
+
};
|
|
457
|
+
}
|
|
458
|
+
let text = args.responseFormat?.type === 'json_object' || args.responseFormat?.type === 'json_schema'
|
|
459
|
+
? `${stubAssistantText(args.model, args.promptText)}\n{"twin_stub":true}`
|
|
460
|
+
: stubAssistantText(args.model, args.promptText);
|
|
461
|
+
if (missTeach) text += missTeach;
|
|
462
|
+
let finish = 'stop';
|
|
463
|
+
// Truncate at the EARLIEST-occurring stop sequence across the whole `stop` list.
|
|
464
|
+
let stopAt = -1;
|
|
465
|
+
for (const s of args.stop ?? []) {
|
|
466
|
+
if (!s) continue;
|
|
467
|
+
const i = text.indexOf(s);
|
|
468
|
+
if (i >= 0 && (stopAt < 0 || i < stopAt)) stopAt = i;
|
|
469
|
+
}
|
|
470
|
+
if (stopAt >= 0) { text = text.slice(0, stopAt); finish = 'stop'; }
|
|
471
|
+
// THE VENDOR DIFFERENCE: exceeding the (stub) context truncates max_tokens by default instead
|
|
472
|
+
// of erroring. The stub context is generous and fixed; the behavior is the fidelity surface.
|
|
473
|
+
const stubContextTokens = 8192;
|
|
474
|
+
if (promptTokens + (args.maxTokens ?? 0) > stubContextTokens) {
|
|
475
|
+
if (args.contextBehavior === 'error') {
|
|
476
|
+
return invalidRequest(`This model's maximum context length is ${stubContextTokens} tokens. However, you requested ${(args.maxTokens ?? 0) + promptTokens} tokens in the messages, which exceeds the model's context limit.`);
|
|
477
|
+
}
|
|
478
|
+
// truncate: max_tokens is lowered to fit.
|
|
479
|
+
if (args.maxTokens !== undefined) args.maxTokens = Math.max(1, stubContextTokens - promptTokens);
|
|
480
|
+
}
|
|
481
|
+
if (args.maxTokens !== undefined && estimateTokens(text) > args.maxTokens) {
|
|
482
|
+
text = text.slice(0, args.maxTokens * 4);
|
|
483
|
+
finish = 'length';
|
|
484
|
+
}
|
|
485
|
+
const message: FireworksChatMessage = { role: 'assistant', content: text };
|
|
486
|
+
// reasoning_effort (any truthy non-'none'/'false' value) surfaces Fireworks' separate
|
|
487
|
+
// `reasoning_content` field — the field the vendor's reasoning models answer with.
|
|
488
|
+
const effortOn = args.reasoningEffort !== undefined && args.reasoningEffort !== 'none' && args.reasoningEffort !== false;
|
|
489
|
+
if (effortOn) message.reasoning_content = stubReasoningText(args.model, args.promptText);
|
|
490
|
+
const completionTokens = estimateTokens(text) + estimateTokens(message.reasoning_content ?? '');
|
|
491
|
+
const usage = { prompt_tokens: promptTokens, total_tokens: promptTokens + completionTokens, completion_tokens: completionTokens };
|
|
492
|
+
const id = `chatcmpl-twin-${stableSuffix(JSON.stringify(args.messages) + args.model)}`;
|
|
493
|
+
return {
|
|
494
|
+
id,
|
|
495
|
+
object: 'chat.completion',
|
|
496
|
+
created: nowEpoch(occurredAt),
|
|
497
|
+
model: args.model,
|
|
498
|
+
choices: [{ index: 0, message, finish_reason: finish, logprobs: null }],
|
|
499
|
+
// usage is carried unconditionally (the vendor's streaming behavior implies it is always
|
|
500
|
+
// computed; the schema marks it nullable only for the echo/logprobs edge paths).
|
|
501
|
+
usage,
|
|
502
|
+
};
|
|
503
|
+
}
|
|
504
|
+
|
|
505
|
+
/** Split text into deterministic streaming chunks (≤ ~20 chars each), preserving order. */
|
|
506
|
+
function chunkText(text: string): string[] {
|
|
507
|
+
if (!text) return [];
|
|
508
|
+
const out: string[] = [];
|
|
509
|
+
for (let i = 0; i < text.length; i += 20) out.push(text.slice(i, i + 20));
|
|
510
|
+
return out;
|
|
511
|
+
}
|
|
512
|
+
|
|
513
|
+
/**
|
|
514
|
+
* Emit the vendor-faithful Fireworks streaming sequence into the injected sink (NO sockets, NO
|
|
515
|
+
* setTimeout). THE VENDOR DIFFERENCE (docs.fireworks.ai/tools-sdks/openai-compatibility): "For
|
|
516
|
+
* streaming responses, the `usage` field is returned in the very last chunk on the response (i.e.
|
|
517
|
+
* the one having `finish_reason` set)" — BY DEFAULT, no `stream_options.include_usage` needed
|
|
518
|
+
* (OpenAI makes it opt-in; Fireworks makes it opt-OUT via `include_usage: false`).
|
|
519
|
+
*/
|
|
520
|
+
function streamChat(args: ChatArgs, sink: SseSink, occurredAt?: string, decision?: ScenarioDecision): FireworksChatCompletion | FireworksResponseEnvelope {
|
|
521
|
+
const built = buildChatCompletion(args, occurredAt, decision);
|
|
522
|
+
if ('status' in built) return built;
|
|
523
|
+
const full = built;
|
|
524
|
+
const base = { id: full.id, object: 'chat.completion.chunk' as const, created: full.created, model: full.model };
|
|
525
|
+
sink({ data: { ...base, choices: [{ index: 0, delta: { role: 'assistant', content: '' }, finish_reason: null }], usage: null } });
|
|
526
|
+
if (full.choices[0]!.message.reasoning_content) {
|
|
527
|
+
sink({ data: { ...base, choices: [{ index: 0, delta: { reasoning_content: full.choices[0]!.message.reasoning_content }, finish_reason: null }], usage: null } });
|
|
528
|
+
}
|
|
529
|
+
for (const piece of chunkText(full.choices[0]!.message.content ?? '')) {
|
|
530
|
+
sink({ data: { ...base, choices: [{ index: 0, delta: { content: piece }, finish_reason: null }], usage: null } });
|
|
531
|
+
}
|
|
532
|
+
// The final chunk carries finish_reason AND — by default — the usage.
|
|
533
|
+
sink({
|
|
534
|
+
data: {
|
|
535
|
+
...base,
|
|
536
|
+
choices: [{ index: 0, delta: {}, finish_reason: full.choices[0]!.finish_reason }],
|
|
537
|
+
...(args.includeUsage ? { usage: full.usage } : { usage: null }),
|
|
538
|
+
},
|
|
539
|
+
});
|
|
540
|
+
sink({ done: true });
|
|
541
|
+
return full;
|
|
542
|
+
}
|
|
543
|
+
|
|
544
|
+
// ── legacy completions (/v1/completions) ────────────────────────────────────────────────
|
|
545
|
+
/** Stream a legacy completion the OpenAI-compatible way: chunks carry the delta in
|
|
546
|
+
* `choices[].text` (`object: 'text_completion'` throughout — the legacy stream has no separate
|
|
547
|
+
* chunk object), ending with the finish chunk that — the Fireworks default — carries `usage`. */
|
|
548
|
+
function streamCompletion(full: FireworksCompletion, sink: SseSink, includeUsage: boolean): FireworksCompletion {
|
|
549
|
+
const base = { id: full.id, object: 'text_completion' as const, created: full.created, model: full.model };
|
|
550
|
+
for (const choice of full.choices) {
|
|
551
|
+
sink({ data: { ...base, choices: [{ index: choice.index, text: choice.text, finish_reason: null, logprobs: choice.logprobs }], usage: null } });
|
|
552
|
+
}
|
|
553
|
+
sink({
|
|
554
|
+
data: {
|
|
555
|
+
...base,
|
|
556
|
+
choices: full.choices.map((c) => ({ index: c.index, text: '', finish_reason: c.finish_reason, logprobs: c.logprobs })),
|
|
557
|
+
...(includeUsage ? { usage: full.usage } : { usage: null }),
|
|
558
|
+
},
|
|
559
|
+
});
|
|
560
|
+
sink({ done: true });
|
|
561
|
+
return full;
|
|
562
|
+
}
|
|
563
|
+
|
|
564
|
+
function handleCompletion(params: Record<string, unknown>, occurredAt?: string, sink?: SseSink, root?: string): FireworksResponseEnvelope {
|
|
565
|
+
if (params.model === undefined) return { status: 422, body: { detail: [{ loc: ['body', 'model'], msg: 'Field required', type: 'missing' }] } };
|
|
566
|
+
if (params.prompt === undefined) return { status: 422, body: { detail: [{ loc: ['body', 'prompt'], msg: 'Field required', type: 'missing' }] } };
|
|
567
|
+
if (typeof params.model !== 'string' || !params.model) return { status: 422, body: { detail: [{ loc: ['body', 'model'], msg: 'Input should be a valid string', type: 'string_type' }] } };
|
|
568
|
+
if (!servedModelIds(root).has(params.model)) return unknownModel(params.model);
|
|
569
|
+
// The same sampling/typing table the chat door applies.
|
|
570
|
+
const sampling = validateSampling(params);
|
|
571
|
+
if (sampling) return sampling;
|
|
572
|
+
const prompts = Array.isArray(params.prompt) ? (params.prompt as unknown[]).map(String) : [String(params.prompt)];
|
|
573
|
+
const maxRaw = params.max_completion_tokens ?? params.max_tokens;
|
|
574
|
+
if (params.max_tokens !== undefined && params.max_tokens !== null && params.max_completion_tokens !== undefined && params.max_completion_tokens !== null) {
|
|
575
|
+
return invalidRequest("'max_tokens' and 'max_completion_tokens' cannot both be specified — 'max_completion_tokens' is an alias for 'max_tokens'");
|
|
576
|
+
}
|
|
577
|
+
let maxTokens: number | undefined;
|
|
578
|
+
if (maxRaw !== undefined && maxRaw !== null) {
|
|
579
|
+
// The same pydantic parse order the chat door applies (string → int_parsing, float →
|
|
580
|
+
// int_from_float, then the ≥1 bound).
|
|
581
|
+
maxTokens = Number(maxRaw);
|
|
582
|
+
if (typeof maxRaw !== 'number' || Number.isNaN(maxRaw)) return { status: 422, body: { detail: [{ loc: ['body', 'max_tokens'], msg: 'Input should be a valid integer', type: 'int_parsing' }] } };
|
|
583
|
+
if (!Number.isInteger(maxTokens)) return { status: 422, body: { detail: [{ loc: ['body', 'max_tokens'], msg: 'Input should be a valid integer', type: 'int_from_float' }] } };
|
|
584
|
+
if (maxTokens < 1) return { status: 422, body: { detail: [{ loc: ['body', 'max_tokens'], msg: 'Input should be a valid integer greater than or equal to 1', type: 'greater_than_equal' }] } };
|
|
585
|
+
}
|
|
586
|
+
const model = String(params.model);
|
|
587
|
+
const promptTokens = prompts.reduce((sum, p) => sum + estimateTokens(p), 0);
|
|
588
|
+
const choices = prompts.map((p, i) => {
|
|
589
|
+
let text = stubAssistantText(model, p);
|
|
590
|
+
let finish = 'stop';
|
|
591
|
+
if (maxTokens !== undefined && estimateTokens(text) > maxTokens) {
|
|
592
|
+
text = text.slice(0, maxTokens * 4);
|
|
593
|
+
finish = 'length';
|
|
594
|
+
}
|
|
595
|
+
return { index: i, text, finish_reason: finish, logprobs: null };
|
|
596
|
+
});
|
|
597
|
+
const completionTokens = choices.reduce((sum, c) => sum + estimateTokens(c.text), 0);
|
|
598
|
+
const body: FireworksCompletion = {
|
|
599
|
+
id: `cmpl-twin-${stableSuffix(JSON.stringify(prompts) + model)}`,
|
|
600
|
+
object: 'text_completion',
|
|
601
|
+
created: nowEpoch(occurredAt),
|
|
602
|
+
model,
|
|
603
|
+
choices,
|
|
604
|
+
// The spec marks `usage` REQUIRED on Completion (nullable only on chat).
|
|
605
|
+
usage: { prompt_tokens: promptTokens, total_tokens: promptTokens + completionTokens, completion_tokens: completionTokens },
|
|
606
|
+
};
|
|
607
|
+
// The server hands a sink only for `stream:true`; ignoring it would answer a 200 event stream
|
|
608
|
+
// with zero frames — a fake success on the wire.
|
|
609
|
+
const streamOptions = params.stream_options as { include_usage?: unknown } | undefined;
|
|
610
|
+
if (sink) return { status: 200, body: streamCompletion(body, sink, streamOptions?.include_usage !== false) };
|
|
611
|
+
return { status: 200, body };
|
|
612
|
+
}
|
|
613
|
+
|
|
614
|
+
// ── Responses API (/v1/responses — stateful CRUD over the kernel log) ───────────────────
|
|
615
|
+
/**
|
|
616
|
+
* Fireworks' Responses API stores responses server-side (`store` defaults true; `store:false`
|
|
617
|
+
* answers with a NULL id per the vendor's own schema: "Will be None if store=False"). The twin
|
|
618
|
+
* mirrors that: stored responses live in the kernel action log and are retrievable/deletable;
|
|
619
|
+
* a store=false response is answered id-less and NOT stored.
|
|
620
|
+
*/
|
|
621
|
+
function responseView(r: Record<string, unknown>): Record<string, unknown> {
|
|
622
|
+
return { id: r.id, ...strip(r) };
|
|
623
|
+
}
|
|
624
|
+
|
|
625
|
+
async function createResponse(params: Record<string, unknown>, req: FireworksRequest): Promise<FireworksResponseEnvelope> {
|
|
626
|
+
if (params.model === undefined) return { status: 422, body: { detail: [{ loc: ['body', 'model'], msg: 'Field required', type: 'missing' }] } };
|
|
627
|
+
if (params.input === undefined) return { status: 422, body: { detail: [{ loc: ['body', 'input'], msg: 'Field required', type: 'missing' }] } };
|
|
628
|
+
if (typeof params.model !== 'string' || !params.model) return { status: 422, body: { detail: [{ loc: ['body', 'model'], msg: 'Input should be a valid string', type: 'string_type' }] } };
|
|
629
|
+
const store = params.store !== false;
|
|
630
|
+
const inputText = typeof params.input === 'string' ? params.input : JSON.stringify(params.input);
|
|
631
|
+
const promptTokens = estimateTokens(inputText);
|
|
632
|
+
const text = stubAssistantText(String(params.model), inputText);
|
|
633
|
+
const completionTokens = estimateTokens(text);
|
|
634
|
+
const outputItem = {
|
|
635
|
+
type: 'message' as const,
|
|
636
|
+
id: `msg_twin_${stableSuffix(text)}`,
|
|
637
|
+
role: 'assistant' as const,
|
|
638
|
+
status: 'completed' as const,
|
|
639
|
+
content: [{ type: 'output_text' as const, text }],
|
|
640
|
+
};
|
|
641
|
+
const base: Record<string, unknown> = {
|
|
642
|
+
object: 'response',
|
|
643
|
+
created_at: nowEpoch(req.occurredAt),
|
|
644
|
+
status: 'completed',
|
|
645
|
+
model: params.model,
|
|
646
|
+
output: [outputItem],
|
|
647
|
+
usage: { prompt_tokens: promptTokens, total_tokens: promptTokens + completionTokens, completion_tokens: completionTokens },
|
|
648
|
+
...(params.instructions !== undefined ? { instructions: params.instructions } : {}),
|
|
649
|
+
...(params.metadata !== undefined ? { metadata: params.metadata } : {}),
|
|
650
|
+
...(params.previous_response_id !== undefined ? { previous_response_id: params.previous_response_id } : {}),
|
|
651
|
+
...(params.temperature !== undefined ? { temperature: params.temperature } : {}),
|
|
652
|
+
...(params.max_output_tokens !== undefined ? { max_output_tokens: params.max_output_tokens } : {}),
|
|
653
|
+
store,
|
|
654
|
+
};
|
|
655
|
+
if (!store) {
|
|
656
|
+
// The vendor's own contract: id is null when store=false, and nothing is retrievable later.
|
|
657
|
+
return { status: 200, body: { ...base, id: null } };
|
|
658
|
+
}
|
|
659
|
+
const id = `resp_twin_${stableSuffix(inputText + String(params.model))}`;
|
|
660
|
+
await applyTwinWrite(SERVICE, {
|
|
661
|
+
operation: 'response.create',
|
|
662
|
+
subjectType: 'response',
|
|
663
|
+
subjectId: id,
|
|
664
|
+
fields: base,
|
|
665
|
+
...(req.occurredAt ? { occurredAt: req.occurredAt } : {}),
|
|
666
|
+
actor: { kind: 'agent' },
|
|
667
|
+
}, req.root);
|
|
668
|
+
return { status: 200, body: responseView(getRow('response', id, undefined, req.root) ?? { ...base, id }) };
|
|
669
|
+
}
|
|
670
|
+
|
|
671
|
+
function handleListResponses(req: FireworksRequest): FireworksResponseEnvelope {
|
|
672
|
+
const url = new URL(req.path, 'http://twin');
|
|
673
|
+
const limit = Number(url.searchParams.get('limit') ?? 20);
|
|
674
|
+
const data = rows('response', undefined, req.root).filter((r) => !isTombstoned(r)).map(responseView);
|
|
675
|
+
const page = data.slice(0, Number.isFinite(limit) && limit > 0 ? limit : 20);
|
|
676
|
+
return { status: 200, body: { object: 'list', data: page, has_more: data.length > page.length, first_id: page[0]?.id ?? null, last_id: page[page.length - 1]?.id ?? null } };
|
|
677
|
+
}
|
|
678
|
+
|
|
679
|
+
function handleGetResponse(id: string, req: FireworksRequest): FireworksResponseEnvelope {
|
|
680
|
+
const r = getRow('response', id, undefined, req.root);
|
|
681
|
+
if (!r || isTombstoned(r)) return notFound(`No response found with id '${id}'.`);
|
|
682
|
+
return { status: 200, body: responseView(r) };
|
|
683
|
+
}
|
|
684
|
+
|
|
685
|
+
async function handleDeleteResponse(id: string, req: FireworksRequest): Promise<FireworksResponseEnvelope> {
|
|
686
|
+
const r = getRow('response', id, undefined, req.root);
|
|
687
|
+
if (!r || isTombstoned(r)) return notFound(`No response found with id '${id}'.`);
|
|
688
|
+
await applyTwinWrite(SERVICE, {
|
|
689
|
+
operation: 'response.delete', subjectType: 'response', subjectId: id, fields: { _deleted: true },
|
|
690
|
+
...(req.occurredAt ? { occurredAt: req.occurredAt } : {}), actor: { kind: 'agent' },
|
|
691
|
+
}, req.root);
|
|
692
|
+
return { status: 200, body: { id, object: 'response', deleted: true } };
|
|
693
|
+
}
|
|
694
|
+
|
|
695
|
+
// ── Anthropic-compatible /v1/messages ───────────────────────────────────────────────────
|
|
696
|
+
/** The Anthropic error envelope — a DIFFERENT shape from the OpenAI-compat plane's. */
|
|
697
|
+
function anthropicError(status: number, type: FireworksAnthropicErrorType, message: string): FireworksResponseEnvelope {
|
|
698
|
+
return { status, body: { type: 'error', error: { type, message }, request_id: null } };
|
|
699
|
+
}
|
|
700
|
+
|
|
701
|
+
function validateAnthropicMessages(params: Record<string, unknown>): { ok: true } | { error: FireworksResponseEnvelope } {
|
|
702
|
+
if (params.model === undefined) return { error: anthropicError(400, 'invalid_request_error', 'model: Field required') };
|
|
703
|
+
if (!Array.isArray(params.messages) || params.messages.length === 0) {
|
|
704
|
+
return { error: anthropicError(400, 'invalid_request_error', 'messages: Field required') };
|
|
705
|
+
}
|
|
706
|
+
for (const m of params.messages as Array<Record<string, unknown>>) {
|
|
707
|
+
if (!m || typeof m !== 'object' || (m.role !== 'user' && m.role !== 'assistant')) {
|
|
708
|
+
return { error: anthropicError(400, 'invalid_request_error', "messages: each message must have role 'user' or 'assistant'") };
|
|
709
|
+
}
|
|
710
|
+
}
|
|
711
|
+
return { ok: true };
|
|
712
|
+
}
|
|
713
|
+
|
|
714
|
+
function buildAnthropicMessage(params: Record<string, unknown>, occurredAt?: string): FireworksResponseEnvelope {
|
|
715
|
+
const bad = validateAnthropicMessages(params);
|
|
716
|
+
if ('error' in bad) return bad.error;
|
|
717
|
+
const messages = params.messages as Array<{ role: string; content: unknown }>;
|
|
718
|
+
const promptText = messages.map((m) => contentToText(m.content)).join('\n');
|
|
719
|
+
const systemText = typeof params.system === 'string' ? params.system : Array.isArray(params.system) ? params.system.map((b) => contentToText((b as { text?: unknown }).text)).join('\n') : '';
|
|
720
|
+
const inputTokens = estimateTokens(promptText + systemText);
|
|
721
|
+
// max_tokens is OPTIONAL on Fireworks (required on Anthropic — the documented difference).
|
|
722
|
+
const maxTokens = typeof params.max_tokens === 'number' ? params.max_tokens : undefined;
|
|
723
|
+
let text = stubAssistantText(String(params.model), promptText.trim() || systemText);
|
|
724
|
+
let stopReason: FireworksAnthropicMessage['stop_reason'] = 'end_turn';
|
|
725
|
+
if (maxTokens !== undefined && estimateTokens(text) > maxTokens) {
|
|
726
|
+
text = text.slice(0, maxTokens * 4);
|
|
727
|
+
stopReason = 'max_tokens';
|
|
728
|
+
}
|
|
729
|
+
const content: FireworksAnthropicContentBlock[] = [{ type: 'text', text, citations: null }];
|
|
730
|
+
const outputTokens = estimateTokens(text);
|
|
731
|
+
const body: FireworksAnthropicMessage = {
|
|
732
|
+
id: `msg_twin_${stableSuffix(promptText + String(params.model))}`,
|
|
733
|
+
type: 'message',
|
|
734
|
+
role: 'assistant',
|
|
735
|
+
content,
|
|
736
|
+
model: String(params.model),
|
|
737
|
+
stop_reason: stopReason,
|
|
738
|
+
stop_sequence: null,
|
|
739
|
+
// Usage is included in BOTH streaming and non-streaming responses (the documented difference
|
|
740
|
+
// from Anthropic, where streaming omits it until the final delta).
|
|
741
|
+
usage: { input_tokens: inputTokens, output_tokens: outputTokens },
|
|
742
|
+
};
|
|
743
|
+
return { status: 200, body };
|
|
744
|
+
}
|
|
745
|
+
|
|
746
|
+
/** The Anthropic SSE sequence: message_start → content_block_start → content_block_delta* →
|
|
747
|
+
* content_block_stop → message_delta (carrying the ACTUAL usage — the one message_delta per
|
|
748
|
+
* stream) → message_stop. */
|
|
749
|
+
function streamAnthropicMessage(params: Record<string, unknown>, sink: SseSink, occurredAt?: string): FireworksResponseEnvelope {
|
|
750
|
+
const built = buildAnthropicMessage(params, occurredAt);
|
|
751
|
+
if (built.status !== 200) return built;
|
|
752
|
+
const full = built.body as FireworksAnthropicMessage;
|
|
753
|
+
const content = full.content;
|
|
754
|
+
const usage = full.usage ?? { input_tokens: 0, output_tokens: 0 };
|
|
755
|
+
sink({ data: { type: 'message_start', message: { ...full, content: [], stop_reason: null, usage: { input_tokens: usage.input_tokens, output_tokens: 0 } } } });
|
|
756
|
+
sink({ data: { type: 'content_block_start', index: 0, content_block: { type: 'text', text: '' } } });
|
|
757
|
+
for (const piece of chunkText(content[0]!.type === 'text' ? content[0]!.text : '')) {
|
|
758
|
+
sink({ data: { type: 'content_block_delta', index: 0, delta: { type: 'text_delta', text: piece } } });
|
|
759
|
+
}
|
|
760
|
+
sink({ data: { type: 'content_block_stop', index: 0 } });
|
|
761
|
+
// ONE message_delta, carrying the ACTUAL token counts (the vendor's own note: the message_start
|
|
762
|
+
// usage is always 0 and should be ignored for metering).
|
|
763
|
+
sink({ data: { type: 'message_delta', delta: { stop_reason: full.stop_reason, stop_sequence: null }, usage: { input_tokens: usage.input_tokens, output_tokens: usage.output_tokens } } });
|
|
764
|
+
sink({ data: { type: 'message_stop' } });
|
|
765
|
+
sink({ done: true });
|
|
766
|
+
return built;
|
|
767
|
+
}
|
|
768
|
+
|
|
769
|
+
// ── embeddings + rerank (deterministic pseudo-vectors / scores) ─────────────────────────
|
|
770
|
+
/** The largest vector the twin will ever allocate from a client number. The vendor's resizable
|
|
771
|
+
* embedding models truncate DOWN to the requested `dimensions` (never produce longer vectors
|
|
772
|
+
* than the model's native output); the catalog's largest native dimensionality across the
|
|
773
|
+
* served embedders is the Qwen3 embedding family's 4096. Every array sized from a REQUEST
|
|
774
|
+
* value in this pack is bounded by this constant or by an input length — audited. */
|
|
775
|
+
const MAX_EMBEDDING_DIMENSIONS = 4096;
|
|
776
|
+
|
|
777
|
+
function handleEmbeddings(params: Record<string, unknown>, root?: string): FireworksResponseEnvelope {
|
|
778
|
+
if (params.model === undefined) return { status: 422, body: { detail: [{ loc: ['body', 'model'], msg: 'Field required', type: 'missing' }] } };
|
|
779
|
+
if (params.input === undefined) return { status: 422, body: { detail: [{ loc: ['body', 'input'], msg: 'Field required', type: 'missing' }] } };
|
|
780
|
+
if (typeof params.model !== 'string' || !servedModelIds(root).has(params.model)) return unknownModel(String(params.model));
|
|
781
|
+
const inputs = Array.isArray(params.input) ? (params.input as unknown[]).map((v) => (typeof v === 'string' ? v : JSON.stringify(v))) : [String(params.input)];
|
|
782
|
+
if (inputs.some((s) => s.length === 0)) return invalidRequest("'input' must not be an empty string");
|
|
783
|
+
// `dimensions` is validated BEFORE any allocation: the spec types it `anyOf [integer, null]`
|
|
784
|
+
// (EmbeddingRequest), so pydantic answers a non-integer with int_parsing; the range below is
|
|
785
|
+
// the vendor's own constraint — dimensions must be ≥ 1 (the API reference schema marks
|
|
786
|
+
// minimum 1) and ≤ the model's native dimensionality (the vendor's resizable models return
|
|
787
|
+
// SHORTER vectors — matryoshka truncation — never longer; the catalog's largest is the Qwen3
|
|
788
|
+
// embedding table's 4096). An unvalidated client number flowed straight into the vector
|
|
789
|
+
// allocation: 1e9 tried to allocate a 12 GB array (one-request DoS), -1/2.5 escaped as a 500,
|
|
790
|
+
// 0 answered a fake 200 with an empty vector, 'x' was silently ignored.
|
|
791
|
+
const dimsRaw = params.dimensions;
|
|
792
|
+
if (dimsRaw !== undefined && dimsRaw !== null) {
|
|
793
|
+
if (typeof dimsRaw !== 'number' || Number.isNaN(dimsRaw)) return validationError(['body', 'dimensions'], 'Input should be a valid integer', 'int_parsing');
|
|
794
|
+
if (!Number.isInteger(dimsRaw)) return validationError(['body', 'dimensions'], 'Input should be a valid integer', 'int_from_float');
|
|
795
|
+
if (dimsRaw < 1) return validationError(['body', 'dimensions'], 'Input should be greater than or equal to 1', 'greater_than_equal');
|
|
796
|
+
if (dimsRaw > MAX_EMBEDDING_DIMENSIONS) return validationError(['body', 'dimensions'], `Input should be less than or equal to ${MAX_EMBEDDING_DIMENSIONS}`, 'less_than_equal');
|
|
797
|
+
}
|
|
798
|
+
const dims = typeof dimsRaw === 'number' ? dimsRaw : 768;
|
|
799
|
+
const encoding = params.encoding_format === undefined ? 'float' : params.encoding_format;
|
|
800
|
+
if (encoding !== 'float' && encoding !== 'base64') return invalidRequest("'encoding_format' must be one of 'float', 'base64'");
|
|
801
|
+
let promptTokens = 0;
|
|
802
|
+
const data = inputs.map((text, index) => {
|
|
803
|
+
promptTokens += estimateTokens(text);
|
|
804
|
+
const vec = pseudoEmbedding(text, dims);
|
|
805
|
+
return {
|
|
806
|
+
object: 'embedding' as const,
|
|
807
|
+
index,
|
|
808
|
+
embedding: encoding === 'base64' ? Buffer.from(new Float32Array(vec).buffer).toString('base64') : vec,
|
|
809
|
+
};
|
|
810
|
+
});
|
|
811
|
+
return {
|
|
812
|
+
status: 200,
|
|
813
|
+
body: {
|
|
814
|
+
object: 'list',
|
|
815
|
+
data,
|
|
816
|
+
model: params.model,
|
|
817
|
+
usage: { prompt_tokens: promptTokens, total_tokens: promptTokens },
|
|
818
|
+
},
|
|
819
|
+
};
|
|
820
|
+
}
|
|
821
|
+
|
|
822
|
+
function handleRerank(params: Record<string, unknown>, root?: string): FireworksResponseEnvelope {
|
|
823
|
+
if (params.query === undefined) return { status: 422, body: { detail: [{ loc: ['body', 'query'], msg: 'Field required', type: 'missing' }] } };
|
|
824
|
+
if (!Array.isArray(params.documents) || params.documents.length === 0) return { status: 422, body: { detail: [{ loc: ['body', 'documents'], msg: 'Field required', type: 'missing' }] } };
|
|
825
|
+
// The vendor's rerank takes a model too (the Qwen3 Reranker family); an unknown id is refused
|
|
826
|
+
// the same way chat/embeddings refuse theirs. A MISSING model is accepted (the twin's own
|
|
827
|
+
// query/document scorer needs no id; the vendor's request schema marks model optional there).
|
|
828
|
+
if (params.model !== undefined && (typeof params.model !== 'string' || !servedModelIds(root).has(params.model))) return unknownModel(String(params.model));
|
|
829
|
+
const query = String(params.query);
|
|
830
|
+
const documents = (params.documents as unknown[]).map(String);
|
|
831
|
+
const returnDocuments = params.return_documents !== false;
|
|
832
|
+
const scored = documents.map((doc, index) => ({ index, score: stubRelevanceScore(query, doc), doc }));
|
|
833
|
+
// Ordered by relevance score (highest first) — the vendor's own contract for the data array.
|
|
834
|
+
scored.sort((a, b) => b.score - a.score);
|
|
835
|
+
// `top_n` is the spec's integer (RerankRequestBody); a non-integer is refused before it can
|
|
836
|
+
// slice, and a value below 1 would answer an EMPTY result list for a valid request — the
|
|
837
|
+
// vendor's own contract keeps at least one result.
|
|
838
|
+
const topRaw = params.top_n;
|
|
839
|
+
if (topRaw !== undefined && topRaw !== null) {
|
|
840
|
+
if (typeof topRaw !== 'number' || Number.isNaN(topRaw)) return validationError(['body', 'top_n'], 'Input should be a valid integer', 'int_parsing');
|
|
841
|
+
if (!Number.isInteger(topRaw)) return validationError(['body', 'top_n'], 'Input should be a valid integer', 'int_from_float');
|
|
842
|
+
if (topRaw < 1) return validationError(['body', 'top_n'], 'Input should be greater than or equal to 1', 'greater_than_equal');
|
|
843
|
+
}
|
|
844
|
+
const topN = typeof topRaw === 'number' ? topRaw : documents.length;
|
|
845
|
+
const data = scored.slice(0, Math.max(0, topN)).map((s) => ({
|
|
846
|
+
index: s.index,
|
|
847
|
+
relevance_score: s.score,
|
|
848
|
+
...(returnDocuments ? { document: s.doc } : {}),
|
|
849
|
+
}));
|
|
850
|
+
const promptTokens = estimateTokens(query) + documents.reduce((sum, d) => sum + estimateTokens(d), 0);
|
|
851
|
+
return {
|
|
852
|
+
status: 200,
|
|
853
|
+
body: {
|
|
854
|
+
object: 'list',
|
|
855
|
+
model: params.model ?? null,
|
|
856
|
+
data,
|
|
857
|
+
usage: { prompt_tokens: promptTokens, total_tokens: promptTokens },
|
|
858
|
+
},
|
|
859
|
+
};
|
|
860
|
+
}
|
|
861
|
+
|
|
862
|
+
// ── CONTROL PLANE (Gateway REST API) ───────────────────────────────────────────────────
|
|
863
|
+
// google.rpc-style resources: `name` fields follow accounts/<account>/…, `state`/`status` are
|
|
864
|
+
// vendor enums, list envelopes carry nextPageToken/totalSize, and CREATE ids arrive as QUERY
|
|
865
|
+
// params (deployments/datasets/users) or in the body (datasets carry datasetId in the body too).
|
|
866
|
+
function resourceView(r: Record<string, unknown>): Record<string, unknown> {
|
|
867
|
+
// The vendor's control-plane schemas (gatewayDeployment, gatewayDataset, …) carry NO `id`
|
|
868
|
+
// field — identity is the hierarchical `name`. The bare id stays a twin-internal alias and
|
|
869
|
+
// must never appear on the wire (defect: strip() was letting it through on every body).
|
|
870
|
+
const { id: _drop, ...rest } = r;
|
|
871
|
+
return strip(rest);
|
|
872
|
+
}
|
|
873
|
+
|
|
874
|
+
function paginate(items: Array<Record<string, unknown>>, url: URL): { items: Array<Record<string, unknown>>; nextPageToken: string | null } {
|
|
875
|
+
const pageSize = Number(url.searchParams.get('pageSize') ?? 0);
|
|
876
|
+
const pageToken = url.searchParams.get('pageToken');
|
|
877
|
+
let start = 0;
|
|
878
|
+
if (pageToken) {
|
|
879
|
+
const n = Number(pageToken);
|
|
880
|
+
start = Number.isFinite(n) && n > 0 ? n : 0;
|
|
881
|
+
}
|
|
882
|
+
const size = Number.isFinite(pageSize) && pageSize > 0 ? pageSize : items.length;
|
|
883
|
+
const page = items.slice(start, start + size);
|
|
884
|
+
const next = start + size < items.length ? String(start + size) : null;
|
|
885
|
+
return { items: page, nextPageToken: next };
|
|
886
|
+
}
|
|
887
|
+
|
|
888
|
+
/** The gateway status embedded on resources: OK once READY, else the resource's own state. */
|
|
889
|
+
function okStatus(): Record<string, unknown> {
|
|
890
|
+
return { code: 'OK', message: '' };
|
|
891
|
+
}
|
|
892
|
+
|
|
893
|
+
async function createControlResource(type: string, prefix: string, accountId: string, params: Record<string, unknown>, req: FireworksRequest, url: URL, buildFields: (params: Record<string, unknown>) => Record<string, unknown>): Promise<FireworksResponseEnvelope> {
|
|
894
|
+
// The vendor passes create ids where its spec puts them: QUERY params for deployments
|
|
895
|
+
// (deploymentId), users (userId), batchInferenceJobId and supervisedFineTuningJobId; BODY
|
|
896
|
+
// fields for datasets ({dataset, datasetId}) and models ({model, modelId}); the resource's own
|
|
897
|
+
// `name` for secrets. Absent → the vendor mints one; so does the twin.
|
|
898
|
+
const queryId = url.searchParams.get(`${type}Id`) ?? url.searchParams.get(`${type}_id`);
|
|
899
|
+
const wrapped = (params[type] && typeof params[type] === 'object' ? params[type] : params) as Record<string, unknown>;
|
|
900
|
+
const bodyId = typeof wrapped[`${type}Id`] === 'string' ? (wrapped[`${type}Id`] as string)
|
|
901
|
+
: typeof params[`${type}Id`] === 'string' ? (params[`${type}Id`] as string)
|
|
902
|
+
// A secret's id is the last segment of the `name` the client supplies in its own body.
|
|
903
|
+
: type === 'secret' && typeof params.name === 'string' ? (params.name.split('/').pop() as string)
|
|
904
|
+
: undefined;
|
|
905
|
+
const id = queryId || bodyId || nextId(type, prefix, accountId, req.root);
|
|
906
|
+
// The vendor's own spec marks the id REQUIRED on datasets (CreateDatasetRequest.required:
|
|
907
|
+
// dataset + datasetId) and models (GatewayGatewayCreateModelBody.required: modelId) — a create
|
|
908
|
+
// without one is a 400 at the vendor, never a mint. The dataset body itself is required too
|
|
909
|
+
// (the same required list); a bare {datasetId} is not a create. The other collections' id
|
|
910
|
+
// params are optional (the vendor mints), so the twin keeps minting there.
|
|
911
|
+
if (type === 'dataset' || type === 'model') {
|
|
912
|
+
if (!queryId && !bodyId) return gatewayError(400, `${type}Id is required`);
|
|
913
|
+
if (type === 'dataset' && !(params.dataset && typeof params.dataset === 'object')) {
|
|
914
|
+
return gatewayError(400, 'dataset is required');
|
|
915
|
+
}
|
|
916
|
+
}
|
|
917
|
+
// TENANCY: the kernel subject is the ACCOUNT-NAMESPACED id (`{account}/{id}` — the vendor's own
|
|
918
|
+
// name grammar), so the kernel's (type, subject) key is unique per tenant. Writing the bare id
|
|
919
|
+
// let a create under account B with an id account A held DESTROY A's row (the overlay folds by
|
|
920
|
+
// the bare id globally); the duplicate check below is therefore per-account by construction.
|
|
921
|
+
const subject = `${accountId}/${id}`;
|
|
922
|
+
if (getRow(type, id, accountId, req.root) && !isTombstoned(getRow(type, id, accountId, req.root)!)) {
|
|
923
|
+
return gatewayError(409, `Resource already exists: ${id}`);
|
|
924
|
+
}
|
|
925
|
+
const fields = buildFields(params);
|
|
926
|
+
// The account the create rode under is part of the resource's identity: every later read is
|
|
927
|
+
// scoped to it, so a row created under acct-A is invisible under acct-B's paths (the vendor's
|
|
928
|
+
// own tenancy). The `_` prefix keeps it off the wire (strip() drops it).
|
|
929
|
+
fields._account = accountId;
|
|
930
|
+
// The id is resolved HERE (query param, body field, or the twin's mint) — so the row's `name`
|
|
931
|
+
// is built from it. A caller-supplied buildFields cannot know the minted id; its PLACEHOLDER
|
|
932
|
+
// stand-in must never survive to the wire (the vendor names every row with the real id).
|
|
933
|
+
if (typeof fields.name === 'string') fields.name = fields.name.replace('PLACEHOLDER', id);
|
|
934
|
+
await applyTwinWrite(SERVICE, {
|
|
935
|
+
operation: `${type}.create`,
|
|
936
|
+
subjectType: type,
|
|
937
|
+
subjectId: subject,
|
|
938
|
+
fields,
|
|
939
|
+
...(req.occurredAt ? { occurredAt: req.occurredAt } : {}),
|
|
940
|
+
actor: { kind: 'agent' },
|
|
941
|
+
}, req.root);
|
|
942
|
+
return { status: 200, body: resourceView(getRow(type, id, accountId, req.root) ?? { id: subject, ...fields }) };
|
|
943
|
+
}
|
|
944
|
+
|
|
945
|
+
async function deleteControlResource(type: string, id: string, accountId: string, req: FireworksRequest): Promise<FireworksResponseEnvelope> {
|
|
946
|
+
const r = getRow(type, id, accountId, req.root);
|
|
947
|
+
if (!r || isTombstoned(r)) return gatewayError(404, `Not found: ${type}/${id}`);
|
|
948
|
+
await applyTwinWrite(SERVICE, {
|
|
949
|
+
operation: `${type}.delete`, subjectType: type, subjectId: `${accountId}/${id}`, fields: { _deleted: true },
|
|
950
|
+
...(req.occurredAt ? { occurredAt: req.occurredAt } : {}), actor: { kind: 'agent' },
|
|
951
|
+
}, req.root);
|
|
952
|
+
// The vendor's delete operations answer `{}` (an empty object per its own spec).
|
|
953
|
+
return { status: 200, body: {} };
|
|
954
|
+
}
|
|
955
|
+
|
|
956
|
+
const DEPLOYMENT_STATES = ['STATE_UNSPECIFIED', 'CREATING', 'READY', 'DELETING', 'FAILED', 'UPDATING', 'DELETED'] as const;
|
|
957
|
+
const JOB_STATES = ['JOB_STATE_UNSPECIFIED', 'JOB_STATE_CREATING', 'JOB_STATE_RUNNING', 'JOB_STATE_COMPLETED', 'JOB_STATE_FAILED', 'JOB_STATE_CANCELLED', 'JOB_STATE_DELETING', 'JOB_STATE_WRITING_RESULTS', 'JOB_STATE_VALIDATING', 'JOB_STATE_DELETING_CLEANING_UP', 'JOB_STATE_PENDING', 'JOB_STATE_EXPIRED', 'JOB_STATE_RE_QUEUEING', 'JOB_STATE_CREATING_INPUT_DATASET', 'JOB_STATE_IDLE', 'JOB_STATE_CANCELLING', 'JOB_STATE_EARLY_STOPPED', 'JOB_STATE_PAUSED', 'JOB_STATE_DELETED', 'JOB_STATE_ARCHIVED'] as const;
|
|
958
|
+
const USER_STATES = ['STATE_UNSPECIFIED', 'CREATING', 'READY', 'UPDATING', 'DELETING'] as const;
|
|
959
|
+
const USER_ROLES = ['admin', 'user', 'contributor', 'inference-user', 'custom'] as const;
|
|
960
|
+
|
|
961
|
+
// ── public entry: cross-cutting protocol (auth) then route ──────────────────────────────
|
|
962
|
+
export async function handleFireworksTwinRequest(req: FireworksRequest): Promise<FireworksResponseEnvelope> {
|
|
963
|
+
const method = req.method.toUpperCase();
|
|
964
|
+
if (req.headers !== undefined || req.apiKey !== undefined) {
|
|
965
|
+
const authErr = checkAuth(req);
|
|
966
|
+
if (authErr) return authErr;
|
|
967
|
+
}
|
|
968
|
+
return routeFireworks(req, method);
|
|
969
|
+
}
|
|
970
|
+
|
|
971
|
+
// ── router ──────────────────────────────────────────────────────────────────────────────
|
|
972
|
+
async function routeFireworks(req: FireworksRequest, method: string): Promise<FireworksResponseEnvelope> {
|
|
973
|
+
const path = (req.path.split('?')[0] ?? '/').replace(/\/+$/, '') || '/';
|
|
974
|
+
const query = req.path.includes('?') ? req.path.slice(req.path.indexOf('?')) : '';
|
|
975
|
+
const params = parseJson(req.body);
|
|
976
|
+
const dec = (s: string) => decodeURIComponent(s);
|
|
977
|
+
|
|
978
|
+
// D3: a read-only twin rejects any mutation with a vendor-shaped error — each plane's OWN
|
|
979
|
+
// envelope (the control plane's google.rpc shape, the inference plane's OpenAI shape).
|
|
980
|
+
if (req.readOnly && method !== 'GET') {
|
|
981
|
+
const message = 'twin is read-only; omit readOnly to accept writes';
|
|
982
|
+
if (path.startsWith(`${FIREWORKS_ACCOUNTS_PREFIX}/`)) return gatewayError(405, message);
|
|
983
|
+
return { status: 405, body: errBody(message, { code: 405 }) };
|
|
984
|
+
}
|
|
985
|
+
|
|
986
|
+
const url = new URL(req.path, 'http://twin');
|
|
987
|
+
|
|
988
|
+
// ── INFERENCE PLANE (/inference/v1/…) ──────────────────────────────────────────────────
|
|
989
|
+
if (path === FIREWORKS_INFERENCE_PREFIX || path.startsWith(`${FIREWORKS_INFERENCE_PREFIX}/`)) {
|
|
990
|
+
const seg = path.slice(FIREWORKS_INFERENCE_PREFIX.length).replace(/^\/+/, '').split('/').filter(Boolean);
|
|
991
|
+
// The Anthropic-compat surface has its OWN envelope family (AnthropicErrorResponse), so it
|
|
992
|
+
// routes separately from the OpenAI-compat operations.
|
|
993
|
+
if (seg[0] === 'messages' && seg.length === 1 && method === 'POST') {
|
|
994
|
+
if (params.stream === true) {
|
|
995
|
+
if (!req.sseSink) return anthropicError(400, 'invalid_request_error', 'streaming requires an SSE-capable connection');
|
|
996
|
+
return streamAnthropicMessage(params, req.sseSink, req.occurredAt);
|
|
997
|
+
}
|
|
998
|
+
return buildAnthropicMessage(params, req.occurredAt);
|
|
999
|
+
}
|
|
1000
|
+
if (seg[0] === 'chat' && seg[1] === 'completions' && seg.length === 2 && method === 'POST') {
|
|
1001
|
+
const validated = validateChat(params);
|
|
1002
|
+
if ('error' in validated) return validated.error;
|
|
1003
|
+
const args = validated.args;
|
|
1004
|
+
// The vendor refuses an unknown model id BEFORE any generation (404 "Model id not found");
|
|
1005
|
+
// an account-owned model (created/pulled through the control plane) is addressable too.
|
|
1006
|
+
if (!servedModelIds(req.root).has(args.model)) return unknownModel(args.model);
|
|
1007
|
+
// R15 — the scenario engine decides AND honors a fault here, before any completion exists:
|
|
1008
|
+
// a `status` fault is this vendor's own refusal envelope (rate limit, server error), a
|
|
1009
|
+
// `slow` has already held the answer, a `drop` never returns. The realizers below only see
|
|
1010
|
+
// a content decision.
|
|
1011
|
+
let decision: ScenarioDecision | undefined;
|
|
1012
|
+
if (req.scenarioEngine) {
|
|
1013
|
+
const served = await req.scenarioEngine.serve({ model: args.model, messages: args.messages, tools: args.tools, serviceTier: args.serviceTier });
|
|
1014
|
+
if (served.kind === 'fault') return { status: served.result.status, body: served.result.body, headers: served.result.headers };
|
|
1015
|
+
decision = served;
|
|
1016
|
+
}
|
|
1017
|
+
const result = args.stream && req.sseSink ? streamChat(args, req.sseSink, req.occurredAt, decision) : buildChatCompletion(args, req.occurredAt, decision);
|
|
1018
|
+
if ('status' in result) return result;
|
|
1019
|
+
return { status: 200, body: result };
|
|
1020
|
+
}
|
|
1021
|
+
if (seg[0] === 'completions' && seg.length === 1 && method === 'POST') return handleCompletion(params, req.occurredAt, req.sseSink, req.root);
|
|
1022
|
+
if (seg[0] === 'responses' && seg.length === 1 && method === 'POST') return createResponse(params, req);
|
|
1023
|
+
if (seg[0] === 'responses' && seg.length === 1 && method === 'GET') return handleListResponses(req);
|
|
1024
|
+
if (seg[0] === 'responses' && seg.length === 2 && method === 'GET') return handleGetResponse(dec(seg[1]!), req);
|
|
1025
|
+
if (seg[0] === 'responses' && seg.length === 2 && method === 'DELETE') return handleDeleteResponse(dec(seg[1]!), req);
|
|
1026
|
+
if (seg[0] === 'embeddings' && seg.length === 1 && method === 'POST') return handleEmbeddings(params, req.root);
|
|
1027
|
+
if (seg[0] === 'rerank' && seg.length === 1 && method === 'POST') return handleRerank(params, req.root);
|
|
1028
|
+
return notFound(`Unknown request URL: ${method} ${path}.`);
|
|
1029
|
+
}
|
|
1030
|
+
|
|
1031
|
+
// ── CONTROL PLANE (/v1/accounts/{account_id}/…) ────────────────────────────────────────
|
|
1032
|
+
if (path.startsWith(`${FIREWORKS_ACCOUNTS_PREFIX}/`)) {
|
|
1033
|
+
const seg = path.slice(FIREWORKS_ACCOUNTS_PREFIX.length).split('/').filter(Boolean).map(dec);
|
|
1034
|
+
const accountId = seg[0];
|
|
1035
|
+
if (!accountId) return gatewayError(404, `Unknown request URL: ${method} ${path}.`);
|
|
1036
|
+
const rest = seg.slice(1);
|
|
1037
|
+
// The account itself: GET /v1/accounts/{account_id}. The twin holds NO account rows — an
|
|
1038
|
+
// account exists only as the path prefix its resources live under, and the vendor's account
|
|
1039
|
+
// row is one-per-credential vendor state the API cannot create. Fabricating a 200 row for
|
|
1040
|
+
// ANY id (the old behavior) was a fake success; the vendor answers 404 for an account id it
|
|
1041
|
+
// does not know, and that is what the twin answers too. The honest gap is filed as
|
|
1042
|
+
// fireworks.control.get_account / fireworks.account.read (todos).
|
|
1043
|
+
if (rest.length === 0 && method === 'GET') {
|
|
1044
|
+
return gatewayError(404, `Not found: accounts/${accountId}`);
|
|
1045
|
+
}
|
|
1046
|
+
const resource = rest[0];
|
|
1047
|
+
// Verb-suffixed custom methods: `:cancel`, `:resume`, `:promote`, … (the vendor's AIP-158
|
|
1048
|
+
// style). The colon rides EITHER the collection segment (`jobs:cancel` — collection-level
|
|
1049
|
+
// verbs) or the id segment (`conf-job:cancel` — resource-level verbs); the id is the segment
|
|
1050
|
+
// before the verb.
|
|
1051
|
+
const colonAt = resource?.indexOf(':') ?? -1;
|
|
1052
|
+
let verb = colonAt >= 0 ? resource!.slice(colonAt + 1) : undefined;
|
|
1053
|
+
const baseResource = colonAt >= 0 ? resource!.slice(0, colonAt) : resource;
|
|
1054
|
+
let idSeg = rest[1];
|
|
1055
|
+
if (verb === undefined && rest.length >= 2 && rest[1]!.includes(':')) {
|
|
1056
|
+
const at = rest[1]!.indexOf(':');
|
|
1057
|
+
verb = rest[1]!.slice(at + 1);
|
|
1058
|
+
idSeg = rest[1]!.slice(0, at);
|
|
1059
|
+
}
|
|
1060
|
+
|
|
1061
|
+
const TYPE_MAP: Record<string, { plural: string; states: readonly string[]; stateKey: string }> = {
|
|
1062
|
+
deployments: { plural: 'deployments', states: DEPLOYMENT_STATES, stateKey: 'state' },
|
|
1063
|
+
datasets: { plural: 'datasets', states: ['STATE_UNSPECIFIED', 'UPLOADING', 'READY'], stateKey: 'state' },
|
|
1064
|
+
batchInferenceJobs: { plural: 'batchInferenceJobs', states: JOB_STATES, stateKey: 'state' },
|
|
1065
|
+
supervisedFineTuningJobs: { plural: 'supervisedFineTuningJobs', states: JOB_STATES, stateKey: 'state' },
|
|
1066
|
+
users: { plural: 'users', states: USER_STATES, stateKey: 'state' },
|
|
1067
|
+
models: { plural: 'models', states: DEPLOYMENT_STATES, stateKey: 'state' },
|
|
1068
|
+
secrets: { plural: 'secrets', states: DEPLOYMENT_STATES, stateKey: 'state' },
|
|
1069
|
+
};
|
|
1070
|
+
const meta = baseResource ? TYPE_MAP[baseResource] : undefined;
|
|
1071
|
+
if (!meta) return gatewayError(404, `Unknown request URL: ${method} ${path}.`);
|
|
1072
|
+
// The PATCH whitelist, per collection: exactly the MUTABLE fields the vendor's own update
|
|
1073
|
+
// operations accept (create-only and output-only fields are absent — a PATCH carrying one is
|
|
1074
|
+
// refused below, the way grpc-gateway refuses an unknown field; googleads-twin.ts is the
|
|
1075
|
+
// estate precedent). Sourced from the pinned spec's own update-operation schemas:
|
|
1076
|
+
// Gateway_UpdateDataset (displayName, exampleCount, userUploaded, evaluationResult,
|
|
1077
|
+
// transformed, splitted, evalProtocol, externalUrl, format, sourceJobName), the deployments
|
|
1078
|
+
// PATCH names the deployment fields, users PATCH names the user fields, secrets and models
|
|
1079
|
+
// the gatewaySecret/gatewayModel fields. The job collections carry NO update op in the spec
|
|
1080
|
+
// (delete+get only) — a PATCH there is the vendor's 404 (route not found), never a served
|
|
1081
|
+
// no-op, so the collections are absent from this map entirely.
|
|
1082
|
+
const PATCHABLE: Record<string, string[]> = {
|
|
1083
|
+
deployments: ['displayName', 'region', 'replicaCount', 'minReplicaCount', 'maxReplicaCount', 'precision'],
|
|
1084
|
+
users: ['displayName', 'email', 'role', 'permissionPreset'],
|
|
1085
|
+
models: ['displayName', 'contextLength', 'description'],
|
|
1086
|
+
secrets: ['keyName'],
|
|
1087
|
+
datasets: ['displayName', 'exampleCount', 'userUploaded', 'evaluationResult', 'transformed', 'splitted', 'evalProtocol', 'externalUrl', 'format', 'sourceJobName'],
|
|
1088
|
+
};
|
|
1089
|
+
// Rows are stored under the SINGULAR type (the same noun createControlResource writes); the
|
|
1090
|
+
// URL carries the plural collection.
|
|
1091
|
+
const singular = baseResource === 'deployments' ? 'deployment' : baseResource === 'datasets' ? 'dataset' : baseResource === 'batchInferenceJobs' ? 'batchInferenceJob' : baseResource === 'supervisedFineTuningJobs' ? 'supervisedFineTuningJob' : baseResource === 'users' ? 'user' : baseResource === 'models' ? 'model' : 'secret';
|
|
1092
|
+
|
|
1093
|
+
// LIST: GET /v1/accounts/{id}/<plural>
|
|
1094
|
+
if (rest.length === 1 && method === 'GET') {
|
|
1095
|
+
const all = rows(singular, accountId, req.root).filter((r) => !isTombstoned(r)).map(resourceView);
|
|
1096
|
+
const { items, nextPageToken } = paginate(all, url);
|
|
1097
|
+
return { status: 200, body: { [meta.plural]: items, nextPageToken, totalSize: all.length } };
|
|
1098
|
+
}
|
|
1099
|
+
// CREATE: POST /v1/accounts/{id}/<plural> — and ONLY that. A verb-suffixed collection
|
|
1100
|
+
// (`<plural>:estimateCost`, `apiKeys:<anything>`) is a custom method, not a create: the
|
|
1101
|
+
// vendor answers it with its own 404 (or serves it), never by running the create builder —
|
|
1102
|
+
// letting `POST …/supervisedFineTuningJobs:estimateCost` with a dataset body CREATE a job
|
|
1103
|
+
// would make an estimate mutate state.
|
|
1104
|
+
if (rest.length === 1 && method === 'POST' && !verb) {
|
|
1105
|
+
const build = (p: Record<string, unknown>): Record<string, unknown> => {
|
|
1106
|
+
// The vendor's create bodies are the RESOURCE DIRECTLY for deployments,
|
|
1107
|
+
// batchInferenceJobs, supervisedFineTuningJobs, users and secrets (requestBody →
|
|
1108
|
+
// $ref gatewayX). Datasets and models wrap: {dataset, datasetId} / {model, modelId}.
|
|
1109
|
+
const inner = (baseResource === 'datasets' && p.dataset && typeof p.dataset === 'object' ? p.dataset
|
|
1110
|
+
: baseResource === 'models' && p.model && typeof p.model === 'object' ? p.model
|
|
1111
|
+
: p) as Record<string, unknown>;
|
|
1112
|
+
const at = new Date(req.occurredAt ?? '1970-01-01T00:00:00Z').toISOString().replace(/\.\d{3}Z$/, 'Z');
|
|
1113
|
+
if (baseResource === 'deployments') {
|
|
1114
|
+
if (inner.baseModel === undefined) return { __error: 'baseModel is required' } as Record<string, unknown>;
|
|
1115
|
+
return {
|
|
1116
|
+
name: `accounts/${accountId}/deployments/PLACEHOLDER`,
|
|
1117
|
+
displayName: inner.displayName ?? '',
|
|
1118
|
+
baseModel: inner.baseModel,
|
|
1119
|
+
state: 'CREATING',
|
|
1120
|
+
status: okStatus(),
|
|
1121
|
+
createTime: at,
|
|
1122
|
+
updateTime: at,
|
|
1123
|
+
...(inner.region !== undefined ? { region: inner.region } : {}),
|
|
1124
|
+
...(inner.replicaCount !== undefined ? { replicaCount: inner.replicaCount } : {}),
|
|
1125
|
+
...(inner.minReplicaCount !== undefined ? { minReplicaCount: inner.minReplicaCount } : {}),
|
|
1126
|
+
...(inner.maxReplicaCount !== undefined ? { maxReplicaCount: inner.maxReplicaCount } : {}),
|
|
1127
|
+
...(inner.precision !== undefined ? { precision: inner.precision } : {}),
|
|
1128
|
+
};
|
|
1129
|
+
}
|
|
1130
|
+
if (baseResource === 'datasets') {
|
|
1131
|
+
return {
|
|
1132
|
+
name: `accounts/${accountId}/datasets/PLACEHOLDER`,
|
|
1133
|
+
displayName: inner.displayName ?? '',
|
|
1134
|
+
state: 'UPLOADING',
|
|
1135
|
+
status: okStatus(),
|
|
1136
|
+
createTime: at,
|
|
1137
|
+
updateTime: at,
|
|
1138
|
+
exampleCount: 0,
|
|
1139
|
+
userUploaded: true,
|
|
1140
|
+
...(inner.format !== undefined ? { format: inner.format } : {}),
|
|
1141
|
+
};
|
|
1142
|
+
}
|
|
1143
|
+
if (baseResource === 'batchInferenceJobs') {
|
|
1144
|
+
// The job's dataset reference must name a dataset that really landed: the vendor
|
|
1145
|
+
// validates the reference at create, and a twin that accepted any string would let a
|
|
1146
|
+
// mistyped id "create" a job over nothing.
|
|
1147
|
+
if (inner.inputDatasetId !== undefined) {
|
|
1148
|
+
const dsId = String(inner.inputDatasetId).split('/').pop() ?? '';
|
|
1149
|
+
const ds = getRow('dataset', dsId, accountId, req.root);
|
|
1150
|
+
if (!ds || isTombstoned(ds)) return { __error: `inputDatasetId not found: ${String(inner.inputDatasetId)}` } as Record<string, unknown>;
|
|
1151
|
+
}
|
|
1152
|
+
return {
|
|
1153
|
+
name: `accounts/${accountId}/batchInferenceJobs/PLACEHOLDER`,
|
|
1154
|
+
displayName: inner.displayName ?? '',
|
|
1155
|
+
state: 'JOB_STATE_CREATING',
|
|
1156
|
+
status: okStatus(),
|
|
1157
|
+
createTime: at,
|
|
1158
|
+
updateTime: at,
|
|
1159
|
+
...(inner.model !== undefined ? { model: inner.model } : {}),
|
|
1160
|
+
...(inner.inputDatasetId !== undefined ? { inputDatasetId: inner.inputDatasetId } : {}),
|
|
1161
|
+
...(inner.outputDatasetId !== undefined ? { outputDatasetId: inner.outputDatasetId } : {}),
|
|
1162
|
+
};
|
|
1163
|
+
}
|
|
1164
|
+
if (baseResource === 'supervisedFineTuningJobs') {
|
|
1165
|
+
if (inner.dataset === undefined) return { __error: 'dataset is required' } as Record<string, unknown>;
|
|
1166
|
+
const dsId = String(inner.dataset).split('/').pop() ?? '';
|
|
1167
|
+
const ds = getRow('dataset', dsId, accountId, req.root);
|
|
1168
|
+
if (!ds || isTombstoned(ds)) return { __error: `dataset not found: ${String(inner.dataset)}` } as Record<string, unknown>;
|
|
1169
|
+
return {
|
|
1170
|
+
name: `accounts/${accountId}/supervisedFineTuningJobs/PLACEHOLDER`,
|
|
1171
|
+
displayName: inner.displayName ?? '',
|
|
1172
|
+
dataset: inner.dataset,
|
|
1173
|
+
state: 'JOB_STATE_CREATING',
|
|
1174
|
+
status: okStatus(),
|
|
1175
|
+
createTime: at,
|
|
1176
|
+
updateTime: at,
|
|
1177
|
+
...(inner.baseModel !== undefined ? { baseModel: inner.baseModel } : {}),
|
|
1178
|
+
...(inner.epochs !== undefined ? { epochs: inner.epochs } : {}),
|
|
1179
|
+
...(inner.learningRate !== undefined ? { learningRate: inner.learningRate } : {}),
|
|
1180
|
+
};
|
|
1181
|
+
}
|
|
1182
|
+
if (baseResource === 'users') {
|
|
1183
|
+
if (inner.role === undefined) return { __error: 'role is required' } as Record<string, unknown>;
|
|
1184
|
+
if (!USER_ROLES.includes(inner.role as (typeof USER_ROLES)[number])) return { __error: `role must be one of ${USER_ROLES.join(', ')}` } as Record<string, unknown>;
|
|
1185
|
+
return {
|
|
1186
|
+
name: `accounts/${accountId}/users/PLACEHOLDER`,
|
|
1187
|
+
displayName: inner.displayName ?? '',
|
|
1188
|
+
email: inner.email ?? null,
|
|
1189
|
+
role: inner.role,
|
|
1190
|
+
state: 'CREATING',
|
|
1191
|
+
status: okStatus(),
|
|
1192
|
+
createTime: at,
|
|
1193
|
+
updateTime: at,
|
|
1194
|
+
...(inner.permissionPreset !== undefined ? { permissionPreset: inner.permissionPreset } : {}),
|
|
1195
|
+
};
|
|
1196
|
+
}
|
|
1197
|
+
if (baseResource === 'secrets') {
|
|
1198
|
+
// The create body IS the gatewaySecret (the spec takes the resource directly), with
|
|
1199
|
+
// `name` and `keyName` required. `name` is the FULL resource name — a bare last
|
|
1200
|
+
// segment is accepted (clients generated from the spec send the short spelling) and
|
|
1201
|
+
// stored canonically; a full name must not be double-prefixed.
|
|
1202
|
+
if (inner.name === undefined || inner.keyName === undefined) return { __error: 'name and keyName are required' } as Record<string, unknown>;
|
|
1203
|
+
const shortName = String(inner.name).split('/').pop() as string;
|
|
1204
|
+
return {
|
|
1205
|
+
name: `accounts/${accountId}/secrets/${shortName}`,
|
|
1206
|
+
keyName: inner.keyName,
|
|
1207
|
+
// The vendor never returns the secret value on ANY read, including the create
|
|
1208
|
+
// response. The twin keeps it out of the stored row entirely — a secret that can be
|
|
1209
|
+
// read back is a custody lie.
|
|
1210
|
+
createTime: at,
|
|
1211
|
+
updateTime: at,
|
|
1212
|
+
};
|
|
1213
|
+
}
|
|
1214
|
+
if (baseResource === 'models') {
|
|
1215
|
+
return {
|
|
1216
|
+
name: `accounts/${accountId}/models/PLACEHOLDER`,
|
|
1217
|
+
displayName: inner.displayName ?? '',
|
|
1218
|
+
state: 'CREATING',
|
|
1219
|
+
status: okStatus(),
|
|
1220
|
+
createTime: at,
|
|
1221
|
+
updateTime: at,
|
|
1222
|
+
public: false,
|
|
1223
|
+
...(inner.baseModelDetails !== undefined ? { baseModelDetails: inner.baseModelDetails } : {}),
|
|
1224
|
+
...(inner.contextLength !== undefined ? { contextLength: inner.contextLength } : {}),
|
|
1225
|
+
...(inner.description !== undefined ? { description: inner.description } : {}),
|
|
1226
|
+
};
|
|
1227
|
+
}
|
|
1228
|
+
return {};
|
|
1229
|
+
};
|
|
1230
|
+
const fields = build(params);
|
|
1231
|
+
if (fields.__error !== undefined) return gatewayError(400, String(fields.__error));
|
|
1232
|
+
// Replace the PLACEHOLDER name with the real id after the id is known. The singular type
|
|
1233
|
+
// name is what the id params are built from (deploymentId, userId, batchInferenceJobId…).
|
|
1234
|
+
const type = baseResource === 'deployments' ? 'deployment' : baseResource === 'datasets' ? 'dataset' : baseResource === 'batchInferenceJobs' ? 'batchInferenceJob' : baseResource === 'supervisedFineTuningJobs' ? 'supervisedFineTuningJob' : baseResource === 'users' ? 'user' : baseResource === 'models' ? 'model' : 'secret';
|
|
1235
|
+
const prefix = baseResource === 'batchInferenceJobs' ? 'bij' : baseResource === 'supervisedFineTuningJobs' ? 'sft' : type === 'secret' ? 'secret' : `${type}`;
|
|
1236
|
+
return createControlResource(type, prefix, accountId, params, req, url, (p) => build(p));
|
|
1237
|
+
}
|
|
1238
|
+
// SINGLE RESOURCE: GET/PATCH/DELETE /v1/accounts/{id}/<plural>/<resource_id>
|
|
1239
|
+
if (rest.length === 2 && !verb) {
|
|
1240
|
+
const rid = idSeg!;
|
|
1241
|
+
// The lookup is ACCOUNT-SCOPED: a row created under another account is invisible here and
|
|
1242
|
+
// answers the same NOT_FOUND an unknown id answers (the vendor's own tenancy — an account
|
|
1243
|
+
// cannot see, patch or delete another account's resource).
|
|
1244
|
+
const r = getRow(singular, rid, accountId, req.root);
|
|
1245
|
+
if (method === 'GET') {
|
|
1246
|
+
if (!r || isTombstoned(r)) return gatewayError(404, `Not found: ${baseResource}/${rid}`);
|
|
1247
|
+
return { status: 200, body: resourceView(r) };
|
|
1248
|
+
}
|
|
1249
|
+
if (method === 'PATCH') {
|
|
1250
|
+
// ONE policy for output-only fields across every collection (grpc-gateway's own): the
|
|
1251
|
+
// spec's readOnly fields are REFUSED with the same unknown-field 400 an invented field
|
|
1252
|
+
// gets — "Cannot find field." — never silently dropped and never written. (Silently
|
|
1253
|
+
// ignoring them on deployments while datasets refused them was two policies on one
|
|
1254
|
+
// plane; the refusal is grpc-gateway's own behavior for a client-set readOnly field.)
|
|
1255
|
+
if (!PATCHABLE[baseResource]) return gatewayError(404, `Unknown request URL: ${method} ${path}.`);
|
|
1256
|
+
if (!r || isTombstoned(r)) return gatewayError(404, `Not found: ${baseResource}/${rid}`);
|
|
1257
|
+
const patchable = { ...(params as Record<string, unknown>) };
|
|
1258
|
+
// The per-collection field whitelist (the same one create uses): an undeclared field is
|
|
1259
|
+
// refused the way grpc-gateway refuses an unknown field, and an output-only field never
|
|
1260
|
+
// passes (googleads-twin.ts is the estate precedent for the refusal shape).
|
|
1261
|
+
const unknown = Object.keys(patchable).filter((k) => !PATCHABLE[baseResource]?.includes(k));
|
|
1262
|
+
if (unknown.length) return gatewayError(400, `Invalid JSON payload received. Unknown name "${unknown[0]}": Cannot find field.`);
|
|
1263
|
+
await applyTwinWrite(SERVICE, {
|
|
1264
|
+
operation: `${singular}.update`, subjectType: singular, subjectId: `${accountId}/${rid}`,
|
|
1265
|
+
fields: patchable,
|
|
1266
|
+
...(req.occurredAt ? { occurredAt: req.occurredAt } : {}), actor: { kind: 'agent' },
|
|
1267
|
+
}, req.root);
|
|
1268
|
+
return { status: 200, body: resourceView(getRow(singular, rid, accountId, req.root) ?? {}) };
|
|
1269
|
+
}
|
|
1270
|
+
if (method === 'DELETE') return deleteControlResource(singular, rid, accountId, req);
|
|
1271
|
+
}
|
|
1272
|
+
// CUSTOM VERB: POST /v1/accounts/{id}/<plural>/<resource_id>:<verb>
|
|
1273
|
+
if (rest.length === 2 && verb && method === 'POST') {
|
|
1274
|
+
const rid = idSeg!;
|
|
1275
|
+
const r = getRow(singular, rid, accountId, req.root);
|
|
1276
|
+
// :undelete addresses a DELETED row by design; every other verb needs a live row.
|
|
1277
|
+
if (!r || (isTombstoned(r) && verb !== 'undelete')) return gatewayError(404, `Not found: ${baseResource}/${rid}`);
|
|
1278
|
+
if (verb === 'cancel' && (baseResource === 'batchInferenceJobs' || baseResource === 'supervisedFineTuningJobs')) {
|
|
1279
|
+
if (r.state === 'JOB_STATE_CANCELLED' || r.state === 'JOB_STATE_CANCELLING') {
|
|
1280
|
+
return gatewayError(400, `Cannot cancel a job in state ${String(r.state)}`);
|
|
1281
|
+
}
|
|
1282
|
+
await applyTwinWrite(SERVICE, {
|
|
1283
|
+
operation: `${singular}.cancel`, subjectType: singular, subjectId: `${accountId}/${rid}`,
|
|
1284
|
+
fields: { state: 'JOB_STATE_CANCELLING' },
|
|
1285
|
+
...(req.occurredAt ? { occurredAt: req.occurredAt } : {}), actor: { kind: 'agent' },
|
|
1286
|
+
}, req.root);
|
|
1287
|
+
// The vendor's cancel answers `{}`.
|
|
1288
|
+
return { status: 200, body: {} };
|
|
1289
|
+
}
|
|
1290
|
+
if (verb === 'resume' && (baseResource === 'batchInferenceJobs' || baseResource === 'supervisedFineTuningJobs')) {
|
|
1291
|
+
await applyTwinWrite(SERVICE, {
|
|
1292
|
+
operation: `${singular}.resume`, subjectType: singular, subjectId: `${accountId}/${rid}`,
|
|
1293
|
+
fields: { state: 'JOB_STATE_RUNNING' },
|
|
1294
|
+
...(req.occurredAt ? { occurredAt: req.occurredAt } : {}), actor: { kind: 'agent' },
|
|
1295
|
+
}, req.root);
|
|
1296
|
+
return { status: 200, body: {} };
|
|
1297
|
+
}
|
|
1298
|
+
if (verb === 'undelete' && baseResource === 'deployments') {
|
|
1299
|
+
await applyTwinWrite(SERVICE, {
|
|
1300
|
+
operation: 'deployment.undelete', subjectType: 'deployment', subjectId: `${accountId}/${rid}`,
|
|
1301
|
+
fields: { _deleted: false, state: 'CREATING' },
|
|
1302
|
+
...(req.occurredAt ? { occurredAt: req.occurredAt } : {}), actor: { kind: 'agent' },
|
|
1303
|
+
}, req.root);
|
|
1304
|
+
return { status: 200, body: resourceView(getRow('deployment', rid, accountId, req.root) ?? {}) };
|
|
1305
|
+
}
|
|
1306
|
+
if (verb === 'scale' && baseResource === 'deployments' && method === 'POST') {
|
|
1307
|
+
const replicaCount = (params as Record<string, unknown>).replicaCount;
|
|
1308
|
+
if (typeof replicaCount !== 'number') return gatewayError(400, 'replicaCount is required');
|
|
1309
|
+
await applyTwinWrite(SERVICE, {
|
|
1310
|
+
operation: 'deployment.scale', subjectType: 'deployment', subjectId: `${accountId}/${rid}`,
|
|
1311
|
+
fields: { replicaCount },
|
|
1312
|
+
...(req.occurredAt ? { occurredAt: req.occurredAt } : {}), actor: { kind: 'agent' },
|
|
1313
|
+
}, req.root);
|
|
1314
|
+
return { status: 200, body: resourceView(getRow('deployment', rid, accountId, req.root) ?? {}) };
|
|
1315
|
+
}
|
|
1316
|
+
return gatewayError(404, `Unknown operation: ${verb} on ${baseResource}`);
|
|
1317
|
+
}
|
|
1318
|
+
// PATCH /v1/accounts/{id}/deployments/{deployment_id}:scale — the spec's own spelling
|
|
1319
|
+
// (Gateway_ScaleDeployment, body {replicaCount}); it answers `{}`, not the resource.
|
|
1320
|
+
if (rest.length === 2 && verb === 'scale' && baseResource === 'deployments' && method === 'PATCH') {
|
|
1321
|
+
const rid = idSeg!;
|
|
1322
|
+
const r = getRow('deployment', rid, accountId, req.root);
|
|
1323
|
+
if (!r || isTombstoned(r)) return gatewayError(404, `Not found: deployments/${rid}`);
|
|
1324
|
+
const replicaCount = (params as Record<string, unknown>).replicaCount;
|
|
1325
|
+
if (typeof replicaCount !== 'number') return gatewayError(400, 'replicaCount is required');
|
|
1326
|
+
await applyTwinWrite(SERVICE, {
|
|
1327
|
+
operation: 'deployment.scale', subjectType: 'deployment', subjectId: `${accountId}/${rid}`,
|
|
1328
|
+
fields: { replicaCount },
|
|
1329
|
+
...(req.occurredAt ? { occurredAt: req.occurredAt } : {}), actor: { kind: 'agent' },
|
|
1330
|
+
}, req.root);
|
|
1331
|
+
return { status: 200, body: {} };
|
|
1332
|
+
}
|
|
1333
|
+
// users/{user_id}/apiKeys — nested under users. The verb-suffixed collection spelling
|
|
1334
|
+
// (`apiKeys:delete`) carries its verb on the apiKeys segment itself.
|
|
1335
|
+
if (baseResource === 'users' && rest.length >= 3 && rest[2]!.split(':')[0] === 'apiKeys') {
|
|
1336
|
+
const userId = rest[1]!;
|
|
1337
|
+
const user = getRow('user', userId, accountId, req.root);
|
|
1338
|
+
if (!user || user._deleted) return gatewayError(404, `Not found: users/${userId}`);
|
|
1339
|
+
const keySeg = rest.slice(3);
|
|
1340
|
+
if (keySeg.length === 0 && method === 'GET') {
|
|
1341
|
+
// The full key is returned ONCE at creation — the LIST answers it never. A list row
|
|
1342
|
+
// carrying `fw_…` would be a custody lie the single-key GET two branches down doesn't
|
|
1343
|
+
// commit, so the strip happens here too.
|
|
1344
|
+
const keys = rows('apiKey', accountId, req.root).filter((r) => !r._deleted && r._user_id === userId).map(resourceView);
|
|
1345
|
+
for (const k of keys) delete k.key;
|
|
1346
|
+
return { status: 200, body: { apiKeys: keys, nextPageToken: null, totalSize: keys.length } };
|
|
1347
|
+
}
|
|
1348
|
+
// POST .../apiKeys:delete — the vendor's own odd verb-suffixed delete with a {keyId} body
|
|
1349
|
+
// (collection-level spelling, Gateway_DeleteApiKey; answers `{}`).
|
|
1350
|
+
if (keySeg.length === 0 && method === 'POST' && rest[2] === 'apiKeys:delete') {
|
|
1351
|
+
const keyId = typeof params.keyId === 'string' ? params.keyId : undefined;
|
|
1352
|
+
if (!keyId) return gatewayError(400, 'keyId is required');
|
|
1353
|
+
const k = getRow('apiKey', keyId, accountId, req.root);
|
|
1354
|
+
if (!k || k._deleted || k._user_id !== userId) return gatewayError(404, `Not found: apiKeys/${keyId}`);
|
|
1355
|
+
await applyTwinWrite(SERVICE, {
|
|
1356
|
+
operation: 'apiKey.delete', subjectType: 'apiKey', subjectId: `${accountId}/${keyId}`, fields: { _deleted: true },
|
|
1357
|
+
...(req.occurredAt ? { occurredAt: req.occurredAt } : {}), actor: { kind: 'agent' },
|
|
1358
|
+
}, req.root);
|
|
1359
|
+
return { status: 200, body: {} };
|
|
1360
|
+
}
|
|
1361
|
+
if (keySeg.length === 0 && method === 'POST' && rest[2] === 'apiKeys') {
|
|
1362
|
+
// The verb check is load-bearing: `apiKeys:frobnicate` must 404 below, never mint a key.
|
|
1363
|
+
const inner = (params.apiKey && typeof params.apiKey === 'object' ? params.apiKey : params) as Record<string, unknown>;
|
|
1364
|
+
const id = nextId('apiKey', 'key', accountId, req.root);
|
|
1365
|
+
const at = new Date(req.occurredAt ?? '1970-01-01T00:00:00Z').toISOString().replace(/\.\d{3}Z$/, 'Z');
|
|
1366
|
+
const key = `fw_${stableSuffix(id + at)}${fnv1a(id).toString(36)}`;
|
|
1367
|
+
await applyTwinWrite(SERVICE, {
|
|
1368
|
+
operation: 'apiKey.create', subjectType: 'apiKey', subjectId: `${accountId}/${id}`,
|
|
1369
|
+
fields: {
|
|
1370
|
+
name: `accounts/${accountId}/users/${userId}/apiKeys/${id}`,
|
|
1371
|
+
displayName: inner.displayName ?? 'default',
|
|
1372
|
+
key, // returned ONCE at creation, never again (the vendor's own contract)
|
|
1373
|
+
prefix: key.slice(0, 6),
|
|
1374
|
+
keyId: id,
|
|
1375
|
+
_user_id: userId,
|
|
1376
|
+
_account: accountId,
|
|
1377
|
+
createTime: at,
|
|
1378
|
+
expireTime: typeof inner.expireTime === 'string' ? inner.expireTime : null,
|
|
1379
|
+
},
|
|
1380
|
+
...(req.occurredAt ? { occurredAt: req.occurredAt } : {}), actor: { kind: 'agent' },
|
|
1381
|
+
}, req.root);
|
|
1382
|
+
return { status: 200, body: resourceView(getRow('apiKey', id, accountId, req.root) ?? {}) };
|
|
1383
|
+
}
|
|
1384
|
+
if (keySeg.length === 1 && method === 'GET') {
|
|
1385
|
+
const k = getRow('apiKey', keySeg[0]!, accountId, req.root);
|
|
1386
|
+
if (!k || k._deleted || k._user_id !== userId) return gatewayError(404, `Not found: apiKeys/${keySeg[0]}`);
|
|
1387
|
+
const view = resourceView(k);
|
|
1388
|
+
delete view.key; // "only available upon creation and not stored thereafter"
|
|
1389
|
+
return { status: 200, body: view };
|
|
1390
|
+
}
|
|
1391
|
+
if (keySeg.length === 1 && method === 'PATCH') {
|
|
1392
|
+
const k = getRow('apiKey', keySeg[0]!, accountId, req.root);
|
|
1393
|
+
if (!k || k._deleted || k._user_id !== userId) return gatewayError(404, `Not found: apiKeys/${keySeg[0]}`);
|
|
1394
|
+
const patchable = { ...(params as Record<string, unknown>) };
|
|
1395
|
+
for (const f of ['key', 'keyId', 'prefix', 'secure', 'email', 'createTime', 'lastUsed', 'isFirepass']) delete patchable[f];
|
|
1396
|
+
const unknown = Object.keys(patchable).filter((f) => !['displayName', 'expireTime'].includes(f));
|
|
1397
|
+
if (unknown.length) return gatewayError(400, `Invalid JSON payload received. Unknown name "${unknown[0]}": Cannot find field.`);
|
|
1398
|
+
await applyTwinWrite(SERVICE, {
|
|
1399
|
+
operation: 'apiKey.update', subjectType: 'apiKey', subjectId: `${accountId}/${keySeg[0]!}`,
|
|
1400
|
+
fields: patchable,
|
|
1401
|
+
...(req.occurredAt ? { occurredAt: req.occurredAt } : {}), actor: { kind: 'agent' },
|
|
1402
|
+
}, req.root);
|
|
1403
|
+
const view = resourceView(getRow('apiKey', keySeg[0]!, accountId, req.root) ?? {});
|
|
1404
|
+
delete view.key;
|
|
1405
|
+
return { status: 200, body: view };
|
|
1406
|
+
}
|
|
1407
|
+
// POST .../apiKeys:delete — the vendor's own odd verb-suffixed delete with a {keyId} body.
|
|
1408
|
+
// The key must belong to the addressed user: the id segment is under that user's own path,
|
|
1409
|
+
// and deleting another user's key through it would skip the scoping the collection-level
|
|
1410
|
+
// spelling enforces one branch up.
|
|
1411
|
+
if (keySeg.length === 1 && keySeg[0]!.endsWith(':delete') && method === 'POST') {
|
|
1412
|
+
const keyId = typeof params.keyId === 'string' ? params.keyId : keySeg[0]!.slice(0, -':delete'.length);
|
|
1413
|
+
const k = getRow('apiKey', keyId, accountId, req.root);
|
|
1414
|
+
if (!k || k._deleted || k._user_id !== userId) return gatewayError(404, `Not found: apiKeys/${keyId}`);
|
|
1415
|
+
await applyTwinWrite(SERVICE, {
|
|
1416
|
+
operation: 'apiKey.delete', subjectType: 'apiKey', subjectId: `${accountId}/${keyId}`, fields: { _deleted: true },
|
|
1417
|
+
...(req.occurredAt ? { occurredAt: req.occurredAt } : {}), actor: { kind: 'agent' },
|
|
1418
|
+
}, req.root);
|
|
1419
|
+
return { status: 200, body: {} };
|
|
1420
|
+
}
|
|
1421
|
+
}
|
|
1422
|
+
return gatewayError(404, `Unknown request URL: ${method} ${path}.`);
|
|
1423
|
+
}
|
|
1424
|
+
|
|
1425
|
+
// Not on either plane — the twin serves nothing else.
|
|
1426
|
+
return notFound(`Unknown request URL: ${method} ${path}.`);
|
|
1427
|
+
}
|