@volter/twin-fireworks 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (39) hide show
  1. package/LICENSE +202 -0
  2. package/README.md +184 -0
  3. package/dist/src/cli.d.ts +2 -0
  4. package/dist/src/cli.js +28 -0
  5. package/dist/src/fireworks-budget.d.ts +54 -0
  6. package/dist/src/fireworks-budget.js +146 -0
  7. package/dist/src/fireworks-capabilities.d.ts +4 -0
  8. package/dist/src/fireworks-capabilities.js +1205 -0
  9. package/dist/src/fireworks-conformance.d.ts +14 -0
  10. package/dist/src/fireworks-conformance.js +514 -0
  11. package/dist/src/fireworks-connector.d.ts +168 -0
  12. package/dist/src/fireworks-connector.js +641 -0
  13. package/dist/src/fireworks-models.d.ts +11 -0
  14. package/dist/src/fireworks-models.js +53 -0
  15. package/dist/src/fireworks-scenario.d.ts +55 -0
  16. package/dist/src/fireworks-scenario.js +171 -0
  17. package/dist/src/fireworks-server.d.ts +16 -0
  18. package/dist/src/fireworks-server.js +144 -0
  19. package/dist/src/fireworks-stub.d.ts +26 -0
  20. package/dist/src/fireworks-stub.js +78 -0
  21. package/dist/src/fireworks-twin.d.ts +51 -0
  22. package/dist/src/fireworks-twin.js +1426 -0
  23. package/dist/src/fireworks-types.d.ts +212 -0
  24. package/dist/src/fireworks-types.js +4 -0
  25. package/dist/src/index.d.ts +9 -0
  26. package/dist/src/index.js +105 -0
  27. package/package.json +52 -0
  28. package/src/cli.ts +27 -0
  29. package/src/fireworks-budget.ts +172 -0
  30. package/src/fireworks-capabilities.ts +1229 -0
  31. package/src/fireworks-conformance.ts +542 -0
  32. package/src/fireworks-connector.ts +700 -0
  33. package/src/fireworks-models.ts +63 -0
  34. package/src/fireworks-scenario.ts +191 -0
  35. package/src/fireworks-server.ts +153 -0
  36. package/src/fireworks-stub.ts +83 -0
  37. package/src/fireworks-twin.ts +1427 -0
  38. package/src/fireworks-types.ts +165 -0
  39. package/src/index.ts +134 -0
@@ -0,0 +1,53 @@
1
+ // Fireworks model catalog — the static slice the twin's INFERENCE plane accepts, on groq's
2
+ // method (groq-models.ts): a static table with provenance comments, merged over any PULLED
3
+ // control-plane model rows (a pulled row of the same id overrides the static one).
4
+ //
5
+ // WHY A CATALOG AT ALL: real Fireworks answers 404 "Model id not found" for an unknown model id
6
+ // (the vendor's own error table: 404 covers "the model doesn't exist, the model is not deployed,
7
+ // or you don't have permission to access it"). A twin that answered 200 for ANY string was a
8
+ // fake success the vendor's own clients would never see.
9
+ //
10
+ // PROVENANCE (what is sourced, and what is not):
11
+ // • the chat ids are the serverless serving paths Fireworks documents (docs.fireworks.ai
12
+ // serverless serving paths + fireconnect model tables, read 2026-09-16): short aliases
13
+ // (`glm-5p2`, `kimi-latest`) and full resource names (`accounts/fireworks/models/…`) are
14
+ // BOTH documented spellings, so both are accepted.
15
+ // • the embedding ids are the Qwen3 Embedding table (docs.fireworks.ai "Generating
16
+ // embeddings": `fireworks/qwen3-embedding-8b` serverless; 4B/0.6B dedicated) — the twin
17
+ // serves the serverless one.
18
+ // • the rerank id is the Qwen3 Reranker table (docs.fireworks.ai "Reranking documents":
19
+ // `fireworks/qwen3-reranker-8b` serverless).
20
+ // • context windows are NOT sourced per-id here (the stub does not model context limits per
21
+ // model — the stub context is fixed at 8192 tokens, documented in fireworks-twin.ts); no
22
+ // capability asserts a per-model window. The catalog's job is the 404 REFUSAL, not metadata.
23
+ // A model absent from this table is deliberately unknown: the honest gap is the catalog's
24
+ // COVERAGE, filed as the fireworks.models.catalog_coverage todo — not a silent 200.
25
+ export const FIREWORKS_MODELS = [
26
+ // Serverless chat — documented serving paths (short aliases + full resource names).
27
+ { id: 'kimi-latest', kind: 'chat' },
28
+ { id: 'kimi-fast-latest', kind: 'chat' },
29
+ { id: 'glm-latest', kind: 'chat' },
30
+ { id: 'glm-fast-latest', kind: 'chat' },
31
+ { id: 'glm-5p2', kind: 'chat' },
32
+ { id: 'glm-5p3-flash', kind: 'chat' },
33
+ { id: 'kimi-k2-instruct', kind: 'chat' },
34
+ { id: 'accounts/fireworks/models/kimi-k2-instruct', kind: 'chat' },
35
+ { id: 'accounts/fireworks/models/kimi-k2-instruct-0905', kind: 'chat' },
36
+ { id: 'accounts/fireworks/models/llama-v3p1-8b-instruct', kind: 'chat' },
37
+ // Serverless embeddings + rerank — the Qwen3 tables (the 4B/0.6B tiers are dedicated-only),
38
+ // plus the legacy Hugging-Face-style embedder the docs name explicitly ("These models are not
39
+ // in the Model Library … they still work on serverless if you pass the exact Hugging Face
40
+ // style id"). The earlier `accounts/fireworks/models/nomic-embed-text-v1_5` spelling this
41
+ // pack first used was never a vendor id — the sourced id is the one below.
42
+ { id: 'fireworks/qwen3-embedding-8b', kind: 'embeddings' },
43
+ { id: 'accounts/fireworks/models/qwen3-embedding-8b', kind: 'embeddings' },
44
+ { id: 'nomic-ai/nomic-embed-text-v1.5', kind: 'embeddings' },
45
+ { id: 'fireworks/qwen3-reranker-8b', kind: 'rerank' },
46
+ { id: 'accounts/fireworks/models/qwen3-reranker-8b', kind: 'rerank' },
47
+ ];
48
+ /** The static ids, for the refusal check. */
49
+ export const FIREWORKS_MODEL_IDS = new Set(FIREWORKS_MODELS.map((m) => m.id));
50
+ /** Which endpoint family a static model id belongs to (undefined = not a chat id). */
51
+ export function catalogKind(id) {
52
+ return FIREWORKS_MODELS.find((m) => m.id === id)?.kind;
53
+ }
@@ -0,0 +1,55 @@
1
+ import { type PackScenarioAdapter, type ScenarioDocument, ScenarioEngine } from '@volter/world-core';
2
+ import type { FireworksMessageParam } from './fireworks-types.js';
3
+ export type FireworksScenarioRequest = {
4
+ model: string;
5
+ messages: FireworksMessageParam[];
6
+ tools?: unknown;
7
+ /** Fireworks-specific: 'auto' | 'default' | 'flex' | 'priority' — only 'priority' is honored
8
+ * by the vendor; the rest behave as 'default'. Scriptable so a world can pin behavior per tier. */
9
+ serviceTier?: string;
10
+ };
11
+ export type FireworksScenarioEngine = ScenarioEngine<FireworksScenarioRequest>;
12
+ export type ScenarioToolCall = {
13
+ name: string;
14
+ arguments: Record<string, unknown>;
15
+ id?: string;
16
+ };
17
+ /**
18
+ * What a fireworks handler may script. `reasoning` is Fireworks-specific (the separate
19
+ * `reasoning_content` field its reasoning models answer with). A scripted FAILURE is not
20
+ * vocabulary of its own — it is the kernel fault grammar (`fault: { kind: 'status', … }`,
21
+ * R15), rendered into Fireworks' envelope by the adapter's renderFault below.
22
+ */
23
+ export type FireworksScenarioRespond = {
24
+ text?: string;
25
+ reasoning?: string;
26
+ toolCalls?: ScenarioToolCall | ScenarioToolCall[];
27
+ finishReason?: 'stop' | 'length' | 'tool_calls';
28
+ };
29
+ export type ScriptedResult = {
30
+ text: string | null;
31
+ reasoning: string | null;
32
+ toolCalls: Array<{
33
+ id: string;
34
+ type: 'function';
35
+ function: {
36
+ name: string;
37
+ arguments: string;
38
+ };
39
+ }>;
40
+ finishReason: 'stop' | 'length' | 'tool_calls';
41
+ };
42
+ export declare const fireworksScenarioAdapter: PackScenarioAdapter<FireworksScenarioRequest>;
43
+ /** Load + STRICTLY validate a scenario document. A malformed file throws at load with what is
44
+ * wrong — never a silent ignore-and-stub (a mis-typed rule falling back is a fake success). */
45
+ export declare function loadFireworksScenarioDocument(path: string): ScenarioDocument;
46
+ export declare function createFireworksScenarioEngine(document?: ScenarioDocument): FireworksScenarioEngine;
47
+ /**
48
+ * Turn a validated `respond` into the pack's own faithful assistant turn.
49
+ *
50
+ * The tool-call id is derived from the scripted call's own content and its position in the
51
+ * handler (never a module-level counter), exactly as the non-scenario path derives its ids — a
52
+ * serve-path determinism requirement: a served response is a pure function of (request, stored
53
+ * state), so two identical scripted requests in one process must answer identically.
54
+ */
55
+ export declare function realizeFireworksRespond(respond: FireworksScenarioRespond): ScriptedResult;
@@ -0,0 +1,171 @@
1
+ // The fireworks pack's scenario system on the kernel's ONE engine (@volter/world-core scenario.ts):
2
+ // Fireworks chat-completions vocabulary + the scripted-turn respond shape. THE KERNEL ENGINE IS NOT
3
+ // RE-IMPLEMENTED HERE — this file is only the pack's adapter (features, matchers, load
4
+ // validation) plus the realizer that turns a validated `respond` into Fireworks' own envelope.
5
+ //
6
+ // The handler FILE (handlers/fireworks.json in a world dir) is the only write surface; the doors
7
+ // (`GET /twin`, `GET /twin/scenario`) are read-only. Scenario support is twin-only scaffolding
8
+ // for eval worlds, NOT vendor surface, so it is deliberately absent from the capability
9
+ // manifest (ADDING_A_TWIN.md §6, "Scenario scripting") and gated by fireworks-scenario.test.ts.
10
+ import { getActiveWorldStore, parseScenarioDocument, ScenarioError, ScenarioEngine } from '@volter/world-core';
11
+ import { contentToText, lastUserText } from "./fireworks-stub.js";
12
+ const RESPOND_KEYS = new Set(['text', 'reasoning', 'toolCalls', 'finishReason']);
13
+ const FINISH_REASONS = new Set(['stop', 'length', 'tool_calls']);
14
+ const SERVICE_TIERS = new Set(['auto', 'default', 'flex', 'priority']);
15
+ const nonEmptyString = (cond) => typeof cond === 'string' && cond.length > 0;
16
+ function lastToolResultNames(messages) {
17
+ const names = new Set();
18
+ const last = messages[messages.length - 1];
19
+ if (!last || last.role !== 'tool' || typeof last.tool_call_id !== 'string')
20
+ return names;
21
+ for (const m of messages) {
22
+ const am = m;
23
+ if (am.role !== 'assistant' || !Array.isArray(am.tool_calls))
24
+ continue;
25
+ for (const tc of am.tool_calls) {
26
+ if (tc?.id === last.tool_call_id && typeof tc?.function?.name === 'string')
27
+ names.add(tc.function.name);
28
+ }
29
+ }
30
+ return names;
31
+ }
32
+ function toolNames(tools) {
33
+ if (!Array.isArray(tools))
34
+ return [];
35
+ return tools.map((t) => typeof t?.function?.name === 'string' ? t.function.name : typeof t?.name === 'string' ? t.name : null).filter((n) => n !== null);
36
+ }
37
+ export const fireworksScenarioAdapter = {
38
+ vendor: 'fireworks',
39
+ // R15 — a status fault in THIS vendor's envelope: the same `{error:{message, code, type}}`
40
+ // OpenAI-compat shape fireworks-twin.ts serves for its own refusals. The vendor's error table
41
+ // (docs.fireworks.ai/api-reference, "Common error codes") keys the type on the status: 429 is
42
+ // the serverless rate limit / deployment-capacity signal, ≥500 is the vendor's server-side
43
+ // family. The kernel adds the `retry-after` header when the fault carries retryAfterSeconds —
44
+ // renderFault must not add it itself.
45
+ renderFault: (f) => ({
46
+ body: {
47
+ error: {
48
+ message: f.message ?? (f.status === 429
49
+ ? 'Rate limit exceeded. Please retry your request.'
50
+ : f.status >= 500
51
+ ? 'Internal Server Error'
52
+ : 'The request was refused by a scripted fault.'),
53
+ code: f.status,
54
+ type: f.status === 429 ? 'rate_limit_exceeded' : f.status >= 500 ? 'internal_server_error' : 'invalid_request_error',
55
+ },
56
+ },
57
+ }),
58
+ features: (req) => ({
59
+ model: req.model,
60
+ lastUserText: lastUserText(req.messages).slice(0, 300),
61
+ tools: toolNames(req.tools),
62
+ lastMessageIsToolResult: req.messages[req.messages.length - 1]?.role === 'tool',
63
+ toolResultFor: [...lastToolResultNames(req.messages)],
64
+ serviceTier: req.serviceTier ?? 'default',
65
+ }),
66
+ matchers: {
67
+ modelEquals: (req, cond) => nonEmptyString(cond) && req.model === cond,
68
+ userTextIncludes: (req, cond) => nonEmptyString(cond) && lastUserText(req.messages).toLowerCase().includes(cond.toLowerCase()),
69
+ anyTextIncludes: (req, cond) => nonEmptyString(cond) && req.messages.map((m) => contentToText(m.content)).join('\n').toLowerCase().includes(cond.toLowerCase()),
70
+ lastMessageIsToolResult: (req, cond) => typeof cond === 'boolean' && (req.messages[req.messages.length - 1]?.role === 'tool') === cond,
71
+ toolResultFor: (req, cond) => nonEmptyString(cond) && lastToolResultNames(req.messages).has(cond),
72
+ hasTool: (req, cond) => nonEmptyString(cond) && toolNames(req.tools).includes(cond),
73
+ // Fireworks-specific: script behaviour per service tier ('priority' is the only honored one).
74
+ serviceTierEquals: (req, cond) => nonEmptyString(cond) && (req.serviceTier ?? 'default') === cond,
75
+ },
76
+ text: (req) => req.messages.map((m) => contentToText(m.content)).join('\n'),
77
+ validateOn: (on) => {
78
+ for (const k of ['modelEquals', 'userTextIncludes', 'anyTextIncludes', 'toolResultFor', 'hasTool']) {
79
+ if (on[k] !== undefined && (typeof on[k] !== 'string' || !on[k]))
80
+ return `on.${k} is a non-empty string`;
81
+ }
82
+ if (on.lastMessageIsToolResult !== undefined && typeof on.lastMessageIsToolResult !== 'boolean')
83
+ return 'on.lastMessageIsToolResult is a boolean';
84
+ if (on.serviceTierEquals !== undefined) {
85
+ if (typeof on.serviceTierEquals !== 'string' || !SERVICE_TIERS.has(on.serviceTierEquals)) {
86
+ return `on.serviceTierEquals is one of ${[...SERVICE_TIERS].join(', ')}`;
87
+ }
88
+ }
89
+ return null;
90
+ },
91
+ validateRespond: (respond) => {
92
+ if (typeof respond !== 'object' || respond === null || Array.isArray(respond))
93
+ return 'respond is an object { text?, reasoning?, toolCalls?, finishReason? }';
94
+ const r = respond;
95
+ for (const k of Object.keys(r))
96
+ if (!RESPOND_KEYS.has(k))
97
+ return `respond: unknown key "${k}" (valid: ${[...RESPOND_KEYS].join(', ')})`;
98
+ if (r.text !== undefined && typeof r.text !== 'string')
99
+ return 'respond.text is a string';
100
+ if (r.reasoning !== undefined && typeof r.reasoning !== 'string')
101
+ return 'respond.reasoning is a string';
102
+ if (r.finishReason !== undefined && (typeof r.finishReason !== 'string' || !FINISH_REASONS.has(r.finishReason)))
103
+ return `respond.finishReason is one of ${[...FINISH_REASONS].join(', ')}`;
104
+ if (r.toolCalls !== undefined) {
105
+ for (const tc of Array.isArray(r.toolCalls) ? r.toolCalls : [r.toolCalls]) {
106
+ const t = tc;
107
+ if (!t || typeof t !== 'object' || Array.isArray(t))
108
+ return 'respond.toolCalls entries are objects';
109
+ if (typeof t.name !== 'string' || !t.name)
110
+ return 'respond.toolCalls[].name is a non-empty string';
111
+ if (!t.arguments || typeof t.arguments !== 'object' || Array.isArray(t.arguments))
112
+ return 'respond.toolCalls[].arguments is an object';
113
+ if (t.id !== undefined && typeof t.id !== 'string')
114
+ return 'respond.toolCalls[].id is a string';
115
+ }
116
+ }
117
+ if (r.text === undefined && r.toolCalls === undefined)
118
+ return 'respond needs text or toolCalls';
119
+ return null;
120
+ },
121
+ };
122
+ /** Load + STRICTLY validate a scenario document. A malformed file throws at load with what is
123
+ * wrong — never a silent ignore-and-stub (a mis-typed rule falling back is a fake success). */
124
+ export function loadFireworksScenarioDocument(path) {
125
+ let parsed;
126
+ try {
127
+ // Read through the ACTIVE WorldStore, never the filesystem directly (runtime contract
128
+ // R12b): the handlers document is WORLD STATE, so a MemoryWorldStore / DO-backed world
129
+ // serves ITS OWN scenario instead of whatever happens to sit on the host disk — and the
130
+ // serve path stays workerd-clean. A missing document keeps the historical ENOENT wording,
131
+ // so the loud load-time failure reads byte-identically to the read it replaces.
132
+ const raw = getActiveWorldStore().read(path);
133
+ if (raw === null)
134
+ throw new Error(`ENOENT: no such file or directory, open '${path}'`);
135
+ parsed = JSON.parse(raw);
136
+ }
137
+ catch (e) {
138
+ throw new ScenarioError(`fireworks scenario: cannot read/parse ${path}: ${e instanceof Error ? e.message : String(e)}`);
139
+ }
140
+ return parseScenarioDocument(parsed, fireworksScenarioAdapter);
141
+ }
142
+ export function createFireworksScenarioEngine(document) {
143
+ return new ScenarioEngine(fireworksScenarioAdapter, document);
144
+ }
145
+ /**
146
+ * Turn a validated `respond` into the pack's own faithful assistant turn.
147
+ *
148
+ * The tool-call id is derived from the scripted call's own content and its position in the
149
+ * handler (never a module-level counter), exactly as the non-scenario path derives its ids — a
150
+ * serve-path determinism requirement: a served response is a pure function of (request, stored
151
+ * state), so two identical scripted requests in one process must answer identically.
152
+ */
153
+ export function realizeFireworksRespond(respond) {
154
+ const toolCalls = [];
155
+ const scripted = respond.toolCalls ? (Array.isArray(respond.toolCalls) ? respond.toolCalls : [respond.toolCalls]) : [];
156
+ scripted.forEach((tc, index) => {
157
+ const seed = `${index}:${tc.name}:${JSON.stringify(tc.arguments)}`;
158
+ let h = 0x811c9dc5;
159
+ for (let i = 0; i < seed.length; i++) {
160
+ h ^= seed.charCodeAt(i);
161
+ h = Math.imul(h, 0x01000193);
162
+ }
163
+ toolCalls.push({ id: tc.id ?? `call_scripted_${(h >>> 0).toString(36)}`, type: 'function', function: { name: tc.name, arguments: JSON.stringify(tc.arguments) } });
164
+ });
165
+ return {
166
+ text: respond.text ?? (toolCalls.length ? null : ''),
167
+ reasoning: respond.reasoning ?? null,
168
+ toolCalls,
169
+ finishReason: respond.finishReason ?? (toolCalls.length ? 'tool_calls' : 'stop'),
170
+ };
171
+ }
@@ -0,0 +1,16 @@
1
+ /** Options every Fireworks-twin HTTP surface needs, independent of who owns the socket. */
2
+ export interface FireworksTwinFetchOptions {
3
+ root?: string;
4
+ readOnly?: boolean;
5
+ scenarioPath?: string;
6
+ }
7
+ export declare function createFireworksTwinFetch(options: FireworksTwinFetchOptions): (request: Request) => Promise<Response>;
8
+ export declare function createFireworksTwinServer(options: {
9
+ root?: string;
10
+ port?: number;
11
+ readOnly?: boolean;
12
+ scenarioPath?: string;
13
+ }): Promise<{
14
+ port: number;
15
+ stop: () => void;
16
+ }>;
@@ -0,0 +1,144 @@
1
+ // Fireworks twin HTTP server — serve the full Fireworks twin handler over HTTP so the real
2
+ // `fireworks` Python SDK (pointed at `base_url: http://127.0.0.1:<port>/inference/v1`), the
3
+ // `@ai-sdk/fireworks` provider and plain OpenAI clients (pointed at
4
+ // `http://127.0.0.1:<port>/inference/v1`) work UNMODIFIED, and the Gateway control plane answers
5
+ // at `/v1/accounts/{account_id}/…`. JSON bodies pass straight through. Writable by default; pass
6
+ // readOnly to reject mutations (D3).
7
+ //
8
+ // Streaming: when a streamable request body has `"stream": true`, the server constructs a REAL SSE
9
+ // response by feeding the handler an sseSink that writes each chunk in the `data: <json>\n\n` wire
10
+ // format, ending with `data: [DONE]\n\n` on the OpenAI-compat paths (the Anthropic-compat
11
+ // /v1/messages ends with its `message_stop` event — no [DONE] sentinel; that is the vendor's own
12
+ // shape). The handler itself stays socket-free — the sink is the only place a socket is touched,
13
+ // on the live HTTP path.
14
+ //
15
+ // FETCH-FIRST (runtime contract R12b): the surface is the plain `createFireworksTwinFetch` and the
16
+ // SERVER is one line of `serveHttp` around it. This is a CUSTOM fetch, not the kernel adapter
17
+ // (`createTwinFetchFromHandler`): SSE streaming on THREE paths (chat/completions, completions,
18
+ // messages) genuinely exceeds the common shape — groq-server.ts is the reference.
19
+ import { serveHttp, twinManifest, worldNow } from '@volter/world-core';
20
+ import { createFireworksScenarioEngine, loadFireworksScenarioDocument } from "./fireworks-scenario.js";
21
+ import { FIREWORKS_INFERENCE_PREFIX, handleFireworksTwinRequest } from "./fireworks-twin.js";
22
+ function wantsStream(body) {
23
+ if (!body)
24
+ return false;
25
+ try {
26
+ return JSON.parse(body)?.stream === true;
27
+ }
28
+ catch {
29
+ return false;
30
+ }
31
+ }
32
+ function encodeSse(event) {
33
+ if (event.done)
34
+ return 'data: [DONE]\n\n';
35
+ return `data: ${JSON.stringify(event.data)}\n\n`;
36
+ }
37
+ // The three inference paths Fireworks streams. The Anthropic-compat /v1/messages streams in the
38
+ // SAME SSE transport but with its own event vocabulary — and the handler already ends its sequence
39
+ // with `message_stop`, so the [DONE] sentinel must NOT be appended there (Anthropic streams do not
40
+ // carry one; appending one would be a wire lie the official clients parse as an unknown event).
41
+ const STREAMABLE = new Set([
42
+ `${FIREWORKS_INFERENCE_PREFIX}/chat/completions`,
43
+ `${FIREWORKS_INFERENCE_PREFIX}/completions`,
44
+ `${FIREWORKS_INFERENCE_PREFIX}/messages`,
45
+ ]);
46
+ export function createFireworksTwinFetch(options) {
47
+ const readOnly = options.readOnly ?? false;
48
+ // Scenario scripting (fireworks-scenario.ts): a JSON scenario file — via the scenarioPath option
49
+ // or the TWIN_FIREWORKS_SCENARIO env var — scripts chat completions. Loaded ONCE at startup (a
50
+ // malformed file fails loudly here, never silently).
51
+ const scenarioPath = options.scenarioPath ?? process.env.TWIN_FIREWORKS_SCENARIO;
52
+ const scenarioEngine = scenarioPath ? createFireworksScenarioEngine(loadFireworksScenarioDocument(scenarioPath)) : undefined;
53
+ return async function fireworksTwinFetch(request) {
54
+ const url = new URL(request.url);
55
+ // THE READ DOORS (TWIN-PROGRAMMING-MODEL): discovery + inspection, read-only.
56
+ if (request.method === 'GET' && url.pathname.replace(/\/+$/, '') === '/twin') {
57
+ return Response.json(twinManifest({
58
+ vendor: 'fireworks',
59
+ twinOf: 'Fireworks AI API (inference plane at /inference/v1, Gateway control plane at /v1/accounts/{account_id})',
60
+ stateSentence: 'Seed control-plane state through the ordinary Gateway API (POST /v1/accounts/{account_id}/deployments?deploymentId=…) with any key.',
61
+ behaviorSentence: "Chat completions are scripted by MSW-shaped handlers in the world dir (handlers/fireworks.json): {on:{userTextIncludes|anyTextIncludes|modelEquals|lastMessageIsToolResult|serviceTierEquals|nthCall}, respond:{text|reasoning, finishReason?} | fault:{kind:'status'|'slow'|'drop', ...}, once?, phase?}. Unmatched requests answer a labeled stub naming this door.",
62
+ exampleHandler: { on: { userTextIncludes: 'summarize' }, respond: { text: 'Scripted summary.' }, once: true },
63
+ engine: scenarioEngine,
64
+ }));
65
+ }
66
+ if (request.method === 'GET' && url.pathname.replace(/\/+$/, '') === '/twin/scenario') {
67
+ return Response.json(scenarioEngine ? scenarioEngine.status() : { vendor: 'fireworks', handlers: [], misses: 0, recentMisses: [] });
68
+ }
69
+ const path = url.pathname + (url.search || '');
70
+ const cleanPath = url.pathname.replace(/\/+$/, '');
71
+ const passHeaders = {};
72
+ for (const k of ['authorization', 'anthropic-version']) {
73
+ const v = request.headers.get(k);
74
+ if (v !== null)
75
+ passHeaders[k] = v;
76
+ }
77
+ let body = '';
78
+ if (request.method !== 'GET')
79
+ body = await request.text();
80
+ // Streaming POST → a real text/event-stream response built from the sink.
81
+ //
82
+ // Events are COLLECTED first, then framed. The handler is synchronous-fast, so buffering
83
+ // costs nothing — and it is what lets a PRE-STREAM failure (validation 422/400, auth 401,
84
+ // rate-limit 429) answer with its REAL status and the vendor's JSON error envelope. Emitting
85
+ // the refusal as a lone `data:` frame inside a 200 text/event-stream was a FAKE SUCCESS the
86
+ // in-process path could not see. Fireworks rejects a bad request BEFORE opening the event
87
+ // stream; this twin decides every refusal before the first chunk.
88
+ if (!readOnly && request.method.toUpperCase() === 'POST' && STREAMABLE.has(cleanPath) && wantsStream(body)) {
89
+ const events = [];
90
+ const { status, body: out, headers: errHeaders } = await handleFireworksTwinRequest({
91
+ ...(scenarioEngine ? { scenarioEngine } : {}),
92
+ method: request.method, path, body, readOnly, occurredAt: worldNow(), headers: passHeaders,
93
+ ...(options.root !== undefined ? { root: options.root } : {}), sseSink: (e) => events.push(e),
94
+ });
95
+ if (status >= 400) {
96
+ return new Response(JSON.stringify(out), { status, headers: { 'content-type': 'application/json', 'x-request-id': 'req_twin', ...(errHeaders ?? {}) } });
97
+ }
98
+ // /v1/messages ends with its own message_stop; the OpenAI-compat paths end with [DONE].
99
+ const anthropicStream = cleanPath === `${FIREWORKS_INFERENCE_PREFIX}/messages`;
100
+ const stream = new ReadableStream({
101
+ start(controller) {
102
+ const enc = new TextEncoder();
103
+ for (const e of events) {
104
+ if (e.done && anthropicStream)
105
+ continue; // no [DONE] on the Anthropic surface
106
+ controller.enqueue(enc.encode(encodeSse(e)));
107
+ }
108
+ controller.close();
109
+ },
110
+ });
111
+ return new Response(stream, { headers: { 'content-type': 'text/event-stream; charset=utf-8', 'cache-control': 'no-cache', 'x-request-id': 'req_twin' } });
112
+ }
113
+ // Defensive: a handler throw must become a vendor-shaped 500, never an escaped exception
114
+ // (ADDING_A_TWIN.md §8, "In-process server harnesses … contain handler errors").
115
+ let status;
116
+ let out;
117
+ let outHeaders;
118
+ try {
119
+ const res = await handleFireworksTwinRequest({
120
+ ...(scenarioEngine ? { scenarioEngine } : {}),
121
+ method: request.method, path, body, readOnly,
122
+ occurredAt: worldNow(), headers: passHeaders,
123
+ ...(options.root !== undefined ? { root: options.root } : {}),
124
+ });
125
+ status = res.status;
126
+ out = res.body;
127
+ outHeaders = res.headers;
128
+ }
129
+ catch (err) {
130
+ return new Response(JSON.stringify({ error: { message: `twin handler error: ${err instanceof Error ? err.message : String(err)}`, code: 500 } }), {
131
+ status: 500, headers: { 'content-type': 'application/json', 'x-request-id': 'req_twin' },
132
+ });
133
+ }
134
+ return new Response(JSON.stringify(out), { status, headers: { 'content-type': 'application/json', 'x-request-id': 'req_twin', ...(outHeaders ?? {}) } });
135
+ };
136
+ }
137
+ export async function createFireworksTwinServer(options) {
138
+ const server = await serveHttp({
139
+ port: options.port ?? 0,
140
+ idleTimeout: 60,
141
+ fetch: createFireworksTwinFetch(options),
142
+ });
143
+ return { port: server.port ?? options.port ?? 0, stop: () => server.stop(true) };
144
+ }
@@ -0,0 +1,26 @@
1
+ /** A cheap stable hash (FNV-1a, 32-bit) — the same primitive the sibling packs use. */
2
+ export declare function fnv1a(text: string): number;
3
+ /** Flatten message/prompt content to plain text (strings joined; content blocks' text parts taken). */
4
+ export declare function contentToText(content: unknown): string;
5
+ /** ~1 token per 4 characters — the same rough estimate the sibling packs use. */
6
+ export declare function estimateTokens(text: string): number;
7
+ export declare function countPromptTokens(messages: Array<{
8
+ content?: unknown;
9
+ }>): number;
10
+ export declare function lastUserText(messages: Array<{
11
+ role?: string;
12
+ content?: unknown;
13
+ }>): string;
14
+ /** THE LABEL. Every generated completion carries it, so nothing can be mistaken for real output. */
15
+ export declare function stubAssistantText(model: string, prompt: string): string;
16
+ /** Deterministic reasoning content, for reasoning_effort on supported models. */
17
+ export declare function stubReasoningText(model: string, prompt: string): string;
18
+ /** Deterministic pseudo-embedding: L2-normalized, derived from the input hash. `dimensions` must
19
+ * already be a validated positive integer — the twin's handler refuses anything else BEFORE
20
+ * calling this (a client-sized allocation from an unvalidated number is a one-request DoS). The
21
+ * guard here is the class-level backstop: NO array in this pack is ever sized from a request
22
+ * value without a bound. */
23
+ export declare function pseudoEmbedding(text: string, dimensions: number): number[];
24
+ /** Deterministic relevance score in [0, 1] for rerank, biased by shared-token overlap so the
25
+ * ORDER of results responds to the query (a constant ordering would be a fake). */
26
+ export declare function stubRelevanceScore(query: string, document: string): number;
@@ -0,0 +1,78 @@
1
+ // Fireworks stub core — the deterministic-content primitives shared by the inference handlers.
2
+ // Mirrors groq-stub.ts / ai-gateway-stub.ts: NO model runs here, so generated content is a
3
+ // DETERMINISTIC, clearly-labeled function of the prompt — never pretending to be real inference.
4
+ // Everything AROUND the stub (the envelope, the field names, the rejections) is vendor-faithful.
5
+ /** A cheap stable hash (FNV-1a, 32-bit) — the same primitive the sibling packs use. */
6
+ export function fnv1a(text) {
7
+ let h = 0x811c9dc5;
8
+ for (let i = 0; i < text.length; i++) {
9
+ h ^= text.charCodeAt(i);
10
+ h = Math.imul(h, 0x01000193);
11
+ }
12
+ return h >>> 0;
13
+ }
14
+ /** Flatten message/prompt content to plain text (strings joined; content blocks' text parts taken). */
15
+ export function contentToText(content) {
16
+ if (typeof content === 'string')
17
+ return content;
18
+ if (Array.isArray(content)) {
19
+ return content
20
+ .map((part) => (typeof part === 'string' ? part : typeof part?.text === 'string' ? (part.text) : ''))
21
+ .join(' ');
22
+ }
23
+ return '';
24
+ }
25
+ /** ~1 token per 4 characters — the same rough estimate the sibling packs use. */
26
+ export function estimateTokens(text) {
27
+ return Math.max(1, Math.ceil(text.length / 4));
28
+ }
29
+ export function countPromptTokens(messages) {
30
+ return messages.reduce((sum, m) => sum + estimateTokens(contentToText(m.content)), 0);
31
+ }
32
+ export function lastUserText(messages) {
33
+ for (let i = messages.length - 1; i >= 0; i--) {
34
+ if (messages[i].role === 'user')
35
+ return contentToText(messages[i].content);
36
+ }
37
+ return '';
38
+ }
39
+ /** THE LABEL. Every generated completion carries it, so nothing can be mistaken for real output. */
40
+ export function stubAssistantText(model, prompt) {
41
+ const h = fnv1a(`${model}\0${prompt}`);
42
+ return `[twin-stub:${model}] Deterministic response to: ${prompt.slice(0, 120) || '(empty prompt)'} (hash ${h.toString(16)})`;
43
+ }
44
+ /** Deterministic reasoning content, for reasoning_effort on supported models. */
45
+ export function stubReasoningText(model, prompt) {
46
+ const h = fnv1a(`reasoning\0${model}\0${prompt}`);
47
+ return `[twin-stub:${model}] reasoning trace (hash ${h.toString(16)})`;
48
+ }
49
+ /** Deterministic pseudo-embedding: L2-normalized, derived from the input hash. `dimensions` must
50
+ * already be a validated positive integer — the twin's handler refuses anything else BEFORE
51
+ * calling this (a client-sized allocation from an unvalidated number is a one-request DoS). The
52
+ * guard here is the class-level backstop: NO array in this pack is ever sized from a request
53
+ * value without a bound. */
54
+ export function pseudoEmbedding(text, dimensions) {
55
+ if (!Number.isInteger(dimensions) || dimensions < 1 || dimensions > 4096) {
56
+ throw new RangeError(`pseudoEmbedding: dimensions must be an integer in [1, 4096], got ${dimensions}`);
57
+ }
58
+ const vec = new Array(dimensions);
59
+ let h = fnv1a(text);
60
+ for (let i = 0; i < dimensions; i++) {
61
+ h = Math.imul(h ^ (i + 1), 0x01000193) >>> 0;
62
+ vec[i] = (h / 0xffffffff) * 2 - 1;
63
+ }
64
+ const norm = Math.hypot(...vec) || 1;
65
+ return vec.map((v) => v / norm);
66
+ }
67
+ /** Deterministic relevance score in [0, 1] for rerank, biased by shared-token overlap so the
68
+ * ORDER of results responds to the query (a constant ordering would be a fake). */
69
+ export function stubRelevanceScore(query, document) {
70
+ const q = new Set(query.toLowerCase().split(/\W+/).filter(Boolean));
71
+ const d = new Set(document.toLowerCase().split(/\W+/).filter(Boolean));
72
+ let overlap = 0;
73
+ for (const t of q)
74
+ if (d.has(t))
75
+ overlap++;
76
+ const base = (fnv1a(`${query}\0${document}`) % 1000) / 10000; // 0..0.1 jitter
77
+ return Math.min(1, overlap / Math.max(1, q.size) + base);
78
+ }
@@ -0,0 +1,51 @@
1
+ import type { FireworksScenarioEngine } from './fireworks-scenario.js';
2
+ /** The inference base path. The vendor's inference server root is `https://api.fireworks.ai/inference`
3
+ * and every OpenAI-compat operation hangs off `/v1/…` under it. */
4
+ export declare const FIREWORKS_INFERENCE_PREFIX = "/inference/v1";
5
+ /** The control-plane base path: everything Gateway REST hangs off `/v1/accounts/{account_id}/…`. */
6
+ export declare const FIREWORKS_ACCOUNTS_PREFIX = "/v1/accounts";
7
+ export type FireworksRequest = {
8
+ method: string;
9
+ path: string;
10
+ body?: string;
11
+ occurredAt?: string;
12
+ root?: string;
13
+ readOnly?: boolean;
14
+ /** The credential the caller presents (the SDKs' bearer `Authorization` header). When a request
15
+ * carries an auth SURFACE (this field set, or `headers` present), the twin holds it to the real
16
+ * vendor rule: a credential is required → 401 on missing/invalid. In-process trusted calls
17
+ * (capability verify, connector) omit BOTH and are not auth-gated. */
18
+ apiKey?: string;
19
+ /** Lower-cased request headers the HTTP server passes through so the handler can model auth. */
20
+ headers?: Record<string, string>;
21
+ /** When set on a streaming POST, chunks are written here (no sockets). */
22
+ sseSink?: SseSink;
23
+ /** Scenario scripting (fireworks-scenario.ts): when set, chat completions consult the engine
24
+ * first — a matching handler scripts the answer, a miss answers the labeled stub with a
25
+ * pointer to the miss. Twin-only scaffolding, never vendor surface. */
26
+ scenarioEngine?: FireworksScenarioEngine;
27
+ };
28
+ export type SseEvent = {
29
+ data?: unknown;
30
+ done?: boolean;
31
+ };
32
+ export type SseSink = (event: SseEvent) => void;
33
+ /** The handler response. `headers` (when present) are response headers the HTTP server should set. */
34
+ export type FireworksResponseEnvelope = {
35
+ status: number;
36
+ body: unknown;
37
+ headers?: Record<string, string>;
38
+ };
39
+ /**
40
+ * Rows of `type`, SCOPED TO ONE ACCOUNT. Every control-plane resource is stored with the
41
+ * `accountId` it was created (or pulled) under, and every read goes through this — a resource
42
+ * addressed under a different account's path is invisible here, which is what makes the
43
+ * cross-account GET/PATCH/DELETE answer the vendor's own NOT_FOUND. `accountId === undefined`
44
+ * means the caller reads a plane with NO account segment (the inference plane's stored
45
+ * responses) — those rows are not account-scoped and the filter is skipped for them.
46
+ */
47
+ /** A row is gone when the TWIN deleted it (`_deleted`, the twin's own marker) or when the
48
+ * CONNECTOR observed it vanish from the vendor (`deleted`, the kernel's tombstone field).
49
+ * Reading only the first served a resource the pull had correctly tombstoned (round-four review). */
50
+ export declare function isTombstoned(r: Record<string, unknown>): boolean;
51
+ export declare function handleFireworksTwinRequest(req: FireworksRequest): Promise<FireworksResponseEnvelope>;