@volter/twin-fireworks 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (39) hide show
  1. package/LICENSE +202 -0
  2. package/README.md +184 -0
  3. package/dist/src/cli.d.ts +2 -0
  4. package/dist/src/cli.js +28 -0
  5. package/dist/src/fireworks-budget.d.ts +54 -0
  6. package/dist/src/fireworks-budget.js +146 -0
  7. package/dist/src/fireworks-capabilities.d.ts +4 -0
  8. package/dist/src/fireworks-capabilities.js +1205 -0
  9. package/dist/src/fireworks-conformance.d.ts +14 -0
  10. package/dist/src/fireworks-conformance.js +514 -0
  11. package/dist/src/fireworks-connector.d.ts +168 -0
  12. package/dist/src/fireworks-connector.js +641 -0
  13. package/dist/src/fireworks-models.d.ts +11 -0
  14. package/dist/src/fireworks-models.js +53 -0
  15. package/dist/src/fireworks-scenario.d.ts +55 -0
  16. package/dist/src/fireworks-scenario.js +171 -0
  17. package/dist/src/fireworks-server.d.ts +16 -0
  18. package/dist/src/fireworks-server.js +144 -0
  19. package/dist/src/fireworks-stub.d.ts +26 -0
  20. package/dist/src/fireworks-stub.js +78 -0
  21. package/dist/src/fireworks-twin.d.ts +51 -0
  22. package/dist/src/fireworks-twin.js +1426 -0
  23. package/dist/src/fireworks-types.d.ts +212 -0
  24. package/dist/src/fireworks-types.js +4 -0
  25. package/dist/src/index.d.ts +9 -0
  26. package/dist/src/index.js +105 -0
  27. package/package.json +52 -0
  28. package/src/cli.ts +27 -0
  29. package/src/fireworks-budget.ts +172 -0
  30. package/src/fireworks-capabilities.ts +1229 -0
  31. package/src/fireworks-conformance.ts +542 -0
  32. package/src/fireworks-connector.ts +700 -0
  33. package/src/fireworks-models.ts +63 -0
  34. package/src/fireworks-scenario.ts +191 -0
  35. package/src/fireworks-server.ts +153 -0
  36. package/src/fireworks-stub.ts +83 -0
  37. package/src/fireworks-twin.ts +1427 -0
  38. package/src/fireworks-types.ts +165 -0
  39. package/src/index.ts +134 -0
@@ -0,0 +1,63 @@
1
+ // Fireworks model catalog — the static slice the twin's INFERENCE plane accepts, on groq's
2
+ // method (groq-models.ts): a static table with provenance comments, merged over any PULLED
3
+ // control-plane model rows (a pulled row of the same id overrides the static one).
4
+ //
5
+ // WHY A CATALOG AT ALL: real Fireworks answers 404 "Model id not found" for an unknown model id
6
+ // (the vendor's own error table: 404 covers "the model doesn't exist, the model is not deployed,
7
+ // or you don't have permission to access it"). A twin that answered 200 for ANY string was a
8
+ // fake success the vendor's own clients would never see.
9
+ //
10
+ // PROVENANCE (what is sourced, and what is not):
11
+ // • the chat ids are the serverless serving paths Fireworks documents (docs.fireworks.ai
12
+ // serverless serving paths + fireconnect model tables, read 2026-09-16): short aliases
13
+ // (`glm-5p2`, `kimi-latest`) and full resource names (`accounts/fireworks/models/…`) are
14
+ // BOTH documented spellings, so both are accepted.
15
+ // • the embedding ids are the Qwen3 Embedding table (docs.fireworks.ai "Generating
16
+ // embeddings": `fireworks/qwen3-embedding-8b` serverless; 4B/0.6B dedicated) — the twin
17
+ // serves the serverless one.
18
+ // • the rerank id is the Qwen3 Reranker table (docs.fireworks.ai "Reranking documents":
19
+ // `fireworks/qwen3-reranker-8b` serverless).
20
+ // • context windows are NOT sourced per-id here (the stub does not model context limits per
21
+ // model — the stub context is fixed at 8192 tokens, documented in fireworks-twin.ts); no
22
+ // capability asserts a per-model window. The catalog's job is the 404 REFUSAL, not metadata.
23
+ // A model absent from this table is deliberately unknown: the honest gap is the catalog's
24
+ // COVERAGE, filed as the fireworks.models.catalog_coverage todo — not a silent 200.
25
+
26
+ /** One catalog row: the id the inference plane accepts. */
27
+ export type FireworksCatalogModel = {
28
+ id: string;
29
+ /** Which inference endpoint families accept it. */
30
+ kind: 'chat' | 'embeddings' | 'rerank';
31
+ };
32
+
33
+ export const FIREWORKS_MODELS: FireworksCatalogModel[] = [
34
+ // Serverless chat — documented serving paths (short aliases + full resource names).
35
+ { id: 'kimi-latest', kind: 'chat' },
36
+ { id: 'kimi-fast-latest', kind: 'chat' },
37
+ { id: 'glm-latest', kind: 'chat' },
38
+ { id: 'glm-fast-latest', kind: 'chat' },
39
+ { id: 'glm-5p2', kind: 'chat' },
40
+ { id: 'glm-5p3-flash', kind: 'chat' },
41
+ { id: 'kimi-k2-instruct', kind: 'chat' },
42
+ { id: 'accounts/fireworks/models/kimi-k2-instruct', kind: 'chat' },
43
+ { id: 'accounts/fireworks/models/kimi-k2-instruct-0905', kind: 'chat' },
44
+ { id: 'accounts/fireworks/models/llama-v3p1-8b-instruct', kind: 'chat' },
45
+ // Serverless embeddings + rerank — the Qwen3 tables (the 4B/0.6B tiers are dedicated-only),
46
+ // plus the legacy Hugging-Face-style embedder the docs name explicitly ("These models are not
47
+ // in the Model Library … they still work on serverless if you pass the exact Hugging Face
48
+ // style id"). The earlier `accounts/fireworks/models/nomic-embed-text-v1_5` spelling this
49
+ // pack first used was never a vendor id — the sourced id is the one below.
50
+ { id: 'fireworks/qwen3-embedding-8b', kind: 'embeddings' },
51
+ { id: 'accounts/fireworks/models/qwen3-embedding-8b', kind: 'embeddings' },
52
+ { id: 'nomic-ai/nomic-embed-text-v1.5', kind: 'embeddings' },
53
+ { id: 'fireworks/qwen3-reranker-8b', kind: 'rerank' },
54
+ { id: 'accounts/fireworks/models/qwen3-reranker-8b', kind: 'rerank' },
55
+ ];
56
+
57
+ /** The static ids, for the refusal check. */
58
+ export const FIREWORKS_MODEL_IDS = new Set(FIREWORKS_MODELS.map((m) => m.id));
59
+
60
+ /** Which endpoint family a static model id belongs to (undefined = not a chat id). */
61
+ export function catalogKind(id: string): FireworksCatalogModel['kind'] | undefined {
62
+ return FIREWORKS_MODELS.find((m) => m.id === id)?.kind;
63
+ }
@@ -0,0 +1,191 @@
1
+ // The fireworks pack's scenario system on the kernel's ONE engine (@volter/world-core scenario.ts):
2
+ // Fireworks chat-completions vocabulary + the scripted-turn respond shape. THE KERNEL ENGINE IS NOT
3
+ // RE-IMPLEMENTED HERE — this file is only the pack's adapter (features, matchers, load
4
+ // validation) plus the realizer that turns a validated `respond` into Fireworks' own envelope.
5
+ //
6
+ // The handler FILE (handlers/fireworks.json in a world dir) is the only write surface; the doors
7
+ // (`GET /twin`, `GET /twin/scenario`) are read-only. Scenario support is twin-only scaffolding
8
+ // for eval worlds, NOT vendor surface, so it is deliberately absent from the capability
9
+ // manifest (ADDING_A_TWIN.md §6, "Scenario scripting") and gated by fireworks-scenario.test.ts.
10
+ import { getActiveWorldStore, parseScenarioDocument, type PackScenarioAdapter, ScenarioError, type ScenarioDocument, ScenarioEngine, type ScenarioFeatures } from '@volter/world-core';
11
+ import { contentToText, lastUserText } from './fireworks-stub.ts';
12
+ import type { FireworksMessageParam } from './fireworks-types.ts';
13
+
14
+ export type FireworksScenarioRequest = {
15
+ model: string;
16
+ messages: FireworksMessageParam[];
17
+ tools?: unknown;
18
+ /** Fireworks-specific: 'auto' | 'default' | 'flex' | 'priority' — only 'priority' is honored
19
+ * by the vendor; the rest behave as 'default'. Scriptable so a world can pin behavior per tier. */
20
+ serviceTier?: string;
21
+ };
22
+ export type FireworksScenarioEngine = ScenarioEngine<FireworksScenarioRequest>;
23
+
24
+ export type ScenarioToolCall = { name: string; arguments: Record<string, unknown>; id?: string };
25
+
26
+ /**
27
+ * What a fireworks handler may script. `reasoning` is Fireworks-specific (the separate
28
+ * `reasoning_content` field its reasoning models answer with). A scripted FAILURE is not
29
+ * vocabulary of its own — it is the kernel fault grammar (`fault: { kind: 'status', … }`,
30
+ * R15), rendered into Fireworks' envelope by the adapter's renderFault below.
31
+ */
32
+ export type FireworksScenarioRespond = {
33
+ text?: string;
34
+ reasoning?: string;
35
+ toolCalls?: ScenarioToolCall | ScenarioToolCall[];
36
+ finishReason?: 'stop' | 'length' | 'tool_calls';
37
+ };
38
+
39
+ export type ScriptedResult = {
40
+ text: string | null;
41
+ reasoning: string | null;
42
+ toolCalls: Array<{ id: string; type: 'function'; function: { name: string; arguments: string } }>;
43
+ finishReason: 'stop' | 'length' | 'tool_calls';
44
+ };
45
+
46
+ const RESPOND_KEYS = new Set(['text', 'reasoning', 'toolCalls', 'finishReason']);
47
+ const FINISH_REASONS = new Set(['stop', 'length', 'tool_calls']);
48
+ const SERVICE_TIERS = new Set(['auto', 'default', 'flex', 'priority']);
49
+ const nonEmptyString = (cond: unknown): cond is string => typeof cond === 'string' && cond.length > 0;
50
+
51
+ function lastToolResultNames(messages: FireworksMessageParam[]): Set<string> {
52
+ const names = new Set<string>();
53
+ const last = messages[messages.length - 1] as { role?: string; tool_call_id?: unknown } | undefined;
54
+ if (!last || last.role !== 'tool' || typeof last.tool_call_id !== 'string') return names;
55
+ for (const m of messages) {
56
+ const am = m as { role?: string; tool_calls?: Array<{ id?: unknown; function?: { name?: unknown } }> };
57
+ if (am.role !== 'assistant' || !Array.isArray(am.tool_calls)) continue;
58
+ for (const tc of am.tool_calls) {
59
+ if (tc?.id === last.tool_call_id && typeof tc?.function?.name === 'string') names.add(tc.function.name);
60
+ }
61
+ }
62
+ return names;
63
+ }
64
+
65
+ function toolNames(tools: unknown): string[] {
66
+ if (!Array.isArray(tools)) return [];
67
+ return (tools as Array<{ function?: { name?: unknown }; name?: unknown }>).map((t) =>
68
+ typeof t?.function?.name === 'string' ? t.function.name : typeof t?.name === 'string' ? t.name : null,
69
+ ).filter((n): n is string => n !== null);
70
+ }
71
+
72
+ export const fireworksScenarioAdapter: PackScenarioAdapter<FireworksScenarioRequest> = {
73
+ vendor: 'fireworks',
74
+ // R15 — a status fault in THIS vendor's envelope: the same `{error:{message, code, type}}`
75
+ // OpenAI-compat shape fireworks-twin.ts serves for its own refusals. The vendor's error table
76
+ // (docs.fireworks.ai/api-reference, "Common error codes") keys the type on the status: 429 is
77
+ // the serverless rate limit / deployment-capacity signal, ≥500 is the vendor's server-side
78
+ // family. The kernel adds the `retry-after` header when the fault carries retryAfterSeconds —
79
+ // renderFault must not add it itself.
80
+ renderFault: (f) => ({
81
+ body: {
82
+ error: {
83
+ message: f.message ?? (f.status === 429
84
+ ? 'Rate limit exceeded. Please retry your request.'
85
+ : f.status >= 500
86
+ ? 'Internal Server Error'
87
+ : 'The request was refused by a scripted fault.'),
88
+ code: f.status,
89
+ type: f.status === 429 ? 'rate_limit_exceeded' : f.status >= 500 ? 'internal_server_error' : 'invalid_request_error',
90
+ },
91
+ },
92
+ }),
93
+ features: (req): ScenarioFeatures => ({
94
+ model: req.model,
95
+ lastUserText: lastUserText(req.messages).slice(0, 300),
96
+ tools: toolNames(req.tools),
97
+ lastMessageIsToolResult: (req.messages[req.messages.length - 1] as { role?: string } | undefined)?.role === 'tool',
98
+ toolResultFor: [...lastToolResultNames(req.messages)],
99
+ serviceTier: req.serviceTier ?? 'default',
100
+ }),
101
+ matchers: {
102
+ modelEquals: (req, cond) => nonEmptyString(cond) && req.model === cond,
103
+ userTextIncludes: (req, cond) => nonEmptyString(cond) && lastUserText(req.messages).toLowerCase().includes(cond.toLowerCase()),
104
+ anyTextIncludes: (req, cond) => nonEmptyString(cond) && req.messages.map((m) => contentToText(m.content)).join('\n').toLowerCase().includes(cond.toLowerCase()),
105
+ lastMessageIsToolResult: (req, cond) => typeof cond === 'boolean' && ((req.messages[req.messages.length - 1] as { role?: string } | undefined)?.role === 'tool') === cond,
106
+ toolResultFor: (req, cond) => nonEmptyString(cond) && lastToolResultNames(req.messages).has(cond),
107
+ hasTool: (req, cond) => nonEmptyString(cond) && toolNames(req.tools).includes(cond),
108
+ // Fireworks-specific: script behaviour per service tier ('priority' is the only honored one).
109
+ serviceTierEquals: (req, cond) => nonEmptyString(cond) && (req.serviceTier ?? 'default') === cond,
110
+ },
111
+ text: (req) => req.messages.map((m) => contentToText(m.content)).join('\n'),
112
+ validateOn: (on) => {
113
+ for (const k of ['modelEquals', 'userTextIncludes', 'anyTextIncludes', 'toolResultFor', 'hasTool'] as const) {
114
+ if (on[k] !== undefined && (typeof on[k] !== 'string' || !on[k])) return `on.${k} is a non-empty string`;
115
+ }
116
+ if (on.lastMessageIsToolResult !== undefined && typeof on.lastMessageIsToolResult !== 'boolean') return 'on.lastMessageIsToolResult is a boolean';
117
+ if (on.serviceTierEquals !== undefined) {
118
+ if (typeof on.serviceTierEquals !== 'string' || !SERVICE_TIERS.has(on.serviceTierEquals)) {
119
+ return `on.serviceTierEquals is one of ${[...SERVICE_TIERS].join(', ')}`;
120
+ }
121
+ }
122
+ return null;
123
+ },
124
+ validateRespond: (respond) => {
125
+ if (typeof respond !== 'object' || respond === null || Array.isArray(respond)) return 'respond is an object { text?, reasoning?, toolCalls?, finishReason? }';
126
+ const r = respond as Record<string, unknown>;
127
+ for (const k of Object.keys(r)) if (!RESPOND_KEYS.has(k)) return `respond: unknown key "${k}" (valid: ${[...RESPOND_KEYS].join(', ')})`;
128
+ if (r.text !== undefined && typeof r.text !== 'string') return 'respond.text is a string';
129
+ if (r.reasoning !== undefined && typeof r.reasoning !== 'string') return 'respond.reasoning is a string';
130
+ if (r.finishReason !== undefined && (typeof r.finishReason !== 'string' || !FINISH_REASONS.has(r.finishReason))) return `respond.finishReason is one of ${[...FINISH_REASONS].join(', ')}`;
131
+ if (r.toolCalls !== undefined) {
132
+ for (const tc of Array.isArray(r.toolCalls) ? r.toolCalls : [r.toolCalls]) {
133
+ const t = tc as Record<string, unknown>;
134
+ if (!t || typeof t !== 'object' || Array.isArray(t)) return 'respond.toolCalls entries are objects';
135
+ if (typeof t.name !== 'string' || !t.name) return 'respond.toolCalls[].name is a non-empty string';
136
+ if (!t.arguments || typeof t.arguments !== 'object' || Array.isArray(t.arguments)) return 'respond.toolCalls[].arguments is an object';
137
+ if (t.id !== undefined && typeof t.id !== 'string') return 'respond.toolCalls[].id is a string';
138
+ }
139
+ }
140
+ if (r.text === undefined && r.toolCalls === undefined) return 'respond needs text or toolCalls';
141
+ return null;
142
+ },
143
+ };
144
+
145
+ /** Load + STRICTLY validate a scenario document. A malformed file throws at load with what is
146
+ * wrong — never a silent ignore-and-stub (a mis-typed rule falling back is a fake success). */
147
+ export function loadFireworksScenarioDocument(path: string): ScenarioDocument {
148
+ let parsed: unknown;
149
+ try {
150
+ // Read through the ACTIVE WorldStore, never the filesystem directly (runtime contract
151
+ // R12b): the handlers document is WORLD STATE, so a MemoryWorldStore / DO-backed world
152
+ // serves ITS OWN scenario instead of whatever happens to sit on the host disk — and the
153
+ // serve path stays workerd-clean. A missing document keeps the historical ENOENT wording,
154
+ // so the loud load-time failure reads byte-identically to the read it replaces.
155
+ const raw = getActiveWorldStore().read(path);
156
+ if (raw === null) throw new Error(`ENOENT: no such file or directory, open '${path}'`);
157
+ parsed = JSON.parse(raw);
158
+ } catch (e) {
159
+ throw new ScenarioError(`fireworks scenario: cannot read/parse ${path}: ${e instanceof Error ? e.message : String(e)}`);
160
+ }
161
+ return parseScenarioDocument(parsed, fireworksScenarioAdapter);
162
+ }
163
+
164
+ export function createFireworksScenarioEngine(document?: ScenarioDocument): FireworksScenarioEngine {
165
+ return new ScenarioEngine(fireworksScenarioAdapter, document);
166
+ }
167
+
168
+ /**
169
+ * Turn a validated `respond` into the pack's own faithful assistant turn.
170
+ *
171
+ * The tool-call id is derived from the scripted call's own content and its position in the
172
+ * handler (never a module-level counter), exactly as the non-scenario path derives its ids — a
173
+ * serve-path determinism requirement: a served response is a pure function of (request, stored
174
+ * state), so two identical scripted requests in one process must answer identically.
175
+ */
176
+ export function realizeFireworksRespond(respond: FireworksScenarioRespond): ScriptedResult {
177
+ const toolCalls: ScriptedResult['toolCalls'] = [];
178
+ const scripted = respond.toolCalls ? (Array.isArray(respond.toolCalls) ? respond.toolCalls : [respond.toolCalls]) : [];
179
+ scripted.forEach((tc, index) => {
180
+ const seed = `${index}:${tc.name}:${JSON.stringify(tc.arguments)}`;
181
+ let h = 0x811c9dc5;
182
+ for (let i = 0; i < seed.length; i++) { h ^= seed.charCodeAt(i); h = Math.imul(h, 0x01000193); }
183
+ toolCalls.push({ id: tc.id ?? `call_scripted_${(h >>> 0).toString(36)}`, type: 'function', function: { name: tc.name, arguments: JSON.stringify(tc.arguments) } });
184
+ });
185
+ return {
186
+ text: respond.text ?? (toolCalls.length ? null : ''),
187
+ reasoning: respond.reasoning ?? null,
188
+ toolCalls,
189
+ finishReason: respond.finishReason ?? (toolCalls.length ? 'tool_calls' : 'stop'),
190
+ };
191
+ }
@@ -0,0 +1,153 @@
1
+ // Fireworks twin HTTP server — serve the full Fireworks twin handler over HTTP so the real
2
+ // `fireworks` Python SDK (pointed at `base_url: http://127.0.0.1:<port>/inference/v1`), the
3
+ // `@ai-sdk/fireworks` provider and plain OpenAI clients (pointed at
4
+ // `http://127.0.0.1:<port>/inference/v1`) work UNMODIFIED, and the Gateway control plane answers
5
+ // at `/v1/accounts/{account_id}/…`. JSON bodies pass straight through. Writable by default; pass
6
+ // readOnly to reject mutations (D3).
7
+ //
8
+ // Streaming: when a streamable request body has `"stream": true`, the server constructs a REAL SSE
9
+ // response by feeding the handler an sseSink that writes each chunk in the `data: <json>\n\n` wire
10
+ // format, ending with `data: [DONE]\n\n` on the OpenAI-compat paths (the Anthropic-compat
11
+ // /v1/messages ends with its `message_stop` event — no [DONE] sentinel; that is the vendor's own
12
+ // shape). The handler itself stays socket-free — the sink is the only place a socket is touched,
13
+ // on the live HTTP path.
14
+ //
15
+ // FETCH-FIRST (runtime contract R12b): the surface is the plain `createFireworksTwinFetch` and the
16
+ // SERVER is one line of `serveHttp` around it. This is a CUSTOM fetch, not the kernel adapter
17
+ // (`createTwinFetchFromHandler`): SSE streaming on THREE paths (chat/completions, completions,
18
+ // messages) genuinely exceeds the common shape — groq-server.ts is the reference.
19
+ import { serveHttp, twinManifest, worldNow } from '@volter/world-core';
20
+ import { createFireworksScenarioEngine, type FireworksScenarioEngine, loadFireworksScenarioDocument } from './fireworks-scenario.ts';
21
+ import { FIREWORKS_INFERENCE_PREFIX, handleFireworksTwinRequest } from './fireworks-twin.ts';
22
+ import type { SseEvent } from './fireworks-twin.ts';
23
+
24
+ function wantsStream(body: string): boolean {
25
+ if (!body) return false;
26
+ try {
27
+ return (JSON.parse(body) as { stream?: unknown })?.stream === true;
28
+ } catch {
29
+ return false;
30
+ }
31
+ }
32
+
33
+ function encodeSse(event: SseEvent): string {
34
+ if (event.done) return 'data: [DONE]\n\n';
35
+ return `data: ${JSON.stringify(event.data)}\n\n`;
36
+ }
37
+
38
+ // The three inference paths Fireworks streams. The Anthropic-compat /v1/messages streams in the
39
+ // SAME SSE transport but with its own event vocabulary — and the handler already ends its sequence
40
+ // with `message_stop`, so the [DONE] sentinel must NOT be appended there (Anthropic streams do not
41
+ // carry one; appending one would be a wire lie the official clients parse as an unknown event).
42
+ const STREAMABLE = new Set([
43
+ `${FIREWORKS_INFERENCE_PREFIX}/chat/completions`,
44
+ `${FIREWORKS_INFERENCE_PREFIX}/completions`,
45
+ `${FIREWORKS_INFERENCE_PREFIX}/messages`,
46
+ ]);
47
+
48
+ /** Options every Fireworks-twin HTTP surface needs, independent of who owns the socket. */
49
+ export interface FireworksTwinFetchOptions {
50
+ root?: string;
51
+ readOnly?: boolean;
52
+ scenarioPath?: string;
53
+ }
54
+
55
+ export function createFireworksTwinFetch(options: FireworksTwinFetchOptions): (request: Request) => Promise<Response> {
56
+ const readOnly = options.readOnly ?? false;
57
+ // Scenario scripting (fireworks-scenario.ts): a JSON scenario file — via the scenarioPath option
58
+ // or the TWIN_FIREWORKS_SCENARIO env var — scripts chat completions. Loaded ONCE at startup (a
59
+ // malformed file fails loudly here, never silently).
60
+ const scenarioPath = options.scenarioPath ?? process.env.TWIN_FIREWORKS_SCENARIO;
61
+ const scenarioEngine: FireworksScenarioEngine | undefined = scenarioPath ? createFireworksScenarioEngine(loadFireworksScenarioDocument(scenarioPath)) : undefined;
62
+ return async function fireworksTwinFetch(request: Request): Promise<Response> {
63
+ const url = new URL(request.url);
64
+ // THE READ DOORS (TWIN-PROGRAMMING-MODEL): discovery + inspection, read-only.
65
+ if (request.method === 'GET' && url.pathname.replace(/\/+$/, '') === '/twin') {
66
+ return Response.json(twinManifest({
67
+ vendor: 'fireworks',
68
+ twinOf: 'Fireworks AI API (inference plane at /inference/v1, Gateway control plane at /v1/accounts/{account_id})',
69
+ stateSentence: 'Seed control-plane state through the ordinary Gateway API (POST /v1/accounts/{account_id}/deployments?deploymentId=…) with any key.',
70
+ behaviorSentence: "Chat completions are scripted by MSW-shaped handlers in the world dir (handlers/fireworks.json): {on:{userTextIncludes|anyTextIncludes|modelEquals|lastMessageIsToolResult|serviceTierEquals|nthCall}, respond:{text|reasoning, finishReason?} | fault:{kind:'status'|'slow'|'drop', ...}, once?, phase?}. Unmatched requests answer a labeled stub naming this door.",
71
+ exampleHandler: { on: { userTextIncludes: 'summarize' }, respond: { text: 'Scripted summary.' }, once: true },
72
+ engine: scenarioEngine as never,
73
+ }));
74
+ }
75
+ if (request.method === 'GET' && url.pathname.replace(/\/+$/, '') === '/twin/scenario') {
76
+ return Response.json(scenarioEngine ? scenarioEngine.status() : { vendor: 'fireworks', handlers: [], misses: 0, recentMisses: [] });
77
+ }
78
+
79
+ const path = url.pathname + (url.search || '');
80
+ const cleanPath = url.pathname.replace(/\/+$/, '');
81
+ const passHeaders: Record<string, string> = {};
82
+ for (const k of ['authorization', 'anthropic-version']) {
83
+ const v = request.headers.get(k);
84
+ if (v !== null) passHeaders[k] = v;
85
+ }
86
+
87
+ let body = '';
88
+ if (request.method !== 'GET') body = await request.text();
89
+
90
+ // Streaming POST → a real text/event-stream response built from the sink.
91
+ //
92
+ // Events are COLLECTED first, then framed. The handler is synchronous-fast, so buffering
93
+ // costs nothing — and it is what lets a PRE-STREAM failure (validation 422/400, auth 401,
94
+ // rate-limit 429) answer with its REAL status and the vendor's JSON error envelope. Emitting
95
+ // the refusal as a lone `data:` frame inside a 200 text/event-stream was a FAKE SUCCESS the
96
+ // in-process path could not see. Fireworks rejects a bad request BEFORE opening the event
97
+ // stream; this twin decides every refusal before the first chunk.
98
+ if (!readOnly && request.method.toUpperCase() === 'POST' && STREAMABLE.has(cleanPath) && wantsStream(body)) {
99
+ const events: SseEvent[] = [];
100
+ const { status, body: out, headers: errHeaders } = await handleFireworksTwinRequest({
101
+ ...(scenarioEngine ? { scenarioEngine } : {}),
102
+ method: request.method, path, body, readOnly, occurredAt: worldNow(), headers: passHeaders,
103
+ ...(options.root !== undefined ? { root: options.root } : {}), sseSink: (e: SseEvent) => events.push(e),
104
+ });
105
+ if (status >= 400) {
106
+ return new Response(JSON.stringify(out), { status, headers: { 'content-type': 'application/json', 'x-request-id': 'req_twin', ...(errHeaders ?? {}) } });
107
+ }
108
+ // /v1/messages ends with its own message_stop; the OpenAI-compat paths end with [DONE].
109
+ const anthropicStream = cleanPath === `${FIREWORKS_INFERENCE_PREFIX}/messages`;
110
+ const stream = new ReadableStream<Uint8Array>({
111
+ start(controller) {
112
+ const enc = new TextEncoder();
113
+ for (const e of events) {
114
+ if (e.done && anthropicStream) continue; // no [DONE] on the Anthropic surface
115
+ controller.enqueue(enc.encode(encodeSse(e)));
116
+ }
117
+ controller.close();
118
+ },
119
+ });
120
+ return new Response(stream, { headers: { 'content-type': 'text/event-stream; charset=utf-8', 'cache-control': 'no-cache', 'x-request-id': 'req_twin' } });
121
+ }
122
+
123
+ // Defensive: a handler throw must become a vendor-shaped 500, never an escaped exception
124
+ // (ADDING_A_TWIN.md §8, "In-process server harnesses … contain handler errors").
125
+ let status: number;
126
+ let out: unknown;
127
+ let outHeaders: Record<string, string> | undefined;
128
+ try {
129
+ const res = await handleFireworksTwinRequest({
130
+ ...(scenarioEngine ? { scenarioEngine } : {}),
131
+ method: request.method, path, body, readOnly,
132
+ occurredAt: worldNow(), headers: passHeaders,
133
+ ...(options.root !== undefined ? { root: options.root } : {}),
134
+ });
135
+ status = res.status; out = res.body; outHeaders = res.headers;
136
+ } catch (err) {
137
+ return new Response(JSON.stringify({ error: { message: `twin handler error: ${err instanceof Error ? err.message : String(err)}`, code: 500 } }), {
138
+ status: 500, headers: { 'content-type': 'application/json', 'x-request-id': 'req_twin' },
139
+ });
140
+ }
141
+
142
+ return new Response(JSON.stringify(out), { status, headers: { 'content-type': 'application/json', 'x-request-id': 'req_twin', ...(outHeaders ?? {}) } });
143
+ };
144
+ }
145
+
146
+ export async function createFireworksTwinServer(options: { root?: string; port?: number; readOnly?: boolean; scenarioPath?: string }): Promise<{ port: number; stop: () => void }> {
147
+ const server = await serveHttp({
148
+ port: options.port ?? 0,
149
+ idleTimeout: 60,
150
+ fetch: createFireworksTwinFetch(options),
151
+ });
152
+ return { port: server.port ?? options.port ?? 0, stop: () => server.stop(true) };
153
+ }
@@ -0,0 +1,83 @@
1
+ // Fireworks stub core — the deterministic-content primitives shared by the inference handlers.
2
+ // Mirrors groq-stub.ts / ai-gateway-stub.ts: NO model runs here, so generated content is a
3
+ // DETERMINISTIC, clearly-labeled function of the prompt — never pretending to be real inference.
4
+ // Everything AROUND the stub (the envelope, the field names, the rejections) is vendor-faithful.
5
+
6
+ /** A cheap stable hash (FNV-1a, 32-bit) — the same primitive the sibling packs use. */
7
+ export function fnv1a(text: string): number {
8
+ let h = 0x811c9dc5;
9
+ for (let i = 0; i < text.length; i++) {
10
+ h ^= text.charCodeAt(i);
11
+ h = Math.imul(h, 0x01000193);
12
+ }
13
+ return h >>> 0;
14
+ }
15
+
16
+ /** Flatten message/prompt content to plain text (strings joined; content blocks' text parts taken). */
17
+ export function contentToText(content: unknown): string {
18
+ if (typeof content === 'string') return content;
19
+ if (Array.isArray(content)) {
20
+ return content
21
+ .map((part) => (typeof part === 'string' ? part : typeof (part as { text?: unknown })?.text === 'string' ? ((part as { text: string }).text) : ''))
22
+ .join(' ');
23
+ }
24
+ return '';
25
+ }
26
+
27
+ /** ~1 token per 4 characters — the same rough estimate the sibling packs use. */
28
+ export function estimateTokens(text: string): number {
29
+ return Math.max(1, Math.ceil(text.length / 4));
30
+ }
31
+
32
+ export function countPromptTokens(messages: Array<{ content?: unknown }>): number {
33
+ return messages.reduce((sum, m) => sum + estimateTokens(contentToText(m.content)), 0);
34
+ }
35
+
36
+ export function lastUserText(messages: Array<{ role?: string; content?: unknown }>): string {
37
+ for (let i = messages.length - 1; i >= 0; i--) {
38
+ if (messages[i].role === 'user') return contentToText(messages[i].content);
39
+ }
40
+ return '';
41
+ }
42
+
43
+ /** THE LABEL. Every generated completion carries it, so nothing can be mistaken for real output. */
44
+ export function stubAssistantText(model: string, prompt: string): string {
45
+ const h = fnv1a(`${model}\0${prompt}`);
46
+ return `[twin-stub:${model}] Deterministic response to: ${prompt.slice(0, 120) || '(empty prompt)'} (hash ${h.toString(16)})`;
47
+ }
48
+
49
+ /** Deterministic reasoning content, for reasoning_effort on supported models. */
50
+ export function stubReasoningText(model: string, prompt: string): string {
51
+ const h = fnv1a(`reasoning\0${model}\0${prompt}`);
52
+ return `[twin-stub:${model}] reasoning trace (hash ${h.toString(16)})`;
53
+ }
54
+
55
+ /** Deterministic pseudo-embedding: L2-normalized, derived from the input hash. `dimensions` must
56
+ * already be a validated positive integer — the twin's handler refuses anything else BEFORE
57
+ * calling this (a client-sized allocation from an unvalidated number is a one-request DoS). The
58
+ * guard here is the class-level backstop: NO array in this pack is ever sized from a request
59
+ * value without a bound. */
60
+ export function pseudoEmbedding(text: string, dimensions: number): number[] {
61
+ if (!Number.isInteger(dimensions) || dimensions < 1 || dimensions > 4096) {
62
+ throw new RangeError(`pseudoEmbedding: dimensions must be an integer in [1, 4096], got ${dimensions}`);
63
+ }
64
+ const vec = new Array<number>(dimensions);
65
+ let h = fnv1a(text);
66
+ for (let i = 0; i < dimensions; i++) {
67
+ h = Math.imul(h ^ (i + 1), 0x01000193) >>> 0;
68
+ vec[i] = (h / 0xffffffff) * 2 - 1;
69
+ }
70
+ const norm = Math.hypot(...vec) || 1;
71
+ return vec.map((v) => v / norm);
72
+ }
73
+
74
+ /** Deterministic relevance score in [0, 1] for rerank, biased by shared-token overlap so the
75
+ * ORDER of results responds to the query (a constant ordering would be a fake). */
76
+ export function stubRelevanceScore(query: string, document: string): number {
77
+ const q = new Set(query.toLowerCase().split(/\W+/).filter(Boolean));
78
+ const d = new Set(document.toLowerCase().split(/\W+/).filter(Boolean));
79
+ let overlap = 0;
80
+ for (const t of q) if (d.has(t)) overlap++;
81
+ const base = (fnv1a(`${query}\0${document}`) % 1000) / 10000; // 0..0.1 jitter
82
+ return Math.min(1, overlap / Math.max(1, q.size) + base);
83
+ }