@volter/twin-fireworks 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +202 -0
- package/README.md +184 -0
- package/dist/src/cli.d.ts +2 -0
- package/dist/src/cli.js +28 -0
- package/dist/src/fireworks-budget.d.ts +54 -0
- package/dist/src/fireworks-budget.js +146 -0
- package/dist/src/fireworks-capabilities.d.ts +4 -0
- package/dist/src/fireworks-capabilities.js +1205 -0
- package/dist/src/fireworks-conformance.d.ts +14 -0
- package/dist/src/fireworks-conformance.js +514 -0
- package/dist/src/fireworks-connector.d.ts +168 -0
- package/dist/src/fireworks-connector.js +641 -0
- package/dist/src/fireworks-models.d.ts +11 -0
- package/dist/src/fireworks-models.js +53 -0
- package/dist/src/fireworks-scenario.d.ts +55 -0
- package/dist/src/fireworks-scenario.js +171 -0
- package/dist/src/fireworks-server.d.ts +16 -0
- package/dist/src/fireworks-server.js +144 -0
- package/dist/src/fireworks-stub.d.ts +26 -0
- package/dist/src/fireworks-stub.js +78 -0
- package/dist/src/fireworks-twin.d.ts +51 -0
- package/dist/src/fireworks-twin.js +1426 -0
- package/dist/src/fireworks-types.d.ts +212 -0
- package/dist/src/fireworks-types.js +4 -0
- package/dist/src/index.d.ts +9 -0
- package/dist/src/index.js +105 -0
- package/package.json +52 -0
- package/src/cli.ts +27 -0
- package/src/fireworks-budget.ts +172 -0
- package/src/fireworks-capabilities.ts +1229 -0
- package/src/fireworks-conformance.ts +542 -0
- package/src/fireworks-connector.ts +700 -0
- package/src/fireworks-models.ts +63 -0
- package/src/fireworks-scenario.ts +191 -0
- package/src/fireworks-server.ts +153 -0
- package/src/fireworks-stub.ts +83 -0
- package/src/fireworks-twin.ts +1427 -0
- package/src/fireworks-types.ts +165 -0
- package/src/index.ts +134 -0
|
@@ -0,0 +1,212 @@
|
|
|
1
|
+
/** One message in a chat request. `reasoning_content` on assistant messages is a Fireworks
|
|
2
|
+
* extension (reasoning models return it; `reasoning_history` controls whether it is replayed). */
|
|
3
|
+
export type FireworksMessageParam = {
|
|
4
|
+
role: string;
|
|
5
|
+
content?: unknown;
|
|
6
|
+
name?: string;
|
|
7
|
+
tool_call_id?: string;
|
|
8
|
+
tool_calls?: Array<{
|
|
9
|
+
id: string;
|
|
10
|
+
type: 'function';
|
|
11
|
+
function: {
|
|
12
|
+
name: string;
|
|
13
|
+
arguments: string;
|
|
14
|
+
};
|
|
15
|
+
}>;
|
|
16
|
+
reasoning_content?: string | null;
|
|
17
|
+
};
|
|
18
|
+
/** The `service_tier` enum is Fireworks' own; only 'priority' is honored — every other value is
|
|
19
|
+
* treated as 'default' (documented; NOT an error). */
|
|
20
|
+
export type FireworksServiceTier = 'auto' | 'default' | 'flex' | 'priority';
|
|
21
|
+
/** Fireworks' documented OpenAI difference: the default is `truncate`, not `error`. */
|
|
22
|
+
export type FireworksContextLengthExceededBehavior = 'error' | 'truncate';
|
|
23
|
+
export type FireworksChatRequest = {
|
|
24
|
+
model: string;
|
|
25
|
+
messages: FireworksMessageParam[];
|
|
26
|
+
temperature?: number | null;
|
|
27
|
+
top_p?: number | null;
|
|
28
|
+
top_k?: number | null;
|
|
29
|
+
frequency_penalty?: number | null;
|
|
30
|
+
presence_penalty?: number | null;
|
|
31
|
+
max_tokens?: number | null;
|
|
32
|
+
max_completion_tokens?: number | null;
|
|
33
|
+
n?: number | null;
|
|
34
|
+
stop?: string | string[] | null;
|
|
35
|
+
stream?: boolean | null;
|
|
36
|
+
stream_options?: {
|
|
37
|
+
include_usage?: boolean;
|
|
38
|
+
buffer_tokens?: number;
|
|
39
|
+
buffer_ms?: number;
|
|
40
|
+
buffer_mode?: string;
|
|
41
|
+
} | null;
|
|
42
|
+
tools?: Array<Record<string, unknown>>;
|
|
43
|
+
tool_choice?: unknown;
|
|
44
|
+
parallel_tool_calls?: boolean | null;
|
|
45
|
+
response_format?: Record<string, unknown> | null;
|
|
46
|
+
reasoning_effort?: string | number | boolean | null;
|
|
47
|
+
reasoning_history?: 'disabled' | 'interleaved' | 'preserved' | null;
|
|
48
|
+
service_tier?: FireworksServiceTier | null;
|
|
49
|
+
context_length_exceeded_behavior?: FireworksContextLengthExceededBehavior | null;
|
|
50
|
+
seed?: number | null;
|
|
51
|
+
user?: string | null;
|
|
52
|
+
echo?: boolean | null;
|
|
53
|
+
logprobs?: boolean | null;
|
|
54
|
+
top_logprobs?: number | null;
|
|
55
|
+
logit_bias?: Record<string, number> | null;
|
|
56
|
+
metadata?: Record<string, unknown> | null;
|
|
57
|
+
};
|
|
58
|
+
export type FireworksChatMessage = {
|
|
59
|
+
role: 'assistant';
|
|
60
|
+
content: string | null;
|
|
61
|
+
reasoning_content?: string | null;
|
|
62
|
+
tool_calls?: Array<{
|
|
63
|
+
id: string;
|
|
64
|
+
type: 'function';
|
|
65
|
+
function: {
|
|
66
|
+
name: string;
|
|
67
|
+
arguments: string;
|
|
68
|
+
};
|
|
69
|
+
}>;
|
|
70
|
+
};
|
|
71
|
+
/** The response envelope: `usage` is nullable on chat (absent when `echo`/logprobs paths skip it),
|
|
72
|
+
* unlike `Completion` below where the spec marks it required. */
|
|
73
|
+
export type FireworksChatCompletion = {
|
|
74
|
+
id: string;
|
|
75
|
+
object: 'chat.completion';
|
|
76
|
+
created: number;
|
|
77
|
+
model: string;
|
|
78
|
+
choices: Array<{
|
|
79
|
+
index: number;
|
|
80
|
+
message: FireworksChatMessage;
|
|
81
|
+
finish_reason: string | null;
|
|
82
|
+
logprobs?: unknown | null;
|
|
83
|
+
token_ids?: number[] | null;
|
|
84
|
+
}>;
|
|
85
|
+
usage?: {
|
|
86
|
+
prompt_tokens: number;
|
|
87
|
+
total_tokens: number;
|
|
88
|
+
completion_tokens: number;
|
|
89
|
+
} | null;
|
|
90
|
+
};
|
|
91
|
+
/** Legacy `/v1/completions`: the spec marks `usage` REQUIRED here (chat's is nullable). */
|
|
92
|
+
export type FireworksCompletion = {
|
|
93
|
+
id: string;
|
|
94
|
+
object: 'text_completion';
|
|
95
|
+
created: number;
|
|
96
|
+
model: string;
|
|
97
|
+
choices: Array<{
|
|
98
|
+
index: number;
|
|
99
|
+
text: string;
|
|
100
|
+
finish_reason: string | null;
|
|
101
|
+
logprobs?: unknown | null;
|
|
102
|
+
}>;
|
|
103
|
+
usage: {
|
|
104
|
+
prompt_tokens: number;
|
|
105
|
+
total_tokens: number;
|
|
106
|
+
completion_tokens: number;
|
|
107
|
+
};
|
|
108
|
+
};
|
|
109
|
+
/** The Responses API (`/v1/responses`): `id` is nullable when `store=false` (the vendor's own
|
|
110
|
+
* contract — a non-stored response has no id to retrieve later). */
|
|
111
|
+
export type FireworksResponseObject = {
|
|
112
|
+
id: string | null;
|
|
113
|
+
object: 'response';
|
|
114
|
+
created_at: number;
|
|
115
|
+
status: 'completed' | 'in_progress' | 'incomplete' | 'failed' | 'cancelled';
|
|
116
|
+
model: string;
|
|
117
|
+
output: Array<FireworksResponseOutputItem>;
|
|
118
|
+
usage?: {
|
|
119
|
+
prompt_tokens: number;
|
|
120
|
+
completion_tokens: number;
|
|
121
|
+
total_tokens: number;
|
|
122
|
+
} | null;
|
|
123
|
+
error?: unknown | null;
|
|
124
|
+
incomplete_details?: unknown | null;
|
|
125
|
+
instructions?: unknown | null;
|
|
126
|
+
max_output_tokens?: number | null;
|
|
127
|
+
metadata?: Record<string, unknown> | null;
|
|
128
|
+
parallel_tool_calls?: boolean | null;
|
|
129
|
+
previous_response_id?: string | null;
|
|
130
|
+
reasoning?: unknown | null;
|
|
131
|
+
store?: boolean | null;
|
|
132
|
+
temperature?: number | null;
|
|
133
|
+
text?: unknown | null;
|
|
134
|
+
tool_choice?: unknown;
|
|
135
|
+
tools?: Array<Record<string, unknown>> | null;
|
|
136
|
+
top_p?: number | null;
|
|
137
|
+
truncation?: string | null;
|
|
138
|
+
user?: string | null;
|
|
139
|
+
};
|
|
140
|
+
export type FireworksResponseOutputItem = {
|
|
141
|
+
type: 'message';
|
|
142
|
+
id: string;
|
|
143
|
+
role: 'assistant';
|
|
144
|
+
status: 'in_progress' | 'completed';
|
|
145
|
+
content: Array<{
|
|
146
|
+
type: string;
|
|
147
|
+
text?: string;
|
|
148
|
+
}>;
|
|
149
|
+
} | {
|
|
150
|
+
type: 'function_call';
|
|
151
|
+
id: string;
|
|
152
|
+
call_id: string;
|
|
153
|
+
name: string;
|
|
154
|
+
arguments: string;
|
|
155
|
+
status?: string;
|
|
156
|
+
} | {
|
|
157
|
+
type: 'function_call_output';
|
|
158
|
+
tool_call_id: string;
|
|
159
|
+
output: unknown;
|
|
160
|
+
};
|
|
161
|
+
/** Anthropic-compatible `/v1/messages`: the response `content` is an array of typed blocks, and
|
|
162
|
+
* `stop_reason` is Anthropic's enum (never OpenAI's `finish_reason`). */
|
|
163
|
+
export type FireworksAnthropicMessage = {
|
|
164
|
+
id: string;
|
|
165
|
+
type: 'message';
|
|
166
|
+
role: 'assistant';
|
|
167
|
+
content: Array<FireworksAnthropicContentBlock>;
|
|
168
|
+
model: string;
|
|
169
|
+
stop_reason: 'end_turn' | 'max_tokens' | 'stop_sequence' | 'tool_use' | 'pause_turn' | 'refusal' | null;
|
|
170
|
+
stop_sequence: string | null;
|
|
171
|
+
usage?: {
|
|
172
|
+
input_tokens: number;
|
|
173
|
+
output_tokens: number;
|
|
174
|
+
};
|
|
175
|
+
};
|
|
176
|
+
export type FireworksAnthropicContentBlock = {
|
|
177
|
+
type: 'text';
|
|
178
|
+
text: string;
|
|
179
|
+
citations?: unknown[] | null;
|
|
180
|
+
} | {
|
|
181
|
+
type: 'thinking';
|
|
182
|
+
thinking: string;
|
|
183
|
+
signature: string;
|
|
184
|
+
} | {
|
|
185
|
+
type: 'redacted_thinking';
|
|
186
|
+
data: string;
|
|
187
|
+
} | {
|
|
188
|
+
type: 'tool_use';
|
|
189
|
+
id: string;
|
|
190
|
+
name: string;
|
|
191
|
+
input: unknown;
|
|
192
|
+
};
|
|
193
|
+
/** The Anthropic error envelope: `{ type: 'error', error: { type, message }, request_id? }` — a
|
|
194
|
+
* DIFFERENT envelope from the OpenAI-compat plane's `{ error: { … } }`. */
|
|
195
|
+
export type FireworksAnthropicErrorType = 'invalid_request_error' | 'authentication_error' | 'billing_error' | 'permission_error' | 'not_found_error' | 'rate_limit_error' | 'timeout_error' | 'api_error' | 'overloaded_error';
|
|
196
|
+
/** google.rpc-style status, embedded on every resource as `status` and (per the vendor's own
|
|
197
|
+
* comment) mimicking google/rpc/status.proto. */
|
|
198
|
+
export type FireworksGatewayStatus = {
|
|
199
|
+
code: FireworksGatewayCode;
|
|
200
|
+
message?: string;
|
|
201
|
+
} | null;
|
|
202
|
+
/** The gatewayCode enum — the gRPC canonical codes, as strings. */
|
|
203
|
+
export type FireworksGatewayCode = 'OK' | 'CANCELLED' | 'UNKNOWN' | 'INVALID_ARGUMENT' | 'DEADLINE_EXCEEDED' | 'NOT_FOUND' | 'ALREADY_EXISTS' | 'PERMISSION_DENIED' | 'UNAUTHENTICATED' | 'RESOURCE_EXHAUSTED' | 'FAILED_PRECONDITION' | 'ABORTED' | 'OUT_OF_RANGE' | 'UNIMPLEMENTED' | 'INTERNAL' | 'UNAVAILABLE' | 'DATA_LOSS';
|
|
204
|
+
/** The list envelope every Gateway list operation answers. */
|
|
205
|
+
export type FireworksListEnvelope<T> = {
|
|
206
|
+
items?: never;
|
|
207
|
+
} & {
|
|
208
|
+
[K in string]: T[];
|
|
209
|
+
} & {
|
|
210
|
+
nextPageToken?: string | null;
|
|
211
|
+
totalSize?: number | null;
|
|
212
|
+
};
|
|
@@ -0,0 +1,4 @@
|
|
|
1
|
+
// Fireworks wire shapes, typed narrowly — the subsets of the vendor's real response shapes this
|
|
2
|
+
// pack reads and writes, grounded in the first-party OpenAPI specs (see specSource in index.ts).
|
|
3
|
+
// Where Fireworks differs from OpenAI the difference is NAMED here, not papered over.
|
|
4
|
+
export {};
|
|
@@ -0,0 +1,9 @@
|
|
|
1
|
+
export { handleFireworksTwinRequest, FIREWORKS_INFERENCE_PREFIX, FIREWORKS_ACCOUNTS_PREFIX } from './fireworks-twin.js';
|
|
2
|
+
export type { FireworksRequest, FireworksResponseEnvelope } from './fireworks-twin.js';
|
|
3
|
+
export { createFireworksTwinFetch, createFireworksTwinServer, type FireworksTwinFetchOptions } from './fireworks-server.js';
|
|
4
|
+
export { fullSyncFireworks, fireworksRequestForAction, liveFireworksExecute, pullFireworksState, pushFireworksAction, pushPendingFireworksActions, syncFireworksFromReal, unpushableReason, } from './fireworks-connector.js';
|
|
5
|
+
export type { FireworksExecute, LiveFireworksOptions } from './fireworks-connector.js';
|
|
6
|
+
export { FIREWORKS_BUDGET_CEILING, FIREWORKS_BUDGET_MAX_RETRY_AFTER_S, FIREWORKS_BUDGET_WINDOW_MS, FIREWORKS_CALL_WEIGHTS, FIREWORKS_RATE_BUDGET, FireworksBudget, FireworksBudgetError, fireworksBudgetPath, fireworksCallWeight, } from './fireworks-budget.js';
|
|
7
|
+
export type { FireworksBudgetErrorKind, FireworksBudgetOptions, FireworksBudgetReservation, FireworksBudgetSnapshot } from './fireworks-budget.js';
|
|
8
|
+
import type { TwinPack } from '@volter/world-core';
|
|
9
|
+
export declare const pack: TwinPack;
|
|
@@ -0,0 +1,105 @@
|
|
|
1
|
+
// @volter/twin-fireworks — the Fireworks twin (one vendor, one package), built on the shared
|
|
2
|
+
// @volter/world-core kernel. Fireworks serves TWO API planes on ONE host (api.fireworks.ai):
|
|
3
|
+
//
|
|
4
|
+
// • the INFERENCE plane — `https://api.fireworks.ai/inference/v1/…`, OpenAI-compatible
|
|
5
|
+
// (chat/completions, completions, responses, embeddings, rerank) plus an Anthropic-compatible
|
|
6
|
+
// POST /v1/messages. Generative output is a labeled deterministic stub; the wire protocol and
|
|
7
|
+
// Fireworks' documented OpenAI differences are the fidelity surface.
|
|
8
|
+
// • the CONTROL plane — the Gateway REST API (`https://api.fireworks.ai/v1/accounts/{account_id}/…`):
|
|
9
|
+
// deployments, datasets, fine-tuning jobs, users/apiKeys, secrets, models, quotas — a stateful
|
|
10
|
+
// google.rpc-style surface (list envelopes {items, nextPageToken, totalSize}, verb-suffixed
|
|
11
|
+
// custom methods like `:cancel`, ids passed via query params on writes).
|
|
12
|
+
//
|
|
13
|
+
// THE DETERMINISTIC STUB IS THE ANSWER: the twin runs no model, so inference returns a
|
|
14
|
+
// DETERMINISTIC STUB completion (clearly labeled `[twin-stub:<model>]`), never pretending to be
|
|
15
|
+
// real inference. Everything around it — the wire protocol, the rejections, the state — is faithful.
|
|
16
|
+
// (Conformance + capability tooling live in @volter/world-tooling, a dev dependency — NOT
|
|
17
|
+
// re-exported here, per E2.)
|
|
18
|
+
export { handleFireworksTwinRequest, FIREWORKS_INFERENCE_PREFIX, FIREWORKS_ACCOUNTS_PREFIX } from "./fireworks-twin.js";
|
|
19
|
+
export { createFireworksTwinFetch, createFireworksTwinServer } from "./fireworks-server.js";
|
|
20
|
+
export { fullSyncFireworks, fireworksRequestForAction, liveFireworksExecute, pullFireworksState, pushFireworksAction, pushPendingFireworksActions, syncFireworksFromReal, unpushableReason, } from "./fireworks-connector.js";
|
|
21
|
+
// The client-side rate budget — the fail-closed backstop `liveFireworksExecute` routes every live
|
|
22
|
+
// request through. The MECHANISM is the kernel's shared, vendor-agnostic `RateBudget`; what lives
|
|
23
|
+
// here is Fireworks' DECLARATION (window/ceiling/per-endpoint weights) plus the vendor-bound
|
|
24
|
+
// bindings. Exported so an operator can inspect spend (`snapshot`) and so a caller can catch
|
|
25
|
+
// `FireworksBudgetError` by type; there is deliberately no export that disables the guard.
|
|
26
|
+
export { FIREWORKS_BUDGET_CEILING, FIREWORKS_BUDGET_MAX_RETRY_AFTER_S, FIREWORKS_BUDGET_WINDOW_MS, FIREWORKS_CALL_WEIGHTS, FIREWORKS_RATE_BUDGET, FireworksBudget, FireworksBudgetError, fireworksBudgetPath, fireworksCallWeight, } from "./fireworks-budget.js";
|
|
27
|
+
import { FIREWORKS_RATE_BUDGET as RATE_BUDGET } from "./fireworks-budget.js";
|
|
28
|
+
import { performFireworksAction, syncFireworksFromRemote } from "./fireworks-connector.js";
|
|
29
|
+
export const pack = {
|
|
30
|
+
vendor: 'fireworks',
|
|
31
|
+
// The SAME object fireworks-budget.ts declares at module load — one source of truth, so
|
|
32
|
+
// registering the pack and importing the connector can never arm two different ceilings.
|
|
33
|
+
rateBudget: RATE_BUDGET,
|
|
34
|
+
transport: 'rest',
|
|
35
|
+
// PROTOCOL 2 (pack contract Part 3): the pack is a plugin — its wire, its tree, and its half of
|
|
36
|
+
// the real state system. The control plane is the stateful surface (deployments, datasets,
|
|
37
|
+
// fine-tuning jobs, users/apiKeys, secrets); inference is generative and stores nothing.
|
|
38
|
+
protocol: '2',
|
|
39
|
+
refresh: { every: '15m', onDemand: { atMost: '60s' } },
|
|
40
|
+
stateSystem: { perform: performFireworksAction, refresh: syncFireworksFromRemote },
|
|
41
|
+
// The round trip: a chat completion — the one call every Fireworks integration makes first, and
|
|
42
|
+
// the one whose answer proves the wire is live end to end — then a DEPLOYMENT, because inference
|
|
43
|
+
// is generative and stores nothing; the control plane is what has state. The deployment carries no
|
|
44
|
+
// `deploymentId`, so the vendor mints one (the spec makes it optional) and the create repeats
|
|
45
|
+
// cleanly on a branch that inherited the first.
|
|
46
|
+
roundTrip: [
|
|
47
|
+
{ method: 'POST', path: '/inference/v1/chat/completions', body: { model: 'accounts/fireworks/models/kimi-k2-instruct', messages: [{ role: 'user', content: 'round trip' }] }, headers: { authorization: 'Bearer round-trip' } },
|
|
48
|
+
{ method: 'POST', path: '/v1/accounts/round-trip/deployments', body: { baseModel: 'accounts/fireworks/models/kimi-k2-instruct', displayName: 'round trip' }, headers: { authorization: 'Bearer round-trip' } },
|
|
49
|
+
],
|
|
50
|
+
parityOrigin: 'http://twin',
|
|
51
|
+
archetype: 'generative',
|
|
52
|
+
bin: 'world-fireworks',
|
|
53
|
+
// The subject types the twin SERVES from its own state — the R2 resource-level claim. Inference
|
|
54
|
+
// (chat/completions, completions, responses, messages, embeddings, rerank) is generative and
|
|
55
|
+
// stores nothing; the control plane is what has state.
|
|
56
|
+
resources: ['deployment', 'dataset', 'batchInferenceJob', 'supervisedFineTuningJob', 'user', 'apiKey', 'secret', 'model'],
|
|
57
|
+
// R2 adopted-as-debt, legibly: declared types no replay can create, each with its reason.
|
|
58
|
+
resourcesUnreachable: {
|
|
59
|
+
account: 'the vendor’s own account row (one per credential, not creatable through the API)',
|
|
60
|
+
quota: 'vendor-assigned limits (read-only in the API; the twin derives them from the account)',
|
|
61
|
+
serverlessModel: 'the vendor catalog (serverless models are curated by Fireworks, not created by callers)',
|
|
62
|
+
},
|
|
63
|
+
specSource: 'Fireworks first-party OpenAPI spec (https://docs.fireworks.ai/merged.openapi.yaml, downloaded 2026-09-16): ' +
|
|
64
|
+
'the merged Gateway REST API 5.10.0 document covering the control plane (api.fireworks.ai/v1/accounts/{account_id}/…) ' +
|
|
65
|
+
'AND the inference plane namespaces (chat/completions, completions, Responses, Anthropic-compatible messages). ' +
|
|
66
|
+
'The embeddings and rerank operations are documented on the vendor api-reference pages only ' +
|
|
67
|
+
'(docs.fireworks.ai/api-reference — post /embeddings, post /rerank), not in the merged spec. ' +
|
|
68
|
+
'The earlier cited URLs (api.fireworks.ai and app.fireworks.ai api/docs/openapi.json, the per-plane ' +
|
|
69
|
+
'api-reference *-openapi.* files) are dead (404, checked 2026-09-16). Envelope-faithful; ' +
|
|
70
|
+
'model output, embedding values and rerank scores are labeled deterministic stubs.',
|
|
71
|
+
description: 'Fireworks AI twin — the OpenAI-compatible inference plane (chat/completions + streaming, completions, ' +
|
|
72
|
+
'Responses API CRUD, Anthropic-compatible /v1/messages, embeddings, rerank) with Fireworks’ documented ' +
|
|
73
|
+
'OpenAI differences (usage in the final stream chunk by default, context_length_exceeded_behavior, ' +
|
|
74
|
+
"service_tier 'priority'-only), plus the stateful Gateway control plane (deployments, datasets, " +
|
|
75
|
+
'batch-inference and supervised-fine-tuning jobs, users/apiKeys, secrets, models) with google.rpc-style ' +
|
|
76
|
+
'statuses; generative output is a labeled deterministic stub.',
|
|
77
|
+
// Fireworks serves BOTH planes on ONE host: the inference plane under /inference/v1/ and the
|
|
78
|
+
// control plane under /v1/accounts/{account_id}/. `apiPathPrefix` is a single startsWith prefix,
|
|
79
|
+
// so '/v' is the narrowest value that routes BOTH surfaces to the twin.
|
|
80
|
+
browserRouting: { apiPathPrefix: '/v', loaderHost: 'https://api.fireworks.ai' },
|
|
81
|
+
// ADOPTION — how an app repo betrays that it talks to Fireworks, so `volter-world covers`/
|
|
82
|
+
// `init` can attribute the signal here. Declared ON THE DESCRIPTOR, never in the central
|
|
83
|
+
// SDK_TWINS / SDK_SCOPE_VENDORS / ENV_STEM_VENDORS / VENDOR_WORLD_IDS maps: declaring a fact in
|
|
84
|
+
// both homes THROWS (world-runtime/src/pack-facts.ts, `overlay*`). `fireworks-ai` is Fireworks'
|
|
85
|
+
// official Python client; `@ai-sdk/fireworks` is the Vercel AI SDK provider. `FIREWORKS` covers
|
|
86
|
+
// FIREWORKS_API_KEY / FIREWORKS_BASE_URL, the names both clients read.
|
|
87
|
+
adoption: {
|
|
88
|
+
pypi: ['fireworks-ai'],
|
|
89
|
+
sdks: ['@ai-sdk/fireworks'],
|
|
90
|
+
envStems: ['FIREWORKS'],
|
|
91
|
+
},
|
|
92
|
+
// INTERCEPTION: the one host both planes are addressed on (the `fireworks` Python client and
|
|
93
|
+
// @ai-sdk/fireworks both default to api.fireworks.ai; the inference base URL is
|
|
94
|
+
// https://api.fireworks.ai/inference).
|
|
95
|
+
hosts: [{ host: 'api.fireworks.ai' }],
|
|
96
|
+
// No app-read endpoint env, deliberately. The `fireworks` Python client reads FIREWORKS_BASE_URL
|
|
97
|
+
// — but the app-read table is the FALLBACK for packs the injector cannot intercept, and this pack
|
|
98
|
+
// declares `hosts`, so FIREWORKS_TWIN_URL + zero-edit interception is the better answer. Emitting
|
|
99
|
+
// both would quietly forgo interception (`init.test.ts`, "the app-read endpoint table cannot go
|
|
100
|
+
// stale behind the injector").
|
|
101
|
+
endpointEnvNone: 'The injector intercepts api.fireworks.ai (see `hosts`), so a world wires this pack through FIREWORKS_TWIN_URL rather than an app-read base-URL var. The `fireworks` Python client would honour FIREWORKS_BASE_URL, but @ai-sdk/fireworks — the dominant client in this repo’s demand census — takes its baseURL as a constructor option and reads no such env, so an app-read var would cover only half the traffic while disabling the interception that covers all of it.',
|
|
102
|
+
};
|
|
103
|
+
// registered at import: the kernel learns the pack's state system (protocol 2)
|
|
104
|
+
import { registerPack } from '@volter/world-core';
|
|
105
|
+
registerPack(pack);
|
package/package.json
ADDED
|
@@ -0,0 +1,52 @@
|
|
|
1
|
+
{
|
|
2
|
+
"name": "@volter/twin-fireworks",
|
|
3
|
+
"version": "0.1.0",
|
|
4
|
+
"description": "Local Fireworks AI twin — OpenAI-compatible inference + Anthropic-compatible messages + the stateful Gateway control plane, envelope-faithful with labeled deterministic stubs. Built on @volter/twin.",
|
|
5
|
+
"author": "Volter (https://github.com/volter-ai)",
|
|
6
|
+
"license": "Apache-2.0",
|
|
7
|
+
"files": [
|
|
8
|
+
"src",
|
|
9
|
+
"README.md",
|
|
10
|
+
"LICENSE",
|
|
11
|
+
"!**/*.test.ts",
|
|
12
|
+
"!**/*.test.tsx",
|
|
13
|
+
"dist"
|
|
14
|
+
],
|
|
15
|
+
"repository": {
|
|
16
|
+
"type": "git",
|
|
17
|
+
"url": "git+https://github.com/volter-ai/twin.git",
|
|
18
|
+
"directory": "packages/twin/fireworks"
|
|
19
|
+
},
|
|
20
|
+
"homepage": "https://github.com/volter-ai/twin/tree/main/packages/twin/fireworks#readme",
|
|
21
|
+
"type": "module",
|
|
22
|
+
"exports": {
|
|
23
|
+
".": {
|
|
24
|
+
"types": "./dist/src/index.d.ts",
|
|
25
|
+
"default": "./dist/src/index.js"
|
|
26
|
+
}
|
|
27
|
+
},
|
|
28
|
+
"bin": {
|
|
29
|
+
"world-fireworks": "dist/src/cli.js"
|
|
30
|
+
},
|
|
31
|
+
"scripts": {
|
|
32
|
+
"test": "bun test src/*.test.ts",
|
|
33
|
+
"typecheck": "tsc --noEmit",
|
|
34
|
+
"build": "node ../../../scripts/publish/build.mjs",
|
|
35
|
+
"prepack": "node ../../../scripts/publish/prepare-publish.mjs prepack",
|
|
36
|
+
"postpack": "node ../../../scripts/publish/prepare-publish.mjs postpack"
|
|
37
|
+
},
|
|
38
|
+
"peerDependencies": {
|
|
39
|
+
"@volter/world-core": "2.0.0"
|
|
40
|
+
},
|
|
41
|
+
"devDependencies": {
|
|
42
|
+
"@ai-sdk/fireworks": "^3.0.52",
|
|
43
|
+
"@types/bun": "^1.2.20",
|
|
44
|
+
"@types/node": "^24.0.0",
|
|
45
|
+
"@volter/world-core": "2.0.0",
|
|
46
|
+
"@volter/world-tooling": "0.1.0",
|
|
47
|
+
"typescript": "^5.9.0"
|
|
48
|
+
},
|
|
49
|
+
"engines": {
|
|
50
|
+
"node": ">=22.3"
|
|
51
|
+
}
|
|
52
|
+
}
|
package/src/cli.ts
ADDED
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
import { keepProcessAlive } from '@volter/world-core/lifecycle';
|
|
3
|
+
// world-fireworks CLI: serve the Fireworks AI twin or run conformance. Fireworks is an
|
|
4
|
+
// API-first vendor — its console is a deployments/usage dashboard, not where the work happens
|
|
5
|
+
// — so this pack ships no mirror.
|
|
6
|
+
import { hasFlag, optionValue } from '@volter/world-core/args';
|
|
7
|
+
import { createFireworksTwinServer } from './fireworks-server.ts';
|
|
8
|
+
|
|
9
|
+
const [cmd, ...rest] = process.argv.slice(2);
|
|
10
|
+
const port = Number(optionValue(rest, '--port', '0')) || undefined;
|
|
11
|
+
const root = optionValue(rest, '--root') || undefined;
|
|
12
|
+
const readOnly = hasFlag(rest, '--read-only'); // a twin accepts writes unless started read-only
|
|
13
|
+
const scenario = optionValue(rest, '--scenario') || undefined; // scripted responses (JSON file)
|
|
14
|
+
|
|
15
|
+
if (cmd === 'serve') {
|
|
16
|
+
const s = await createFireworksTwinServer({ readOnly, ...(root ? { root } : {}), ...(port ? { port } : {}), ...(scenario ? { scenarioPath: scenario } : {}) });
|
|
17
|
+
process.stdout.write(`fireworks twin (Gateway control plane at /v1/accounts/{account_id}; inference at /inference/v1)${readOnly ? ' [read-only]' : ''}${scenario ? ` [scenario: ${scenario}]` : ''} at http://127.0.0.1:${s.port}\n`);
|
|
18
|
+
await keepProcessAlive();
|
|
19
|
+
} else if (cmd === 'conformance') {
|
|
20
|
+
// dev-only; lazy so the bin runs without @volter/world-tooling
|
|
21
|
+
const { checkFireworksConformance } = await import('./fireworks-conformance.ts');
|
|
22
|
+
const report = await checkFireworksConformance({ ...(root ? { root } : {}) });
|
|
23
|
+
process.stdout.write(`${JSON.stringify(report, null, 2)}\n`);
|
|
24
|
+
if (!report.ok) process.exitCode = 1;
|
|
25
|
+
} else {
|
|
26
|
+
process.stdout.write('Usage: world-fireworks serve|conformance [--port N] [--root DIR] [--read-only] [--scenario FILE]\n');
|
|
27
|
+
}
|
|
@@ -0,0 +1,172 @@
|
|
|
1
|
+
// Fireworks' CLIENT-SIDE RATE BUDGET — the pack's DECLARATION (the numbers) plus the thin typed
|
|
2
|
+
// bindings `liveFireworksExecute` uses. The MECHANISM — the durable token-keyed ledger, the rolling
|
|
3
|
+
// window, reserve-under-lock, the `Retry-After`/429 cooldown, fail-CLOSED on a corrupt ledger —
|
|
4
|
+
// lives ONCE in the vendor-agnostic kernel (`@volter/world-core` → `rateBudget.ts`). Read that
|
|
5
|
+
// module's header for the full rationale AND for the honest list of what the guard does not
|
|
6
|
+
// guarantee.
|
|
7
|
+
//
|
|
8
|
+
// ── WHY THIS EXISTS ─────────────────────────────────────────────────────────────────────────
|
|
9
|
+
// A real ~4.5-DAY vendor lockout (Figma, 2026-07-25) happened because raw API calls were made
|
|
10
|
+
// outside the pack's connector — no cache, no batching, no ceiling. Discipline only binds the code
|
|
11
|
+
// that follows it; a BUDGET binds the code that does not.
|
|
12
|
+
//
|
|
13
|
+
// ── HOW THE CEILING WAS CHOSEN (live-read, 2026-09-16) ───────────────────────────────────────
|
|
14
|
+
// https://docs.fireworks.ai/serverless/rate-limits (read 2026-09-16) publishes NO scalar per-key
|
|
15
|
+
// request limit a client could be bound by. Fireworks meters THREE token-per-minute metrics —
|
|
16
|
+
// Total Prompt TPM, Uncached Prompt TPM and Generated TPM — each "per account and per model", and
|
|
17
|
+
// the ceilings are ADAPTIVE: an account's allowance scales within its size tier based on recent
|
|
18
|
+
// usage, so the only place the real numbers live is the account's own dashboard. The docs also
|
|
19
|
+
// state rate-limit headers are not currently returned, so there is no response-derived backstop
|
|
20
|
+
// signal to bind.
|
|
21
|
+
//
|
|
22
|
+
// So this declaration does not model Fireworks' limit, and nothing here is more permissive than
|
|
23
|
+
// the kernel's undeclared fallback: window 60s, ceiling 60, defaultWeight 2 — i.e. 30 calls/minute,
|
|
24
|
+
// EXACTLY `DEFAULT_RATE_BUDGET`, with no endpoint priced cheaper than the fallback would price it.
|
|
25
|
+
// What the declaration buys is RESOLUTION, and it deliberately buys it DOWNWARD: the inference
|
|
26
|
+
// endpoints (the token-metered ones — every TPM metric above is about inference traffic) cost 6,
|
|
27
|
+
// so at most TEN of them land in a window. Being stricter than the fallback needs no vendor
|
|
28
|
+
// justification; being looser would, and there is none to have.
|
|
29
|
+
//
|
|
30
|
+
// It bounds the 60s AVERAGE; it does not pace (the kernel refuses, it never sleeps). The backstop
|
|
31
|
+
// for a sub-second burst is the `Retry-After`/429 cooldown, and it is the ONLY backstop. Fireworks
|
|
32
|
+
// DOES document rate-limit response headers — `X-Ratelimit-Limit-Tokens-Prompt`,
|
|
33
|
+
// `X-Ratelimit-Limit-Tokens-Cache-Adjusted-Prompt`, `X-Ratelimit-Limit-Tokens-Generated` — but they
|
|
34
|
+
// carry TOKEN CEILING values, not a remaining-count or a retry-after, so none of them is a signal
|
|
35
|
+
// the kernel's cooldown can arm on. Unlike the groq declaration there is no header-derived
|
|
36
|
+
// cooldown trigger: only a real 429 (or an explicit `Retry-After`) can arm it.
|
|
37
|
+
//
|
|
38
|
+
// ── HOW THE WEIGHTS WERE CHOSEN (and what is a judgement call) ───────────────────────────────
|
|
39
|
+
// • The inference endpoints cost 6: all THREE published TPM metrics are about inference traffic,
|
|
40
|
+
// so one chat/completions/embeddings/rerank request consumes far more of the account's real
|
|
41
|
+
// allowance than a deployments list poll. 60/6 = 10 per window is a deliberate tightening,
|
|
42
|
+
// not a matched published figure — Fireworks publishes no RPM at all.
|
|
43
|
+
// • Everything else (control-plane CRUD on the Gateway REST surface) costs 2 — the fallback's
|
|
44
|
+
// own default.
|
|
45
|
+
import {
|
|
46
|
+
declareRateBudget,
|
|
47
|
+
rateBudgetPath,
|
|
48
|
+
rateBudgetWeight,
|
|
49
|
+
RateBudget,
|
|
50
|
+
type RateBudgetDeclaration,
|
|
51
|
+
type RateBudgetOptions,
|
|
52
|
+
type RateBudgetReservation,
|
|
53
|
+
type RateBudgetSnapshot,
|
|
54
|
+
} from '@volter/world-core';
|
|
55
|
+
|
|
56
|
+
const VENDOR = 'fireworks';
|
|
57
|
+
|
|
58
|
+
/** Rolling window, in ms. Spend older than this is pruned. */
|
|
59
|
+
export const FIREWORKS_BUDGET_WINDOW_MS = 60_000;
|
|
60
|
+
|
|
61
|
+
/**
|
|
62
|
+
* Weighted units allowed inside one window. 60/60s at `defaultWeight` 2 = 30 calls a minute —
|
|
63
|
+
* EXACTLY the kernel's undeclared fallback, because Fireworks publishes no scalar that would
|
|
64
|
+
* justify more. See the header.
|
|
65
|
+
*/
|
|
66
|
+
export const FIREWORKS_BUDGET_CEILING = 60;
|
|
67
|
+
|
|
68
|
+
/** Seconds. A `retry-after` above this means the key is throttled hard — fail loudly, don't sleep. */
|
|
69
|
+
export const FIREWORKS_BUDGET_MAX_RETRY_AFTER_S = 300;
|
|
70
|
+
|
|
71
|
+
/** Per-call cost, keyed by `"<METHOD> <path>"`. See the header for what is documented vs. judged. */
|
|
72
|
+
export const FIREWORKS_CALL_WEIGHTS = {
|
|
73
|
+
/** The token-metered inference surface: `/inference/v1/chat/completions`,
|
|
74
|
+
* `/inference/v1/completions`, `/inference/v1/responses`, `/inference/v1/embeddings`,
|
|
75
|
+
* `/inference/v1/rerank` and `/inference/v1/messages`. Every published TPM metric (Total
|
|
76
|
+
* Prompt, Uncached Prompt, Generated) is about exactly this traffic. At 60/6 = 10 per window —
|
|
77
|
+
* a deliberate tightening below the fallback's 30, since Fireworks publishes no request-rate
|
|
78
|
+
* figure at all. */
|
|
79
|
+
inference: 6,
|
|
80
|
+
/** Everything else: Gateway control-plane CRUD (deployments, datasets, fine-tuning jobs, …). */
|
|
81
|
+
other: 2,
|
|
82
|
+
} as const;
|
|
83
|
+
|
|
84
|
+
/** THE PACK'S DECLARATION — pure data, the only Fireworks-specific thing in the whole budget. */
|
|
85
|
+
export const FIREWORKS_RATE_BUDGET: RateBudgetDeclaration = {
|
|
86
|
+
windowMs: FIREWORKS_BUDGET_WINDOW_MS,
|
|
87
|
+
ceiling: FIREWORKS_BUDGET_CEILING,
|
|
88
|
+
defaultWeight: FIREWORKS_CALL_WEIGHTS.other,
|
|
89
|
+
maxRetryAfterSeconds: FIREWORKS_BUDGET_MAX_RETRY_AFTER_S,
|
|
90
|
+
rules: [
|
|
91
|
+
{ match: '^POST /inference/v1/(chat/completions|completions|responses|embeddings|rerank|messages)$', weight: FIREWORKS_CALL_WEIGHTS.inference },
|
|
92
|
+
],
|
|
93
|
+
reason:
|
|
94
|
+
'Fireworks publishes NO scalar per-key request limit (docs.fireworks.ai/serverless/rate-limits, ' +
|
|
95
|
+
'read 2026-09-16): it meters THREE token-per-minute metrics — Total Prompt TPM, Uncached Prompt ' +
|
|
96
|
+
'TPM and Generated TPM — each per account AND per model, with ADAPTIVE ceilings that scale within ' +
|
|
97
|
+
"a size tier based on recent usage, so the real numbers live only on the account's dashboard. " +
|
|
98
|
+
'Its documented rate-limit response headers (X-Ratelimit-Limit-Tokens-Prompt / ' +
|
|
99
|
+
'-Cache-Adjusted-Prompt / -Generated) carry TOKEN CEILING values, not a remaining count or a ' +
|
|
100
|
+
'retry-after, so no header is a signal the cooldown can arm on. Because no published figure ' +
|
|
101
|
+
'justifies going higher, the ceiling is pinned at the kernel fallback in EVERY dimension — ' +
|
|
102
|
+
'60 units / 60s at defaultWeight 2 = 30 calls/min — and the declaration buys resolution ' +
|
|
103
|
+
'DOWNWARD, never headroom: the inference endpoints cost 6, so at most 10 land in a window. ' +
|
|
104
|
+
'The window bounds the 60s AVERAGE and does not pace; the `retry-after`/429 cooldown is the ' +
|
|
105
|
+
'only sub-second-burst backstop.',
|
|
106
|
+
};
|
|
107
|
+
|
|
108
|
+
// Declared at module load, so merely importing this module (which `fireworks-connector.ts` does) is
|
|
109
|
+
// enough to arm the real ceiling.
|
|
110
|
+
declareRateBudget(VENDOR, FIREWORKS_RATE_BUDGET);
|
|
111
|
+
|
|
112
|
+
/**
|
|
113
|
+
* Price one call. The key is `"<METHOD> <path>"` with the query string split off, so a rule can
|
|
114
|
+
* price by method (a write is not a read) without the kernel knowing anything about Fireworks. An
|
|
115
|
+
* unclassified endpoint still costs `defaultWeight` — nothing is ever free.
|
|
116
|
+
*/
|
|
117
|
+
export function fireworksCallWeight(method: string, path: string): number {
|
|
118
|
+
const { bare, query } = splitQuery(path);
|
|
119
|
+
// UPPER-CASE the method: `fetch` normalizes a known lowercase method before sending, so
|
|
120
|
+
// `execute('post', …)` really does issue a POST and must be priced as one.
|
|
121
|
+
return rateBudgetWeight(VENDOR, `${String(method).toUpperCase()} ${bare}`, query);
|
|
122
|
+
}
|
|
123
|
+
|
|
124
|
+
/**
|
|
125
|
+
* `/v1/x?a=1` -> `{ bare: '/v1/x', query: { a: '1' } }`. Rules match the path; `whenQuery*` the
|
|
126
|
+
* query. NORMALIZED, because the anchored rules are otherwise trivially evaded: `fetch` upper-cases
|
|
127
|
+
* a known method before sending, so `execute('post', …)` issues a real WRITE that a `^POST ` rule
|
|
128
|
+
* would price as a read; and a trailing slash makes a path miss a `$` anchor while most routers
|
|
129
|
+
* treat it as the same endpoint.
|
|
130
|
+
*/
|
|
131
|
+
function splitQuery(path: string): { bare: string; query: Record<string, string> } {
|
|
132
|
+
const at = path.indexOf('?');
|
|
133
|
+
const query: Record<string, string> = {};
|
|
134
|
+
if (at !== -1) for (const [k, v] of new URLSearchParams(path.slice(at + 1))) query[k] = v;
|
|
135
|
+
// Collapse REPEATED slashes as well as a trailing one: `/inference/v1//chat/completions` reaches
|
|
136
|
+
// the same endpoint on most routers but misses a `^POST /inference/v1/(chat/completions|…)$` rule,
|
|
137
|
+
// which would price an inference call as a 2-unit read.
|
|
138
|
+
const raw = (at === -1 ? path : path.slice(0, at)).replace(/\/{2,}/g, '/');
|
|
139
|
+
const bare = raw.length > 1 && raw.endsWith('/') ? raw.replace(/\/+$/, '') : raw;
|
|
140
|
+
return { bare, query };
|
|
141
|
+
}
|
|
142
|
+
|
|
143
|
+
/** Where Fireworks' ledger lives. Token-keyed and cwd-independent by default (Fireworks' limits
|
|
144
|
+
* are per ACCOUNT, i.e. per key, so a cwd-scoped ledger would hand the same key a fresh allowance
|
|
145
|
+
* in every checkout, worktree and CI matrix leg); pass `root` for world-scoped accounting. */
|
|
146
|
+
export function fireworksBudgetPath(opts: { root?: string; token?: string } | string = {}): string {
|
|
147
|
+
const o = typeof opts === 'string' ? { root: opts } : opts;
|
|
148
|
+
// VENDOR spread LAST: a loosely-typed `{ vendor: 'other', … }` slipping through (TypeScript's
|
|
149
|
+
// excess-property check only catches object literals) must not redirect this pack's ledger.
|
|
150
|
+
return rateBudgetPath({ ...o, vendor: VENDOR });
|
|
151
|
+
}
|
|
152
|
+
|
|
153
|
+
/** Construction options for Fireworks' budget. The vendor is fixed; everything else may only TIGHTEN. */
|
|
154
|
+
export type FireworksBudgetOptions = Omit<RateBudgetOptions, 'vendor'>;
|
|
155
|
+
|
|
156
|
+
/**
|
|
157
|
+
* Fireworks' budget — the shared kernel guard bound to this vendor's declaration. A real subclass,
|
|
158
|
+
* not an alias, so `budget instanceof FireworksBudget` in `liveFireworksExecute` means "a budget
|
|
159
|
+
* that accounts against FIREWORKS' ledger under FIREWORKS' ceiling".
|
|
160
|
+
*/
|
|
161
|
+
export class FireworksBudget extends RateBudget {
|
|
162
|
+
constructor(opts: FireworksBudgetOptions = {}) {
|
|
163
|
+
super({ ...opts, vendor: VENDOR });
|
|
164
|
+
}
|
|
165
|
+
}
|
|
166
|
+
|
|
167
|
+
/** The typed refusal. One error class shared with every other vendor's budget; `err.vendor` says
|
|
168
|
+
* which one refused, and `err.kind` says why. */
|
|
169
|
+
export { RateBudgetError as FireworksBudgetError } from '@volter/world-core';
|
|
170
|
+
export type { RateBudgetErrorKind as FireworksBudgetErrorKind } from '@volter/world-core';
|
|
171
|
+
export type FireworksBudgetReservation = RateBudgetReservation;
|
|
172
|
+
export type FireworksBudgetSnapshot = RateBudgetSnapshot;
|