@volter/twin-fireworks 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (39) hide show
  1. package/LICENSE +202 -0
  2. package/README.md +184 -0
  3. package/dist/src/cli.d.ts +2 -0
  4. package/dist/src/cli.js +28 -0
  5. package/dist/src/fireworks-budget.d.ts +54 -0
  6. package/dist/src/fireworks-budget.js +146 -0
  7. package/dist/src/fireworks-capabilities.d.ts +4 -0
  8. package/dist/src/fireworks-capabilities.js +1205 -0
  9. package/dist/src/fireworks-conformance.d.ts +14 -0
  10. package/dist/src/fireworks-conformance.js +514 -0
  11. package/dist/src/fireworks-connector.d.ts +168 -0
  12. package/dist/src/fireworks-connector.js +641 -0
  13. package/dist/src/fireworks-models.d.ts +11 -0
  14. package/dist/src/fireworks-models.js +53 -0
  15. package/dist/src/fireworks-scenario.d.ts +55 -0
  16. package/dist/src/fireworks-scenario.js +171 -0
  17. package/dist/src/fireworks-server.d.ts +16 -0
  18. package/dist/src/fireworks-server.js +144 -0
  19. package/dist/src/fireworks-stub.d.ts +26 -0
  20. package/dist/src/fireworks-stub.js +78 -0
  21. package/dist/src/fireworks-twin.d.ts +51 -0
  22. package/dist/src/fireworks-twin.js +1426 -0
  23. package/dist/src/fireworks-types.d.ts +212 -0
  24. package/dist/src/fireworks-types.js +4 -0
  25. package/dist/src/index.d.ts +9 -0
  26. package/dist/src/index.js +105 -0
  27. package/package.json +52 -0
  28. package/src/cli.ts +27 -0
  29. package/src/fireworks-budget.ts +172 -0
  30. package/src/fireworks-capabilities.ts +1229 -0
  31. package/src/fireworks-conformance.ts +542 -0
  32. package/src/fireworks-connector.ts +700 -0
  33. package/src/fireworks-models.ts +63 -0
  34. package/src/fireworks-scenario.ts +191 -0
  35. package/src/fireworks-server.ts +153 -0
  36. package/src/fireworks-stub.ts +83 -0
  37. package/src/fireworks-twin.ts +1427 -0
  38. package/src/fireworks-types.ts +165 -0
  39. package/src/index.ts +134 -0
@@ -0,0 +1,165 @@
1
+ // Fireworks wire shapes, typed narrowly — the subsets of the vendor's real response shapes this
2
+ // pack reads and writes, grounded in the first-party OpenAPI specs (see specSource in index.ts).
3
+ // Where Fireworks differs from OpenAI the difference is NAMED here, not papered over.
4
+
5
+ // ── Inference plane (OpenAI-compatible, at /inference/v1) ────────────────────────────────────
6
+
7
+ /** One message in a chat request. `reasoning_content` on assistant messages is a Fireworks
8
+ * extension (reasoning models return it; `reasoning_history` controls whether it is replayed). */
9
+ export type FireworksMessageParam = {
10
+ role: string;
11
+ content?: unknown;
12
+ name?: string;
13
+ tool_call_id?: string;
14
+ tool_calls?: Array<{ id: string; type: 'function'; function: { name: string; arguments: string } }>;
15
+ reasoning_content?: string | null;
16
+ };
17
+
18
+ /** The `service_tier` enum is Fireworks' own; only 'priority' is honored — every other value is
19
+ * treated as 'default' (documented; NOT an error). */
20
+ export type FireworksServiceTier = 'auto' | 'default' | 'flex' | 'priority';
21
+ /** Fireworks' documented OpenAI difference: the default is `truncate`, not `error`. */
22
+ export type FireworksContextLengthExceededBehavior = 'error' | 'truncate';
23
+
24
+ export type FireworksChatRequest = {
25
+ model: string;
26
+ messages: FireworksMessageParam[];
27
+ temperature?: number | null;
28
+ top_p?: number | null;
29
+ top_k?: number | null;
30
+ frequency_penalty?: number | null;
31
+ presence_penalty?: number | null;
32
+ max_tokens?: number | null;
33
+ max_completion_tokens?: number | null;
34
+ n?: number | null;
35
+ stop?: string | string[] | null;
36
+ stream?: boolean | null;
37
+ stream_options?: { include_usage?: boolean; buffer_tokens?: number; buffer_ms?: number; buffer_mode?: string } | null;
38
+ tools?: Array<Record<string, unknown>>;
39
+ tool_choice?: unknown;
40
+ parallel_tool_calls?: boolean | null;
41
+ response_format?: Record<string, unknown> | null;
42
+ reasoning_effort?: string | number | boolean | null;
43
+ reasoning_history?: 'disabled' | 'interleaved' | 'preserved' | null;
44
+ service_tier?: FireworksServiceTier | null;
45
+ context_length_exceeded_behavior?: FireworksContextLengthExceededBehavior | null;
46
+ seed?: number | null;
47
+ user?: string | null;
48
+ echo?: boolean | null;
49
+ logprobs?: boolean | null;
50
+ top_logprobs?: number | null;
51
+ logit_bias?: Record<string, number> | null;
52
+ metadata?: Record<string, unknown> | null;
53
+ };
54
+
55
+ export type FireworksChatMessage = {
56
+ role: 'assistant';
57
+ content: string | null;
58
+ reasoning_content?: string | null;
59
+ tool_calls?: Array<{ id: string; type: 'function'; function: { name: string; arguments: string } }>;
60
+ };
61
+
62
+ /** The response envelope: `usage` is nullable on chat (absent when `echo`/logprobs paths skip it),
63
+ * unlike `Completion` below where the spec marks it required. */
64
+ export type FireworksChatCompletion = {
65
+ id: string;
66
+ object: 'chat.completion';
67
+ created: number;
68
+ model: string;
69
+ choices: Array<{
70
+ index: number;
71
+ message: FireworksChatMessage;
72
+ finish_reason: string | null;
73
+ logprobs?: unknown | null;
74
+ token_ids?: number[] | null;
75
+ }>;
76
+ usage?: {
77
+ prompt_tokens: number;
78
+ total_tokens: number;
79
+ completion_tokens: number;
80
+ } | null;
81
+ };
82
+
83
+ /** Legacy `/v1/completions`: the spec marks `usage` REQUIRED here (chat's is nullable). */
84
+ export type FireworksCompletion = {
85
+ id: string;
86
+ object: 'text_completion';
87
+ created: number;
88
+ model: string;
89
+ choices: Array<{ index: number; text: string; finish_reason: string | null; logprobs?: unknown | null }>;
90
+ usage: { prompt_tokens: number; total_tokens: number; completion_tokens: number };
91
+ };
92
+
93
+ /** The Responses API (`/v1/responses`): `id` is nullable when `store=false` (the vendor's own
94
+ * contract — a non-stored response has no id to retrieve later). */
95
+ export type FireworksResponseObject = {
96
+ id: string | null;
97
+ object: 'response';
98
+ created_at: number;
99
+ status: 'completed' | 'in_progress' | 'incomplete' | 'failed' | 'cancelled';
100
+ model: string;
101
+ output: Array<FireworksResponseOutputItem>;
102
+ usage?: { prompt_tokens: number; completion_tokens: number; total_tokens: number } | null;
103
+ error?: unknown | null;
104
+ incomplete_details?: unknown | null;
105
+ instructions?: unknown | null;
106
+ max_output_tokens?: number | null;
107
+ metadata?: Record<string, unknown> | null;
108
+ parallel_tool_calls?: boolean | null;
109
+ previous_response_id?: string | null;
110
+ reasoning?: unknown | null;
111
+ store?: boolean | null;
112
+ temperature?: number | null;
113
+ text?: unknown | null;
114
+ tool_choice?: unknown;
115
+ tools?: Array<Record<string, unknown>> | null;
116
+ top_p?: number | null;
117
+ truncation?: string | null;
118
+ user?: string | null;
119
+ };
120
+
121
+ export type FireworksResponseOutputItem =
122
+ | { type: 'message'; id: string; role: 'assistant'; status: 'in_progress' | 'completed'; content: Array<{ type: string; text?: string }> }
123
+ | { type: 'function_call'; id: string; call_id: string; name: string; arguments: string; status?: string }
124
+ | { type: 'function_call_output'; tool_call_id: string; output: unknown };
125
+
126
+ /** Anthropic-compatible `/v1/messages`: the response `content` is an array of typed blocks, and
127
+ * `stop_reason` is Anthropic's enum (never OpenAI's `finish_reason`). */
128
+ export type FireworksAnthropicMessage = {
129
+ id: string;
130
+ type: 'message';
131
+ role: 'assistant';
132
+ content: Array<FireworksAnthropicContentBlock>;
133
+ model: string;
134
+ stop_reason: 'end_turn' | 'max_tokens' | 'stop_sequence' | 'tool_use' | 'pause_turn' | 'refusal' | null;
135
+ stop_sequence: string | null;
136
+ usage?: { input_tokens: number; output_tokens: number };
137
+ };
138
+
139
+ export type FireworksAnthropicContentBlock =
140
+ | { type: 'text'; text: string; citations?: unknown[] | null }
141
+ | { type: 'thinking'; thinking: string; signature: string }
142
+ | { type: 'redacted_thinking'; data: string }
143
+ | { type: 'tool_use'; id: string; name: string; input: unknown };
144
+
145
+ /** The Anthropic error envelope: `{ type: 'error', error: { type, message }, request_id? }` — a
146
+ * DIFFERENT envelope from the OpenAI-compat plane's `{ error: { … } }`. */
147
+ export type FireworksAnthropicErrorType =
148
+ | 'invalid_request_error' | 'authentication_error' | 'billing_error' | 'permission_error'
149
+ | 'not_found_error' | 'rate_limit_error' | 'timeout_error' | 'api_error' | 'overloaded_error';
150
+
151
+ // ── Control plane (Gateway REST API, at /v1/accounts/{account_id}) ───────────────────────────
152
+
153
+ /** google.rpc-style status, embedded on every resource as `status` and (per the vendor's own
154
+ * comment) mimicking google/rpc/status.proto. */
155
+ export type FireworksGatewayStatus = { code: FireworksGatewayCode; message?: string } | null;
156
+
157
+ /** The gatewayCode enum — the gRPC canonical codes, as strings. */
158
+ export type FireworksGatewayCode =
159
+ | 'OK' | 'CANCELLED' | 'UNKNOWN' | 'INVALID_ARGUMENT' | 'DEADLINE_EXCEEDED' | 'NOT_FOUND'
160
+ | 'ALREADY_EXISTS' | 'PERMISSION_DENIED' | 'UNAUTHENTICATED' | 'RESOURCE_EXHAUSTED'
161
+ | 'FAILED_PRECONDITION' | 'ABORTED' | 'OUT_OF_RANGE' | 'UNIMPLEMENTED' | 'INTERNAL'
162
+ | 'UNAVAILABLE' | 'DATA_LOSS';
163
+
164
+ /** The list envelope every Gateway list operation answers. */
165
+ export type FireworksListEnvelope<T> = { items?: never } & { [K in string]: T[] } & { nextPageToken?: string | null; totalSize?: number | null };
package/src/index.ts ADDED
@@ -0,0 +1,134 @@
1
+ // @volter/twin-fireworks — the Fireworks twin (one vendor, one package), built on the shared
2
+ // @volter/world-core kernel. Fireworks serves TWO API planes on ONE host (api.fireworks.ai):
3
+ //
4
+ // • the INFERENCE plane — `https://api.fireworks.ai/inference/v1/…`, OpenAI-compatible
5
+ // (chat/completions, completions, responses, embeddings, rerank) plus an Anthropic-compatible
6
+ // POST /v1/messages. Generative output is a labeled deterministic stub; the wire protocol and
7
+ // Fireworks' documented OpenAI differences are the fidelity surface.
8
+ // • the CONTROL plane — the Gateway REST API (`https://api.fireworks.ai/v1/accounts/{account_id}/…`):
9
+ // deployments, datasets, fine-tuning jobs, users/apiKeys, secrets, models, quotas — a stateful
10
+ // google.rpc-style surface (list envelopes {items, nextPageToken, totalSize}, verb-suffixed
11
+ // custom methods like `:cancel`, ids passed via query params on writes).
12
+ //
13
+ // THE DETERMINISTIC STUB IS THE ANSWER: the twin runs no model, so inference returns a
14
+ // DETERMINISTIC STUB completion (clearly labeled `[twin-stub:<model>]`), never pretending to be
15
+ // real inference. Everything around it — the wire protocol, the rejections, the state — is faithful.
16
+ // (Conformance + capability tooling live in @volter/world-tooling, a dev dependency — NOT
17
+ // re-exported here, per E2.)
18
+ export { handleFireworksTwinRequest, FIREWORKS_INFERENCE_PREFIX, FIREWORKS_ACCOUNTS_PREFIX } from './fireworks-twin.ts';
19
+ export type { FireworksRequest, FireworksResponseEnvelope } from './fireworks-twin.ts';
20
+ export { createFireworksTwinFetch, createFireworksTwinServer, type FireworksTwinFetchOptions } from './fireworks-server.ts';
21
+ export {
22
+ fullSyncFireworks,
23
+ fireworksRequestForAction,
24
+ liveFireworksExecute,
25
+ pullFireworksState,
26
+ pushFireworksAction,
27
+ pushPendingFireworksActions,
28
+ syncFireworksFromReal,
29
+ unpushableReason,
30
+ } from './fireworks-connector.ts';
31
+ export type { FireworksExecute, LiveFireworksOptions } from './fireworks-connector.ts';
32
+ // The client-side rate budget — the fail-closed backstop `liveFireworksExecute` routes every live
33
+ // request through. The MECHANISM is the kernel's shared, vendor-agnostic `RateBudget`; what lives
34
+ // here is Fireworks' DECLARATION (window/ceiling/per-endpoint weights) plus the vendor-bound
35
+ // bindings. Exported so an operator can inspect spend (`snapshot`) and so a caller can catch
36
+ // `FireworksBudgetError` by type; there is deliberately no export that disables the guard.
37
+ export {
38
+ FIREWORKS_BUDGET_CEILING,
39
+ FIREWORKS_BUDGET_MAX_RETRY_AFTER_S,
40
+ FIREWORKS_BUDGET_WINDOW_MS,
41
+ FIREWORKS_CALL_WEIGHTS,
42
+ FIREWORKS_RATE_BUDGET,
43
+ FireworksBudget,
44
+ FireworksBudgetError,
45
+ fireworksBudgetPath,
46
+ fireworksCallWeight,
47
+ } from './fireworks-budget.ts';
48
+ export type { FireworksBudgetErrorKind, FireworksBudgetOptions, FireworksBudgetReservation, FireworksBudgetSnapshot } from './fireworks-budget.ts';
49
+
50
+ import type { TwinPack } from '@volter/world-core';
51
+ import { FIREWORKS_RATE_BUDGET as RATE_BUDGET } from './fireworks-budget.ts';
52
+ import { performFireworksAction, syncFireworksFromRemote } from './fireworks-connector.ts';
53
+
54
+ export const pack: TwinPack = {
55
+ vendor: 'fireworks',
56
+ // The SAME object fireworks-budget.ts declares at module load — one source of truth, so
57
+ // registering the pack and importing the connector can never arm two different ceilings.
58
+ rateBudget: RATE_BUDGET,
59
+ transport: 'rest',
60
+ // PROTOCOL 2 (pack contract Part 3): the pack is a plugin — its wire, its tree, and its half of
61
+ // the real state system. The control plane is the stateful surface (deployments, datasets,
62
+ // fine-tuning jobs, users/apiKeys, secrets); inference is generative and stores nothing.
63
+ protocol: '2',
64
+ refresh: { every: '15m', onDemand: { atMost: '60s' } },
65
+ stateSystem: { perform: performFireworksAction, refresh: syncFireworksFromRemote },
66
+ // The round trip: a chat completion — the one call every Fireworks integration makes first, and
67
+ // the one whose answer proves the wire is live end to end — then a DEPLOYMENT, because inference
68
+ // is generative and stores nothing; the control plane is what has state. The deployment carries no
69
+ // `deploymentId`, so the vendor mints one (the spec makes it optional) and the create repeats
70
+ // cleanly on a branch that inherited the first.
71
+ roundTrip: [
72
+ { method: 'POST', path: '/inference/v1/chat/completions', body: { model: 'accounts/fireworks/models/kimi-k2-instruct', messages: [{ role: 'user', content: 'round trip' }] }, headers: { authorization: 'Bearer round-trip' } },
73
+ { method: 'POST', path: '/v1/accounts/round-trip/deployments', body: { baseModel: 'accounts/fireworks/models/kimi-k2-instruct', displayName: 'round trip' }, headers: { authorization: 'Bearer round-trip' } },
74
+ ],
75
+ parityOrigin: 'http://twin',
76
+ archetype: 'generative',
77
+ bin: 'world-fireworks',
78
+ // The subject types the twin SERVES from its own state — the R2 resource-level claim. Inference
79
+ // (chat/completions, completions, responses, messages, embeddings, rerank) is generative and
80
+ // stores nothing; the control plane is what has state.
81
+ resources: ['deployment', 'dataset', 'batchInferenceJob', 'supervisedFineTuningJob', 'user', 'apiKey', 'secret', 'model'],
82
+ // R2 adopted-as-debt, legibly: declared types no replay can create, each with its reason.
83
+ resourcesUnreachable: {
84
+ account: 'the vendor’s own account row (one per credential, not creatable through the API)',
85
+ quota: 'vendor-assigned limits (read-only in the API; the twin derives them from the account)',
86
+ serverlessModel: 'the vendor catalog (serverless models are curated by Fireworks, not created by callers)',
87
+ },
88
+ specSource:
89
+ 'Fireworks first-party OpenAPI spec (https://docs.fireworks.ai/merged.openapi.yaml, downloaded 2026-09-16): ' +
90
+ 'the merged Gateway REST API 5.10.0 document covering the control plane (api.fireworks.ai/v1/accounts/{account_id}/…) ' +
91
+ 'AND the inference plane namespaces (chat/completions, completions, Responses, Anthropic-compatible messages). ' +
92
+ 'The embeddings and rerank operations are documented on the vendor api-reference pages only ' +
93
+ '(docs.fireworks.ai/api-reference — post /embeddings, post /rerank), not in the merged spec. ' +
94
+ 'The earlier cited URLs (api.fireworks.ai and app.fireworks.ai api/docs/openapi.json, the per-plane ' +
95
+ 'api-reference *-openapi.* files) are dead (404, checked 2026-09-16). Envelope-faithful; ' +
96
+ 'model output, embedding values and rerank scores are labeled deterministic stubs.',
97
+ description:
98
+ 'Fireworks AI twin — the OpenAI-compatible inference plane (chat/completions + streaming, completions, ' +
99
+ 'Responses API CRUD, Anthropic-compatible /v1/messages, embeddings, rerank) with Fireworks’ documented ' +
100
+ 'OpenAI differences (usage in the final stream chunk by default, context_length_exceeded_behavior, ' +
101
+ "service_tier 'priority'-only), plus the stateful Gateway control plane (deployments, datasets, " +
102
+ 'batch-inference and supervised-fine-tuning jobs, users/apiKeys, secrets, models) with google.rpc-style ' +
103
+ 'statuses; generative output is a labeled deterministic stub.',
104
+ // Fireworks serves BOTH planes on ONE host: the inference plane under /inference/v1/ and the
105
+ // control plane under /v1/accounts/{account_id}/. `apiPathPrefix` is a single startsWith prefix,
106
+ // so '/v' is the narrowest value that routes BOTH surfaces to the twin.
107
+ browserRouting: { apiPathPrefix: '/v', loaderHost: 'https://api.fireworks.ai' },
108
+ // ADOPTION — how an app repo betrays that it talks to Fireworks, so `volter-world covers`/
109
+ // `init` can attribute the signal here. Declared ON THE DESCRIPTOR, never in the central
110
+ // SDK_TWINS / SDK_SCOPE_VENDORS / ENV_STEM_VENDORS / VENDOR_WORLD_IDS maps: declaring a fact in
111
+ // both homes THROWS (world-runtime/src/pack-facts.ts, `overlay*`). `fireworks-ai` is Fireworks'
112
+ // official Python client; `@ai-sdk/fireworks` is the Vercel AI SDK provider. `FIREWORKS` covers
113
+ // FIREWORKS_API_KEY / FIREWORKS_BASE_URL, the names both clients read.
114
+ adoption: {
115
+ pypi: ['fireworks-ai'],
116
+ sdks: ['@ai-sdk/fireworks'],
117
+ envStems: ['FIREWORKS'],
118
+ },
119
+ // INTERCEPTION: the one host both planes are addressed on (the `fireworks` Python client and
120
+ // @ai-sdk/fireworks both default to api.fireworks.ai; the inference base URL is
121
+ // https://api.fireworks.ai/inference).
122
+ hosts: [{ host: 'api.fireworks.ai' }],
123
+ // No app-read endpoint env, deliberately. The `fireworks` Python client reads FIREWORKS_BASE_URL
124
+ // — but the app-read table is the FALLBACK for packs the injector cannot intercept, and this pack
125
+ // declares `hosts`, so FIREWORKS_TWIN_URL + zero-edit interception is the better answer. Emitting
126
+ // both would quietly forgo interception (`init.test.ts`, "the app-read endpoint table cannot go
127
+ // stale behind the injector").
128
+ endpointEnvNone:
129
+ 'The injector intercepts api.fireworks.ai (see `hosts`), so a world wires this pack through FIREWORKS_TWIN_URL rather than an app-read base-URL var. The `fireworks` Python client would honour FIREWORKS_BASE_URL, but @ai-sdk/fireworks — the dominant client in this repo’s demand census — takes its baseURL as a constructor option and reads no such env, so an app-read var would cover only half the traffic while disabling the interception that covers all of it.',
130
+ };
131
+
132
+ // registered at import: the kernel learns the pack's state system (protocol 2)
133
+ import { registerPack } from '@volter/world-core';
134
+ registerPack(pack);