pi-fireworks-provider 1.1.0 → 1.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +50 -13
- package/index.ts +266 -5
- package/package.json +19 -2
- package/patch.json +0 -30
- package/.github/FUNDING.yml +0 -4
- package/.pi/messenger/channels/memory.jsonl +0 -1
- package/.pi/messenger/session-id +0 -1
- package/AGENTS.md +0 -56
- package/scripts/update-models.js +0 -349
package/README.md
CHANGED
|
@@ -15,9 +15,11 @@ _Kimi, MiniMax, GLM, DeepSeek, GPT-OSS — via Fireworks AI's Anthropic Messages
|
|
|
15
15
|
|
|
16
16
|
## Features
|
|
17
17
|
|
|
18
|
-
- **
|
|
18
|
+
- **39+ AI Models** including Kimi K2.5, MiniMax M2.5, GLM 4.5/4.7/5, DeepSeek V3.1/V3.2, DeepSeek V4 Flash, and GPT-OSS
|
|
19
19
|
- **Dual API support** via Fireworks AI's Anthropic Messages and OpenAI-compatible completions endpoints (per-model routing, matching pi core's Fireworks provider)
|
|
20
20
|
- **Service tiers** — toggle Fireworks `priority` vs `standard` per request on supported models (with priority pricing reflected in cost tracking), via a keybinding, `/fireworks-tier`, and a footer status area
|
|
21
|
+
- **Preserved thinking** — toggle Fireworks' `reasoning_history: "preserved"` so prior assistant reasoning is retained across turns (better multi-turn recall; uses more tokens), via the `/fireworks-settings` panel, with a model-select notification. Matches neuralwatt/makora's settings-only UX, adapted to Fireworks' single global `reasoning_history` knob
|
|
22
|
+
- **Settings panel** — `/fireworks-settings` (TUI) to configure preserved thinking, service tier, and display preferences; persisted to `~/.pi/agent/extensions/fireworks.json`
|
|
21
23
|
- **Cost Tracking** with per-model pricing for budget management
|
|
22
24
|
- **Reasoning Models** support for advanced reasoning capabilities
|
|
23
25
|
- **Vision Support** for image-capable models
|
|
@@ -71,8 +73,8 @@ pi
|
|
|
71
73
|
|-------|------|---------|------------|------------|-------------|
|
|
72
74
|
| DeepSeek V3.1 | Text | 164K | 164K | $0.56 | $1.68 |
|
|
73
75
|
| DeepSeek V3.2 | Text | 164K | 160K | $0.56 | $1.68 |
|
|
74
|
-
| DeepSeek V4 Flash | Text | 1.0M |
|
|
75
|
-
| DeepSeek V4 Pro | Text | 1.0M |
|
|
76
|
+
| DeepSeek V4 Flash | Text | 1.0M | 384K | $0.14 | $0.28 |
|
|
77
|
+
| DeepSeek V4 Pro | Text | 1.0M | 384K | $1.74 | $3.48 |
|
|
76
78
|
| DeepSeek V4 Pro (router) | Text | 1.0M | 1.0M | $1.74 | $3.48 |
|
|
77
79
|
| Gemma 4 26B A4B IT | Text + Image | 262K | 0 | Free | Free |
|
|
78
80
|
| Gemma 4 31B IT | Text + Image | 262K | 0 | Free | Free |
|
|
@@ -80,27 +82,31 @@ pi
|
|
|
80
82
|
| GLM 4.5 Air | Text | 131K | 131K | $0.22 | $0.88 |
|
|
81
83
|
| GLM 4.7 | Text | 203K | 198K | $0.60 | $2.20 |
|
|
82
84
|
| GLM 5 | Text | 203K | 131K | $1.00 | $3.20 |
|
|
83
|
-
| GLM 5 Fast
|
|
85
|
+
| GLM 5 Fast | Text | 203K | 131K | $1.00 | $3.20 |
|
|
84
86
|
| GLM 5.1 | Text | 203K | 131K | $1.40 | $4.40 |
|
|
85
|
-
| GLM 5.1 Fast
|
|
86
|
-
| GLM 5.2 | Text | 1.0M |
|
|
87
|
+
| GLM 5.1 Fast | Text | 203K | 131K | $2.80 | $8.80 |
|
|
88
|
+
| GLM 5.2 | Text | 1.0M | 131K | $1.40 | $4.40 |
|
|
89
|
+
| GLM 5.2 Fast | Text | 1.0M | 131K | $2.10 | $6.60 |
|
|
87
90
|
| GPT OSS 120B | Text | 131K | 33K | $0.15 | $0.60 |
|
|
88
|
-
| GPT OSS 20B | Text | 131K | 33K | $0.
|
|
91
|
+
| GPT OSS 20B | Text | 131K | 33K | $0.07 | $0.30 |
|
|
89
92
|
| Kimi K2 Instruct | Text | 131K | 16K | $1.00 | $3.00 |
|
|
90
93
|
| Kimi K2 Thinking | Text | 262K | 256K | $0.60 | $2.50 |
|
|
91
94
|
| Kimi K2.5 | Text + Image | 262K | 256K | $0.60 | $3.00 |
|
|
92
95
|
| Kimi K2.5 Fast (router) | Text + Image | 262K | 256K | $0.60 | $3.00 |
|
|
93
96
|
| Kimi K2.6 | Text + Image | 262K | 262K | $0.95 | $4.00 |
|
|
94
97
|
| Kimi K2.6 (router) | Text + Image | 262K | 262K | $0.95 | $4.00 |
|
|
95
|
-
| Kimi K2.6
|
|
96
|
-
| Kimi K2.
|
|
98
|
+
| Kimi K2.6 Fast | Text + Image | 262K | 262K | $2.00 | $8.00 |
|
|
99
|
+
| Kimi K2.6 Turbo | Text + Image | 262K | 262K | $2.00 | $8.00 |
|
|
100
|
+
| Kimi K2.7 Code | Text + Image | 262K | 262K | $0.95 | $4.00 |
|
|
101
|
+
| Kimi K2.7 Code Fast | Text + Image | 262K | 262K | $1.90 | $8.00 |
|
|
97
102
|
| Llama 3.3 70B Instruct | Text | 131K | 0 | Free | Free |
|
|
98
103
|
| MiniMax M2.7 (router) | Text | 204K | 0 | $0.30 | $1.20 |
|
|
99
|
-
| Minimax M3 | Text + Image | 512K | 0 | Free | Free |
|
|
100
104
|
| MiniMax-M2.1 | Text | 197K | 200K | $0.30 | $1.20 |
|
|
101
105
|
| MiniMax-M2.5 | Text | 197K | 197K | $0.30 | $1.20 |
|
|
102
106
|
| MiniMax-M2.7 | Text | 197K | 197K | $0.30 | $1.20 |
|
|
107
|
+
| MiniMax-M3 | Text | 512K | 512K | $0.30 | $1.20 |
|
|
103
108
|
| NVIDIA Nemotron 3 Ultra NVFP4 | Text | 262K | 0 | Free | Free |
|
|
109
|
+
| Qwen 3.7 Plus | Text + Image | 262K | 66K | $0.40 | $1.60 |
|
|
104
110
|
| Qwen3 8B | Text | 41K | 0 | Free | Free |
|
|
105
111
|
| Qwen3 VL 30B A3B Instruct | Text + Image | 262K | 0 | Free | Free |
|
|
106
112
|
| Qwen3 VL 30B A3B Thinking | Text + Image | 262K | 0 | Free | Free |
|
|
@@ -140,16 +146,35 @@ The selection is persisted per session (survives `/reload` and resume). When `pr
|
|
|
140
146
|
"default": "standard",
|
|
141
147
|
"keybinding": "ctrl+shift+l",
|
|
142
148
|
"display": "statusbar"
|
|
149
|
+
},
|
|
150
|
+
"preserveThinking": {
|
|
151
|
+
"default": false
|
|
143
152
|
}
|
|
144
153
|
}
|
|
145
154
|
```
|
|
146
155
|
|
|
147
|
-
- `default` — tier used until you toggle (`standard` | `priority`).
|
|
148
|
-
- `keybinding` — any [pi key format](https://github.com/earendil-works/pi-coding-agent/blob/main/docs/keybindings.md) (e.g. `ctrl+shift+l`, `ctrl+shift+k`). Requires `/reload` after changing. On macOS browser terminals (localterm), avoid `alt`/`ctrl+alt` (Option produces special chars) and `ctrl+shift+t/w/n/c/v` (browser/localterm tab + copy/paste shortcuts).
|
|
149
|
-
- `display` — `statusbar` (footer status area) or `off` (hide the tier indicator).
|
|
156
|
+
- `serviceTier.default` — tier used until you toggle (`standard` | `priority`).
|
|
157
|
+
- `serviceTier.keybinding` — any [pi key format](https://github.com/earendil-works/pi-coding-agent/blob/main/docs/keybindings.md) (e.g. `ctrl+shift+l`, `ctrl+shift+k`). Requires `/reload` after changing. On macOS browser terminals (localterm), avoid `alt`/`ctrl+alt` (Option produces special chars) and `ctrl+shift+t/w/n/c/v` (browser/localterm tab + copy/paste shortcuts).
|
|
158
|
+
- `serviceTier.display` — `statusbar` (footer status area) or `off` (hide the tier indicator).
|
|
159
|
+
- `preserveThinking.default` — whether `reasoning_history: "preserved"` is injected (`true` | `false`, default `false`). Also settable via `/fireworks-settings`.
|
|
150
160
|
|
|
151
161
|
> **Note:** The OpenAI completions endpoint accepts `service_tier` directly (per Fireworks' API). The Anthropic Messages endpoint passes the top-level field through as an extra. If a supported Anthropic-routed model rejects it, file an issue so we can gate injection by API.
|
|
152
162
|
|
|
163
|
+
## Preserved Thinking
|
|
164
|
+
|
|
165
|
+
Fireworks exposes a top-level `reasoning_history` request parameter. The only accepted value is `"preserved"`; omitting it (the default) means prior assistant reasoning is **stripped** from the model's context each turn. Setting `reasoning_history: "preserved"` makes Fireworks render prior assistant reasoning into the model's context, improving multi-turn recall at the cost of extra tokens. See [the Fireworks reasoning guide](https://docs.fireworks.ai/guides/reasoning#preserved-thinking).
|
|
166
|
+
|
|
167
|
+
This works on **both** transports — the OpenAI completions endpoint (assistant `reasoning_content` field) and the Anthropic Messages endpoint (assistant `thinking` content blocks, for which Fireworks returns a `signature` so pi-ai replays them). pi-ai already replays the reasoning field/block on prior assistant turns; this extension's only job is injecting the top-level `reasoning_history: "preserved"` flag that makes Fireworks honor it.
|
|
168
|
+
|
|
169
|
+
Unlike neuralwatt/makora (which use per-model vLLM `chat_template_kwargs` flags like `preserve_thinking`/`clear_thinking`), Fireworks' knob is a single global parameter that applies to every reasoning model, so we expose it as one on/off toggle rather than a per-model submenu.
|
|
170
|
+
|
|
171
|
+
**Toggle it:**
|
|
172
|
+
|
|
173
|
+
- **`/fireworks-settings`** (TUI) → *Preserved thinking* → `on` / `off`. Takes effect immediately and persists as the default for future sessions.
|
|
174
|
+
- A dim **model-select notification** tells you the current state when you switch to a Fireworks reasoning model (`Preserved thinking ON for …` / `… OFF for …`).
|
|
175
|
+
|
|
176
|
+
Preserved thinking is **off by default** to match pi core and Fireworks' default (stripped). There is intentionally no `/fireworks-preserve` command or keybinding — it's settings-panel-only, mirroring neuralwatt/makora.
|
|
177
|
+
|
|
153
178
|
## Usage
|
|
154
179
|
|
|
155
180
|
After loading the extension, use the `/model` command in pi to select your preferred model:
|
|
@@ -190,6 +215,18 @@ Add to your pi configuration for automatic loading:
|
|
|
190
215
|
}
|
|
191
216
|
```
|
|
192
217
|
|
|
218
|
+
## Development
|
|
219
|
+
|
|
220
|
+
```bash
|
|
221
|
+
pnpm install # install dev tooling (vitest, knip, typescript)
|
|
222
|
+
pnpm test # run the test suite (vitest)
|
|
223
|
+
pnpm run test:watch # watch mode
|
|
224
|
+
pnpm run lint:dead # dead-code / unused-export scan (knip)
|
|
225
|
+
pnpm run check # typecheck (tsc) + tests + knip, all green or exit non-zero
|
|
226
|
+
```
|
|
227
|
+
|
|
228
|
+
Tests live in `tests/` and stub the `@earendil-works/pi-coding-agent` / `@earendil-works/pi-tui` peer dependencies (see `tests/__mocks__/`) so they run without the real pi packages installed. A per-run temp dir is used for `~/.pi/agent` (via `PI_CODING_AGENT_DIR` in `tests/vitest.setup.ts`) so config/cache reads and writes never touch your real environment.
|
|
229
|
+
|
|
193
230
|
## License
|
|
194
231
|
|
|
195
232
|
MIT
|
package/index.ts
CHANGED
|
@@ -366,14 +366,21 @@ const PRIORITY_PRICING: Record<string, { input: number; output: number; cacheRea
|
|
|
366
366
|
|
|
367
367
|
type ServiceTier = "standard" | "priority";
|
|
368
368
|
|
|
369
|
+
type PreserveMode = boolean; // true = inject reasoning_history:"preserved"; false = default (stripped)
|
|
370
|
+
|
|
369
371
|
interface ServiceTierConfig {
|
|
370
372
|
default: ServiceTier;
|
|
371
373
|
keybinding: string;
|
|
372
374
|
display: "statusbar" | "off";
|
|
373
375
|
}
|
|
374
376
|
|
|
377
|
+
interface PreserveThinkingConfig {
|
|
378
|
+
default: PreserveMode;
|
|
379
|
+
}
|
|
380
|
+
|
|
375
381
|
interface FireworksConfig {
|
|
376
382
|
serviceTier: ServiceTierConfig;
|
|
383
|
+
preserveThinking: PreserveThinkingConfig;
|
|
377
384
|
}
|
|
378
385
|
|
|
379
386
|
const FIREWORKS_CONFIG_PATH = path.join(getAgentDir(), "extensions", "fireworks.json");
|
|
@@ -384,22 +391,40 @@ const DEFAULT_SERVICE_TIER_CONFIG: ServiceTierConfig = {
|
|
|
384
391
|
keybinding: "ctrl+shift+l",
|
|
385
392
|
display: "statusbar",
|
|
386
393
|
};
|
|
387
|
-
|
|
394
|
+
// Preserved thinking is OFF by default to match pi core and Fireworks' default
|
|
395
|
+
// (`reasoning_history` omitted = prior reasoning stripped). Opt in via the
|
|
396
|
+
// /fireworks-settings panel (mirrors neuralwatt/makora: settings-only, no
|
|
397
|
+
// command or keybinding).
|
|
398
|
+
const DEFAULT_PRESERVE_CONFIG: PreserveThinkingConfig = {
|
|
399
|
+
default: false,
|
|
400
|
+
};
|
|
401
|
+
const DEFAULT_FIREWORKS_CONFIG: FireworksConfig = {
|
|
402
|
+
serviceTier: DEFAULT_SERVICE_TIER_CONFIG,
|
|
403
|
+
preserveThinking: DEFAULT_PRESERVE_CONFIG,
|
|
404
|
+
};
|
|
388
405
|
|
|
389
406
|
function isValidTier(v: unknown): v is ServiceTier {
|
|
390
407
|
return v === "standard" || v === "priority";
|
|
391
408
|
}
|
|
392
409
|
|
|
410
|
+
function isValidKeybinding(v: unknown): v is string {
|
|
411
|
+
return typeof v === "string" && v.length > 0;
|
|
412
|
+
}
|
|
413
|
+
|
|
393
414
|
function loadFireworksConfig(): FireworksConfig {
|
|
394
415
|
try {
|
|
395
416
|
const raw = JSON.parse(fs.readFileSync(FIREWORKS_CONFIG_PATH, "utf8"));
|
|
396
417
|
const st = raw?.serviceTier ?? {};
|
|
418
|
+
const pt = raw?.preserveThinking ?? {};
|
|
397
419
|
return {
|
|
398
420
|
serviceTier: {
|
|
399
421
|
default: isValidTier(st.default) ? st.default : DEFAULT_SERVICE_TIER_CONFIG.default,
|
|
400
|
-
keybinding:
|
|
422
|
+
keybinding: isValidKeybinding(st.keybinding) ? st.keybinding : DEFAULT_SERVICE_TIER_CONFIG.keybinding,
|
|
401
423
|
display: st.display === "off" ? "off" : "statusbar",
|
|
402
424
|
},
|
|
425
|
+
preserveThinking: {
|
|
426
|
+
default: typeof pt.default === "boolean" ? pt.default : DEFAULT_PRESERVE_CONFIG.default,
|
|
427
|
+
},
|
|
403
428
|
};
|
|
404
429
|
} catch {
|
|
405
430
|
// Config missing or invalid — write defaults so the user can discover it.
|
|
@@ -409,7 +434,27 @@ function loadFireworksConfig(): FireworksConfig {
|
|
|
409
434
|
} catch {
|
|
410
435
|
// Write failure is non-fatal — defaults still work in memory.
|
|
411
436
|
}
|
|
412
|
-
return { serviceTier: { ...DEFAULT_SERVICE_TIER_CONFIG } };
|
|
437
|
+
return { serviceTier: { ...DEFAULT_SERVICE_TIER_CONFIG }, preserveThinking: { ...DEFAULT_PRESERVE_CONFIG } };
|
|
438
|
+
}
|
|
439
|
+
}
|
|
440
|
+
|
|
441
|
+
// Read-modify-write the raw config JSON without re-validating, so unrelated
|
|
442
|
+
// fields a user added survive a settings-UI write. `loadFireworksConfig()`
|
|
443
|
+
// (validated) is still called after writing to refresh the in-memory config.
|
|
444
|
+
function readRawFireworksConfig(): Record<string, any> {
|
|
445
|
+
try {
|
|
446
|
+
return JSON.parse(fs.readFileSync(FIREWORKS_CONFIG_PATH, "utf8"));
|
|
447
|
+
} catch {
|
|
448
|
+
return JSON.parse(JSON.stringify(DEFAULT_FIREWORKS_CONFIG));
|
|
449
|
+
}
|
|
450
|
+
}
|
|
451
|
+
|
|
452
|
+
function writeRawFireworksConfig(raw: Record<string, any>): void {
|
|
453
|
+
try {
|
|
454
|
+
fs.mkdirSync(path.dirname(FIREWORKS_CONFIG_PATH), { recursive: true });
|
|
455
|
+
fs.writeFileSync(FIREWORKS_CONFIG_PATH, JSON.stringify(raw, null, 2) + "\n");
|
|
456
|
+
} catch {
|
|
457
|
+
// Write failure is non-fatal — the in-memory refresh still applies.
|
|
413
458
|
}
|
|
414
459
|
}
|
|
415
460
|
|
|
@@ -508,14 +553,108 @@ function recomputePriorityCost(message: any): any {
|
|
|
508
553
|
return { ...message, usage: { ...usage, cost } };
|
|
509
554
|
}
|
|
510
555
|
|
|
556
|
+
// ─── Preserved Thinking (reasoning_history) ──────────────────────────────────
|
|
557
|
+
|
|
558
|
+
// Fireworks exposes a top-level `reasoning_history` request parameter. The
|
|
559
|
+
// only accepted value is `"preserved"`; omitting it (or any other value)
|
|
560
|
+
// means prior assistant reasoning is STRIPPED from the model's context
|
|
561
|
+
// (verified by e2e: reasoning_content on an assistant message yields 0%
|
|
562
|
+
// multi-turn recall without the flag, 100% with it). This works on BOTH
|
|
563
|
+
// transports — the OpenAI completions endpoint (assistant `reasoning_content`
|
|
564
|
+
// field) and the Anthropic Messages endpoint (assistant `thinking` content
|
|
565
|
+
// block, which Firebooks returns a `signature` for so pi-ai replays it).
|
|
566
|
+
//
|
|
567
|
+
// Unlike neuralwatt/makora (which use per-model vLLM `chat_template_kwargs`
|
|
568
|
+
// flags like preserve_thinking/clear_thinking), Fireworks' knob is a single
|
|
569
|
+
// global top-level param that applies to every reasoning model. We expose it
|
|
570
|
+
// as one on/off toggle. See https://docs.fireworks.ai/guides/reasoning#preserved-thinking.
|
|
571
|
+
|
|
572
|
+
// A Fireworks model is preserve-eligible if it's a reasoning model. We read
|
|
573
|
+
// `reasoning` off the registered model when available, but also accept a
|
|
574
|
+
// truthy `reasoning` flag on ctx.model (pi-ai attaches it).
|
|
575
|
+
function isPreserveEligible(model: any): boolean {
|
|
576
|
+
if (!model || model.provider !== "fireworks") return false;
|
|
577
|
+
return model.reasoning === true;
|
|
578
|
+
}
|
|
579
|
+
|
|
580
|
+
// Runtime state: whether preserved thinking is active. Initialized from the
|
|
581
|
+
// config-file default at session_start (mirrors neuralwatt/makora, which drive
|
|
582
|
+
// preserve state from the config file, not session entries) and updated by the
|
|
583
|
+
// /fireworks-settings panel, which also persists the new default.
|
|
584
|
+
let preserveOn: boolean = fireworksConfig.preserveThinking.default;
|
|
585
|
+
|
|
586
|
+
function setPreserve(on: boolean): void {
|
|
587
|
+
preserveOn = on;
|
|
588
|
+
}
|
|
589
|
+
|
|
511
590
|
// ─── Extension Entry Point ────────────────────────────────────────────────────
|
|
512
591
|
|
|
592
|
+
export {
|
|
593
|
+
applyPatch,
|
|
594
|
+
buildModels,
|
|
595
|
+
isChatModel,
|
|
596
|
+
transformApiModel,
|
|
597
|
+
mergeWithEmbedded,
|
|
598
|
+
loadStaleModels,
|
|
599
|
+
isFireworksKimiModel,
|
|
600
|
+
sanitizePattern,
|
|
601
|
+
sanitizeSchemaForKimi,
|
|
602
|
+
stripAnchorBleedInPlace,
|
|
603
|
+
isValidTier,
|
|
604
|
+
isValidKeybinding,
|
|
605
|
+
loadFireworksConfig,
|
|
606
|
+
readRawFireworksConfig,
|
|
607
|
+
writeRawFireworksConfig,
|
|
608
|
+
isPriorityApplicable,
|
|
609
|
+
PRIORITY_PRICING,
|
|
610
|
+
recomputePriorityCost,
|
|
611
|
+
replayTierState,
|
|
612
|
+
setTier,
|
|
613
|
+
updateTierStatus,
|
|
614
|
+
isPreserveEligible,
|
|
615
|
+
setPreserve,
|
|
616
|
+
};
|
|
617
|
+
|
|
618
|
+
export type {
|
|
619
|
+
JsonModel,
|
|
620
|
+
PatchEntry,
|
|
621
|
+
PatchData,
|
|
622
|
+
FireworksApi,
|
|
623
|
+
ServiceTier,
|
|
624
|
+
ServiceTierConfig,
|
|
625
|
+
PreserveThinkingConfig,
|
|
626
|
+
FireworksConfig,
|
|
627
|
+
};
|
|
628
|
+
|
|
513
629
|
export default function (pi: ExtensionAPI) {
|
|
514
630
|
piRef = pi;
|
|
515
631
|
const embeddedModels = modelsData as JsonModel[];
|
|
516
632
|
const customModels = customModelsData as JsonModel[];
|
|
517
633
|
const patches = patchData as PatchData;
|
|
518
634
|
|
|
635
|
+
// Deferred model_select notify timer — cleared on rapid re-switch and on
|
|
636
|
+
// session_shutdown so only the latest switch notifies. Mirrors neuralwatt's
|
|
637
|
+
// pattern so pi core's (and other extensions') notifications land first.
|
|
638
|
+
let modelSelectNotifyTimer: ReturnType<typeof setTimeout> | null = null;
|
|
639
|
+
const MODEL_SELECT_NOTIFY_DELAY_MS = 250;
|
|
640
|
+
|
|
641
|
+
// Notify preserved-thinking state for a Fireworks reasoning model. Deferred
|
|
642
|
+
// so pi core's notifications land first; cancelled on re-switch/shutdown so
|
|
643
|
+
// only the latest shows. Always level "info" — the text conveys the
|
|
644
|
+
// preserve/strip tradeoff, not a warning.
|
|
645
|
+
function notifyPreserveOnSelect(model: any, ctx: any): void {
|
|
646
|
+
if (!model || model.provider !== "fireworks") return;
|
|
647
|
+
if (!isPreserveEligible(model)) return;
|
|
648
|
+
const msg = preserveOn
|
|
649
|
+
? `Preserved thinking ON for ${model.name || model.id} — full reasoning history retained across turns (better multi-turn recall; uses more tokens). Open /fireworks-settings to change.`
|
|
650
|
+
: `Preserved thinking OFF for ${model.name || model.id} — reasoning stripped each turn (Fireworks default; lighter, weaker multi-turn recall). Open /fireworks-settings to change.`;
|
|
651
|
+
if (modelSelectNotifyTimer) clearTimeout(modelSelectNotifyTimer);
|
|
652
|
+
modelSelectNotifyTimer = setTimeout(() => {
|
|
653
|
+
modelSelectNotifyTimer = null;
|
|
654
|
+
try { ctx.ui.notify(msg, "info"); } catch { /* notify is a no-op without a UI runner */ }
|
|
655
|
+
}, MODEL_SELECT_NOTIFY_DELAY_MS);
|
|
656
|
+
}
|
|
657
|
+
|
|
519
658
|
const staleBase = loadStaleModels(embeddedModels);
|
|
520
659
|
const staleModels = buildModels(staleBase, customModels, patches);
|
|
521
660
|
|
|
@@ -532,7 +671,9 @@ export default function (pi: ExtensionAPI) {
|
|
|
532
671
|
const signal = revalidateAbort.signal;
|
|
533
672
|
fireworksConfig = loadFireworksConfig();
|
|
534
673
|
replayTierState(ctx, fireworksConfig.serviceTier.default);
|
|
674
|
+
preserveOn = fireworksConfig.preserveThinking.default;
|
|
535
675
|
updateTierStatus(ctx);
|
|
676
|
+
notifyPreserveOnSelect(ctx.model, ctx);
|
|
536
677
|
resolveApiKey(ctx.modelRegistry).then(() => {
|
|
537
678
|
revalidateModels(cachedApiKey, embeddedModels, signal).then((freshBase) => {
|
|
538
679
|
if (freshBase && !signal.aborted) {
|
|
@@ -550,6 +691,7 @@ export default function (pi: ExtensionAPI) {
|
|
|
550
691
|
pi.on("session_shutdown", (_event, ctx) => {
|
|
551
692
|
revalidateAbort?.abort();
|
|
552
693
|
try { ctx.ui.setStatus(TIER_STATUS_KEY, undefined); } catch {}
|
|
694
|
+
if (modelSelectNotifyTimer) { clearTimeout(modelSelectNotifyTimer); modelSelectNotifyTimer = null; }
|
|
553
695
|
});
|
|
554
696
|
|
|
555
697
|
// Sanitize JSON Schema patterns for Kimi models before sending to Fireworks.
|
|
@@ -576,6 +718,19 @@ export default function (pi: ExtensionAPI) {
|
|
|
576
718
|
modified = true;
|
|
577
719
|
}
|
|
578
720
|
|
|
721
|
+
// Preserved thinking: inject top-level `reasoning_history: "preserved"`
|
|
722
|
+
// so Fireworks renders prior assistant reasoning (reasoning_content on the
|
|
723
|
+
// OpenAI endpoint, thinking blocks on the Anthropic endpoint) into the
|
|
724
|
+
// model's context instead of stripping it. The only accepted value is
|
|
725
|
+
// "preserved"; omitted = stripped (Fireworks default / pi core). Applies to
|
|
726
|
+
// any Fireworks reasoning model on both transports. pi-ai already replays
|
|
727
|
+
// the reasoning field/block on prior assistant turns; this flag is what
|
|
728
|
+
// makes Fireworks honor it. See https://docs.fireworks.ai/guides/reasoning.
|
|
729
|
+
if (preserveOn && isPreserveEligible(model)) {
|
|
730
|
+
payload.reasoning_history = "preserved";
|
|
731
|
+
modified = true;
|
|
732
|
+
}
|
|
733
|
+
|
|
579
734
|
// Kimi anchor-bleed sanitization (Kimi K2.x pattern bug). Only applies to
|
|
580
735
|
// Kimi models, but a single request can be both Kimi and priority-tiered.
|
|
581
736
|
if (isFireworksKimiModel(model)) {
|
|
@@ -623,9 +778,12 @@ export default function (pi: ExtensionAPI) {
|
|
|
623
778
|
}
|
|
624
779
|
});
|
|
625
780
|
|
|
626
|
-
// Refresh the tier status area when the active model changes
|
|
627
|
-
|
|
781
|
+
// Refresh the tier status area when the active model changes, and notify
|
|
782
|
+
// preserved-thinking state for the selected reasoning model (mirrors
|
|
783
|
+
// neuralwatt/makora: notification only, no preserve status area).
|
|
784
|
+
pi.on("model_select", async (event, ctx) => {
|
|
628
785
|
updateTierStatus(ctx);
|
|
786
|
+
notifyPreserveOnSelect(event.model ?? ctx.model, ctx);
|
|
629
787
|
});
|
|
630
788
|
|
|
631
789
|
// Recompute finalized assistant-message cost against priority pricing so
|
|
@@ -672,4 +830,107 @@ export default function (pi: ExtensionAPI) {
|
|
|
672
830
|
}
|
|
673
831
|
},
|
|
674
832
|
});
|
|
833
|
+
|
|
834
|
+
// /fireworks-settings: TUI settings panel (mirrors /neuralwatt-settings &
|
|
835
|
+
// /makora-settings). Opens a SettingsList via ctx.ui.custom(). Toggles write
|
|
836
|
+
// to ~/.pi/agent/extensions/fireworks.json (raw read-modify-write so unknown
|
|
837
|
+
// fields survive) and refresh the in-memory config. Preserved thinking is
|
|
838
|
+
// settings-only (no command/keybinding), exactly like the siblings; the
|
|
839
|
+
// service-tier keybinding is load-time only, so keybinding changes need /reload.
|
|
840
|
+
pi.registerCommand("fireworks-settings", {
|
|
841
|
+
description: "Configure Fireworks: preserved thinking + service tier + display",
|
|
842
|
+
async handler(_args, ctx) {
|
|
843
|
+
if (ctx.mode !== "tui") {
|
|
844
|
+
ctx.ui.notify("/fireworks-settings requires TUI mode.", "error");
|
|
845
|
+
return;
|
|
846
|
+
}
|
|
847
|
+
const { SettingsList, Container } = await import("@earendil-works/pi-tui");
|
|
848
|
+
const { getSettingsListTheme, DynamicBorder } = await import("@earendil-works/pi-coding-agent");
|
|
849
|
+
|
|
850
|
+
await ctx.ui.custom((_tui, theme, _kb, done) => {
|
|
851
|
+
const border = () => new DynamicBorder((s: string) => theme.fg("border", s));
|
|
852
|
+
|
|
853
|
+
const items: any[] = [
|
|
854
|
+
{
|
|
855
|
+
id: "preserveThinking",
|
|
856
|
+
label: "Preserved thinking",
|
|
857
|
+
description: "Inject reasoning_history:\"preserved\" so Fireworks retains prior assistant reasoning across turns (better multi-turn recall; uses more tokens). Off = Fireworks default (stripped). Applies to every Fireworks reasoning model on both endpoints.",
|
|
858
|
+
currentValue: preserveOn ? "on" : "off",
|
|
859
|
+
values: ["on", "off"],
|
|
860
|
+
},
|
|
861
|
+
{
|
|
862
|
+
id: "serviceTier",
|
|
863
|
+
label: "Service tier",
|
|
864
|
+
description: "Fireworks service tier for supported models. priority = higher throughput / latency at ~1.2\u20131.5\u00d7 cost (priority pricing reflected in cost tracking). standard = default.",
|
|
865
|
+
currentValue: currentTier,
|
|
866
|
+
values: ["standard", "priority"],
|
|
867
|
+
},
|
|
868
|
+
{
|
|
869
|
+
id: "serviceTier.display",
|
|
870
|
+
label: "Service tier display",
|
|
871
|
+
description: "Where the \u201ctier: standard / \u26a1priority\u201d indicator is shown: footer status area or hidden",
|
|
872
|
+
currentValue: fireworksConfig.serviceTier.display,
|
|
873
|
+
values: ["statusbar", "off"],
|
|
874
|
+
},
|
|
875
|
+
];
|
|
876
|
+
|
|
877
|
+
const container = new Container();
|
|
878
|
+
container.addChild(border());
|
|
879
|
+
|
|
880
|
+
const settingsList = new SettingsList(
|
|
881
|
+
items,
|
|
882
|
+
Math.min(items.length + 2, 15),
|
|
883
|
+
getSettingsListTheme(),
|
|
884
|
+
(id: string, newValue: string) => {
|
|
885
|
+
if (id === "preserveThinking") {
|
|
886
|
+
const on = newValue === "on";
|
|
887
|
+
setPreserve(on);
|
|
888
|
+
// Persist as the default too, so it survives new sessions.
|
|
889
|
+
const raw = readRawFireworksConfig();
|
|
890
|
+
raw.preserveThinking = { ...(raw.preserveThinking ?? {}), default: on };
|
|
891
|
+
writeRawFireworksConfig(raw);
|
|
892
|
+
fireworksConfig = loadFireworksConfig();
|
|
893
|
+
ctx.ui.notify(`Preserved thinking ${on ? "on" : "off"} \u2014 takes effect now.`, "info");
|
|
894
|
+
} else if (id === "serviceTier") {
|
|
895
|
+
let model: any;
|
|
896
|
+
try { model = ctx.model; } catch { model = undefined; }
|
|
897
|
+
if (!model || model.provider !== "fireworks" || !isPriorityApplicable(model.id)) {
|
|
898
|
+
ctx.ui.notify("Service tier only applies to supported Fireworks models.", "info");
|
|
899
|
+
return;
|
|
900
|
+
}
|
|
901
|
+
const tier = newValue as ServiceTier;
|
|
902
|
+
setTier(ctx, tier);
|
|
903
|
+
const raw = readRawFireworksConfig();
|
|
904
|
+
raw.serviceTier = { ...(raw.serviceTier ?? {}), default: tier };
|
|
905
|
+
writeRawFireworksConfig(raw);
|
|
906
|
+
fireworksConfig = loadFireworksConfig();
|
|
907
|
+
ctx.ui.notify(`Fireworks service tier: ${tier}`, "info");
|
|
908
|
+
} else if (id === "serviceTier.display") {
|
|
909
|
+
const raw = readRawFireworksConfig();
|
|
910
|
+
raw.serviceTier = { ...(raw.serviceTier ?? {}), display: newValue };
|
|
911
|
+
writeRawFireworksConfig(raw);
|
|
912
|
+
fireworksConfig = loadFireworksConfig();
|
|
913
|
+
updateTierStatus(ctx);
|
|
914
|
+
}
|
|
915
|
+
},
|
|
916
|
+
() => done(undefined),
|
|
917
|
+
{ enableSearch: true },
|
|
918
|
+
);
|
|
919
|
+
container.addChild(settingsList);
|
|
920
|
+
container.addChild(border());
|
|
921
|
+
|
|
922
|
+
return {
|
|
923
|
+
render(width: number) {
|
|
924
|
+
return container.render(width);
|
|
925
|
+
},
|
|
926
|
+
invalidate() {
|
|
927
|
+
container.invalidate();
|
|
928
|
+
},
|
|
929
|
+
handleInput(data: string) {
|
|
930
|
+
settingsList.handleInput?.(data);
|
|
931
|
+
},
|
|
932
|
+
};
|
|
933
|
+
});
|
|
934
|
+
},
|
|
935
|
+
});
|
|
675
936
|
}
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "pi-fireworks-provider",
|
|
3
|
-
"version": "1.
|
|
3
|
+
"version": "1.2.0",
|
|
4
4
|
"description": "Fireworks AI provider extension for pi - Access Kimi, MiniMax, GLM, DeepSeek, and GPT-OSS models through the Fireworks AI API",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"main": "index.ts",
|
|
@@ -19,6 +19,20 @@
|
|
|
19
19
|
],
|
|
20
20
|
"author": "",
|
|
21
21
|
"license": "MIT",
|
|
22
|
+
"files": [
|
|
23
|
+
"index.ts",
|
|
24
|
+
"models.json",
|
|
25
|
+
"custom-models.json",
|
|
26
|
+
"patch.json",
|
|
27
|
+
"README.md",
|
|
28
|
+
"LICENSE"
|
|
29
|
+
],
|
|
30
|
+
"devDependencies": {
|
|
31
|
+
"@types/node": "^22.20.0",
|
|
32
|
+
"knip": "6.14.1",
|
|
33
|
+
"typescript": "6.0.3",
|
|
34
|
+
"vitest": "4.1.7"
|
|
35
|
+
},
|
|
22
36
|
"pi": {
|
|
23
37
|
"extensions": [
|
|
24
38
|
"./index.ts"
|
|
@@ -27,7 +41,10 @@
|
|
|
27
41
|
"scripts": {
|
|
28
42
|
"clean": "echo 'nothing to clean'",
|
|
29
43
|
"build": "echo 'nothing to build'",
|
|
30
|
-
"check": "
|
|
44
|
+
"check": "tsc --noEmit && vitest run && knip --no-gitignore",
|
|
45
|
+
"lint:dead": "knip --no-gitignore",
|
|
46
|
+
"test": "vitest run",
|
|
47
|
+
"test:watch": "vitest",
|
|
31
48
|
"update-models": "node scripts/update-models.js"
|
|
32
49
|
}
|
|
33
50
|
}
|
package/patch.json
CHANGED
|
@@ -417,22 +417,6 @@
|
|
|
417
417
|
"supportsLongCacheRetention": false
|
|
418
418
|
}
|
|
419
419
|
},
|
|
420
|
-
"accounts/fireworks/models/qwen3p6-plus": {
|
|
421
|
-
"name": "Qwen 3.6 Plus",
|
|
422
|
-
"reasoning": true,
|
|
423
|
-
"input": ["text", "image"],
|
|
424
|
-
"cost": {
|
|
425
|
-
"input": 0.5,
|
|
426
|
-
"output": 3,
|
|
427
|
-
"cacheRead": 0.1,
|
|
428
|
-
"cacheWrite": 0
|
|
429
|
-
},
|
|
430
|
-
"contextWindow": 262144,
|
|
431
|
-
"maxTokens": 8192,
|
|
432
|
-
"compat": {
|
|
433
|
-
"supportsReasoningEffort": true
|
|
434
|
-
}
|
|
435
|
-
},
|
|
436
420
|
"accounts/fireworks/models/qwen3p7-plus": {
|
|
437
421
|
"name": "Qwen 3.7 Plus",
|
|
438
422
|
"api": "anthropic-messages",
|
|
@@ -544,20 +528,6 @@
|
|
|
544
528
|
"supportsReasoningEffort": true
|
|
545
529
|
}
|
|
546
530
|
},
|
|
547
|
-
"accounts/fireworks/routers/kimi-k2p5-turbo": {
|
|
548
|
-
"name": "Kimi K2.5 Turbo (router)",
|
|
549
|
-
"reasoning": true,
|
|
550
|
-
"cost": {
|
|
551
|
-
"input": 0.6,
|
|
552
|
-
"output": 3,
|
|
553
|
-
"cacheRead": 0.1,
|
|
554
|
-
"cacheWrite": 0
|
|
555
|
-
},
|
|
556
|
-
"maxTokens": 256000,
|
|
557
|
-
"compat": {
|
|
558
|
-
"supportsReasoningEffort": true
|
|
559
|
-
}
|
|
560
|
-
},
|
|
561
531
|
"accounts/fireworks/routers/kimi-k2p6": {
|
|
562
532
|
"name": "Kimi K2.6 (router)",
|
|
563
533
|
"reasoning": true,
|
package/.github/FUNDING.yml
DELETED
|
@@ -1 +0,0 @@
|
|
|
1
|
-
{"_meta":true,"v":1,"id":"memory","type":"named","createdAt":"2026-05-25T08:03:16.155Z","description":"Cross-session knowledge and insights"}
|
package/.pi/messenger/session-id
DELETED
|
@@ -1 +0,0 @@
|
|
|
1
|
-
019e98a1-b474-7d6d-b80e-b1dedf22aa71
|
package/AGENTS.md
DELETED
|
@@ -1,56 +0,0 @@
|
|
|
1
|
-
# AGENTS.md
|
|
2
|
-
|
|
3
|
-
## DO NOT EDIT — Auto-generated Files
|
|
4
|
-
|
|
5
|
-
The following files are **idempotent** and regenerated by `scripts/update-models.js`. Never edit them directly — your changes will be overwritten on the next model sync.
|
|
6
|
-
|
|
7
|
-
| File | Why it's auto-generated |
|
|
8
|
-
|------|------------------------|
|
|
9
|
-
| `models.json` | Built from the provider API. `update-models.js` fetches models, preserves curated data for known IDs, and writes this file. |
|
|
10
|
-
| `README.md` (model table) | The table under `## Available Models` is replaced in-place by `update-models.js` after merging base models → patch → custom models. |
|
|
11
|
-
|
|
12
|
-
## Correct Files to Edit
|
|
13
|
-
|
|
14
|
-
When a model needs overrides, new properties, or corrections, edit the appropriate source file below. These are the **source of truth** that the update script reads but never writes.
|
|
15
|
-
|
|
16
|
-
| File | Purpose |
|
|
17
|
-
|------|---------|
|
|
18
|
-
| `patch.json` | Per-model overrides keyed by model ID. Add reasoning flags, compat settings, pricing corrections, thinking level maps, etc. Applied on top of `models.json` at runtime and for README generation. |
|
|
19
|
-
| `custom-models.json` | Models that don't exist in the provider API (hidden models, router endpoints, cross-provider aliases). Merged after patch. Format: array of full model objects (same schema as `models.json` entries). |
|
|
20
|
-
| `index.ts` | Provider extension code. |
|
|
21
|
-
| `scripts/update-models.js` | The sync script itself (edit only if changing how models are fetched/transformed). |
|
|
22
|
-
|
|
23
|
-
## Data Flow
|
|
24
|
-
|
|
25
|
-
```
|
|
26
|
-
Provider API ──fetch──► models.json ──apply──► patch.json ──merge──► custom-models.json
|
|
27
|
-
│ │ │
|
|
28
|
-
└────────────────────────────┴──────────────────────┘
|
|
29
|
-
│
|
|
30
|
-
README model table
|
|
31
|
-
```
|
|
32
|
-
|
|
33
|
-
1. `models.json` — base data from the provider API (auto-generated, DO NOT EDIT)
|
|
34
|
-
2. `patch.json` — overrides applied on top (EDIT THIS for corrections/enrichments)
|
|
35
|
-
3. `custom-models.json` — additional models not in the API (EDIT THIS for new models)
|
|
36
|
-
4. README table — rendered from the merged result of all three (auto-generated, DO NOT EDIT)
|
|
37
|
-
|
|
38
|
-
## Common Tasks
|
|
39
|
-
|
|
40
|
-
### Add a compat setting or override pricing for an existing model
|
|
41
|
-
→ Edit `patch.json`. Add an entry keyed by the model's `id`.
|
|
42
|
-
|
|
43
|
-
### Add a model not available in the provider API
|
|
44
|
-
→ Edit `custom-models.json`. Add a full model object to the array.
|
|
45
|
-
|
|
46
|
-
### Update models from the provider API
|
|
47
|
-
→ Run `node scripts/update-models.js` (may require an API key env var).
|
|
48
|
-
|
|
49
|
-
### Regenerate the README model table
|
|
50
|
-
→ Run `node scripts/update-models.js` — it updates both `models.json` and the README table.
|
|
51
|
-
|
|
52
|
-
## TL;DR
|
|
53
|
-
|
|
54
|
-
- **Never edit `models.json`** — edit `patch.json` instead.
|
|
55
|
-
- **Never edit the README model table** — run the update script instead.
|
|
56
|
-
- `patch.json` and `custom-models.json` are the source files you should modify.
|
package/scripts/update-models.js
DELETED
|
@@ -1,349 +0,0 @@
|
|
|
1
|
-
#!/usr/bin/env node
|
|
2
|
-
|
|
3
|
-
/**
|
|
4
|
-
* Script to update fireworks models from the Fireworks API
|
|
5
|
-
*
|
|
6
|
-
* Uses the official Fireworks Gateway REST API to discover available models:
|
|
7
|
-
* GET /v1/accounts/fireworks/models
|
|
8
|
-
*
|
|
9
|
-
* Requires FIREWORKS_API_KEY environment variable.
|
|
10
|
-
* Usage: FIREWORKS_API_KEY=your-key node scripts/update-models.js
|
|
11
|
-
*
|
|
12
|
-
* Data flow:
|
|
13
|
-
* models.json → auto-generated from Fireworks API (model discovery)
|
|
14
|
-
* patch.json → manual overrides (pricing, reasoning, limits, etc.)
|
|
15
|
-
* custom-models.json → hidden/router models not in the API
|
|
16
|
-
*
|
|
17
|
-
* The API provides: id, displayName, contextLength, supportsImageInput,
|
|
18
|
-
* supportsTools, supportsServerless, state, kind, moe, parameterCount, etc.
|
|
19
|
-
*
|
|
20
|
-
* It does NOT provide: pricing, max output tokens, reasoning mode, or
|
|
21
|
-
* interleaved thinking details. Those come from patch.json.
|
|
22
|
-
*
|
|
23
|
-
* Merge order for README: models.json → apply patch.json → merge custom-models.json
|
|
24
|
-
*/
|
|
25
|
-
|
|
26
|
-
import https from 'https';
|
|
27
|
-
import fs from 'fs';
|
|
28
|
-
import path from 'path';
|
|
29
|
-
import { fileURLToPath } from 'url';
|
|
30
|
-
|
|
31
|
-
const __filename = fileURLToPath(import.meta.url);
|
|
32
|
-
const __dirname = path.dirname(__filename);
|
|
33
|
-
|
|
34
|
-
const FIREWORKS_API_BASE = 'https://api.fireworks.ai';
|
|
35
|
-
const ACCOUNT_ID = 'fireworks';
|
|
36
|
-
const MODELS_PATH = path.join(process.cwd(), 'models.json');
|
|
37
|
-
const CUSTOM_MODELS_PATH = path.join(process.cwd(), 'custom-models.json');
|
|
38
|
-
const PATCH_PATH = path.join(process.cwd(), 'patch.json');
|
|
39
|
-
|
|
40
|
-
// ─── HTTP helpers ───────────────────────────────────────────────────────────
|
|
41
|
-
|
|
42
|
-
function fetchJSON(url, headers = {}) {
|
|
43
|
-
return new Promise((resolve, reject) => {
|
|
44
|
-
const req = https.get(url, { headers }, (res) => {
|
|
45
|
-
let data = '';
|
|
46
|
-
res.on('data', (chunk) => (data += chunk));
|
|
47
|
-
res.on('end', () => {
|
|
48
|
-
try {
|
|
49
|
-
resolve(JSON.parse(data));
|
|
50
|
-
} catch (e) {
|
|
51
|
-
reject(new Error(`Failed to parse JSON from ${url}: ${e.message}`));
|
|
52
|
-
}
|
|
53
|
-
});
|
|
54
|
-
});
|
|
55
|
-
req.on('error', reject);
|
|
56
|
-
});
|
|
57
|
-
}
|
|
58
|
-
|
|
59
|
-
/**
|
|
60
|
-
* Paginate through the Fireworks account models API.
|
|
61
|
-
* Returns all models across all pages.
|
|
62
|
-
*/
|
|
63
|
-
async function fetchAllFireworksModels(apiKey) {
|
|
64
|
-
const headers = {};
|
|
65
|
-
if (apiKey) headers.Authorization = `Bearer ${apiKey}`;
|
|
66
|
-
|
|
67
|
-
const allModels = [];
|
|
68
|
-
let pageToken = undefined;
|
|
69
|
-
let page = 0;
|
|
70
|
-
|
|
71
|
-
do {
|
|
72
|
-
let url = `${FIREWORKS_API_BASE}/v1/accounts/${ACCOUNT_ID}/models?pageSize=200`;
|
|
73
|
-
if (pageToken) url += `&pageToken=${pageToken}`;
|
|
74
|
-
|
|
75
|
-
const data = await fetchJSON(url, headers);
|
|
76
|
-
const models = data.models || [];
|
|
77
|
-
allModels.push(...models);
|
|
78
|
-
|
|
79
|
-
pageToken = data.nextPageToken || undefined;
|
|
80
|
-
page++;
|
|
81
|
-
console.log(` Page ${page}: fetched ${models.length} models (total so far: ${allModels.length})`);
|
|
82
|
-
} while (pageToken);
|
|
83
|
-
|
|
84
|
-
return allModels;
|
|
85
|
-
}
|
|
86
|
-
|
|
87
|
-
// ─── File I/O ───────────────────────────────────────────────────────────────
|
|
88
|
-
|
|
89
|
-
function loadJSON(filePath) {
|
|
90
|
-
try {
|
|
91
|
-
if (!fs.existsSync(filePath)) return [];
|
|
92
|
-
const data = fs.readFileSync(filePath, 'utf8');
|
|
93
|
-
const parsed = JSON.parse(data);
|
|
94
|
-
console.log(`✓ Loaded ${Array.isArray(parsed) ? parsed.length : Object.keys(parsed).length} entries from ${path.basename(filePath)}`);
|
|
95
|
-
return parsed;
|
|
96
|
-
} catch (e) {
|
|
97
|
-
console.warn(`Warning: Could not load ${path.basename(filePath)}: ${e.message}`);
|
|
98
|
-
return {};
|
|
99
|
-
}
|
|
100
|
-
}
|
|
101
|
-
|
|
102
|
-
function saveJSON(filePath, data) {
|
|
103
|
-
fs.writeFileSync(filePath, JSON.stringify(data, null, 2) + '\n');
|
|
104
|
-
const count = Array.isArray(data) ? data.length : Object.keys(data).length;
|
|
105
|
-
console.log(`✓ Saved ${count} entries to ${path.basename(filePath)}`);
|
|
106
|
-
}
|
|
107
|
-
|
|
108
|
-
// ─── Model filtering & mapping ──────────────────────────────────────────────
|
|
109
|
-
|
|
110
|
-
/**
|
|
111
|
-
* Filter: only serverless chat-capable LLM base models that are READY.
|
|
112
|
-
*
|
|
113
|
-
* We only include models that are serverless (pay-per-token) because those
|
|
114
|
-
* are the ones relevant for the pi provider. Non-serverless models can only
|
|
115
|
-
* be used via on-demand deployments, which isn't what this provider targets.
|
|
116
|
-
*
|
|
117
|
-
* Exceptions: models present in the existing models.json are kept even if
|
|
118
|
-
* they lose serverless status (they may still work via routers/firepass).
|
|
119
|
-
*/
|
|
120
|
-
function isRelevantModel(m, existingIds = new Set()) {
|
|
121
|
-
const kind = m.kind || '';
|
|
122
|
-
// Only HuggingFace base models
|
|
123
|
-
if (kind !== 'HF_BASE_MODEL') return false;
|
|
124
|
-
// Must be READY
|
|
125
|
-
if (m.state !== 'READY') return false;
|
|
126
|
-
// Must have a context length
|
|
127
|
-
if (!m.contextLength || m.contextLength === 0) return false;
|
|
128
|
-
// Must be serverless, OR already exist in our curated list
|
|
129
|
-
if (!m.supportsServerless && !existingIds.has(m.name)) return false;
|
|
130
|
-
return true;
|
|
131
|
-
}
|
|
132
|
-
|
|
133
|
-
/**
|
|
134
|
-
* Build a display name from the API displayName, falling back to the model id.
|
|
135
|
-
*/
|
|
136
|
-
function buildDisplayName(m) {
|
|
137
|
-
let name = m.displayName || m.name || '';
|
|
138
|
-
if (!name || name === m.name) {
|
|
139
|
-
name = m.name.split('/').pop() || m.name;
|
|
140
|
-
}
|
|
141
|
-
return name;
|
|
142
|
-
}
|
|
143
|
-
|
|
144
|
-
/**
|
|
145
|
-
* Convert a Fireworks API model to Pi-native models.json format.
|
|
146
|
-
* Only includes data the API provides — no pricing, reasoning, or output limits.
|
|
147
|
-
* Those come from patch.json.
|
|
148
|
-
*/
|
|
149
|
-
function convertModel(apiModel) {
|
|
150
|
-
const id = apiModel.name;
|
|
151
|
-
const name = buildDisplayName(apiModel);
|
|
152
|
-
const input = ['text'];
|
|
153
|
-
if (apiModel.supportsImageInput) input.push('image');
|
|
154
|
-
|
|
155
|
-
return {
|
|
156
|
-
id,
|
|
157
|
-
name,
|
|
158
|
-
reasoning: false,
|
|
159
|
-
input,
|
|
160
|
-
cost: {
|
|
161
|
-
input: 0,
|
|
162
|
-
output: 0,
|
|
163
|
-
cacheRead: 0,
|
|
164
|
-
cacheWrite: 0,
|
|
165
|
-
},
|
|
166
|
-
contextWindow: apiModel.contextLength || 0,
|
|
167
|
-
maxTokens: 0,
|
|
168
|
-
};
|
|
169
|
-
}
|
|
170
|
-
|
|
171
|
-
/**
|
|
172
|
-
* Deep merge a patch into a model. Nested objects (cost) are merged
|
|
173
|
-
* field-by-field; scalar fields are replaced.
|
|
174
|
-
*/
|
|
175
|
-
function applyPatch(model, patch) {
|
|
176
|
-
const result = { ...model };
|
|
177
|
-
|
|
178
|
-
if (patch.name !== undefined) result.name = patch.name;
|
|
179
|
-
if (patch.family !== undefined) result.family = patch.family;
|
|
180
|
-
if (patch.reasoning !== undefined) result.reasoning = patch.reasoning;
|
|
181
|
-
if (patch.interleaved !== undefined) result.interleaved = patch.interleaved;
|
|
182
|
-
|
|
183
|
-
if (patch.input !== undefined) result.input = patch.input;
|
|
184
|
-
if (patch.contextWindow !== undefined) result.contextWindow = patch.contextWindow;
|
|
185
|
-
if (patch.maxTokens !== undefined) result.maxTokens = patch.maxTokens;
|
|
186
|
-
|
|
187
|
-
if (patch.cost) {
|
|
188
|
-
result.cost = {
|
|
189
|
-
input: patch.cost.input ?? result.cost?.input ?? 0,
|
|
190
|
-
output: patch.cost.output ?? result.cost?.output ?? 0,
|
|
191
|
-
cacheRead: patch.cost.cacheRead ?? result.cost?.cacheRead ?? 0,
|
|
192
|
-
cacheWrite: patch.cost.cacheWrite ?? result.cost?.cacheWrite ?? 0,
|
|
193
|
-
};
|
|
194
|
-
}
|
|
195
|
-
|
|
196
|
-
return result;
|
|
197
|
-
}
|
|
198
|
-
|
|
199
|
-
// ─── README generation ──────────────────────────────────────────────────────
|
|
200
|
-
|
|
201
|
-
function formatCost(cost) {
|
|
202
|
-
if (cost === null || cost === undefined) return '-';
|
|
203
|
-
if (cost === 0) return 'Free';
|
|
204
|
-
return `$${cost.toFixed(2)}`;
|
|
205
|
-
}
|
|
206
|
-
|
|
207
|
-
function formatNumber(num) {
|
|
208
|
-
if (num === null || num === undefined) return '-';
|
|
209
|
-
if (num >= 1000000) return `${(num / 1000000).toFixed(1)}M`;
|
|
210
|
-
if (num >= 1000) return `${(num / 1000).toFixed(0)}K`;
|
|
211
|
-
return num.toString();
|
|
212
|
-
}
|
|
213
|
-
|
|
214
|
-
function getInputTypes(inputTypes) {
|
|
215
|
-
const types = inputTypes || ['text'];
|
|
216
|
-
const hasImage = types.includes('image');
|
|
217
|
-
const hasText = types.includes('text');
|
|
218
|
-
if (hasImage && hasText) return 'Text + Image';
|
|
219
|
-
if (hasImage) return 'Image';
|
|
220
|
-
return 'Text';
|
|
221
|
-
}
|
|
222
|
-
|
|
223
|
-
function generateReadmeRow(model) {
|
|
224
|
-
const cost = model.cost || {};
|
|
225
|
-
return `| ${model.name} | ${getInputTypes(model.input)} | ${formatNumber(model.contextWindow)} | ${formatNumber(model.maxTokens)} | ${formatCost(cost.input)} | ${formatCost(cost.output)} |`;
|
|
226
|
-
}
|
|
227
|
-
|
|
228
|
-
function updateReadme(models) {
|
|
229
|
-
const readmePath = path.join(process.cwd(), 'README.md');
|
|
230
|
-
let readme = fs.readFileSync(readmePath, 'utf8');
|
|
231
|
-
|
|
232
|
-
const sortedModels = [...models].sort((a, b) => {
|
|
233
|
-
const familyA = a.family || '';
|
|
234
|
-
const familyB = b.family || '';
|
|
235
|
-
if (familyA !== familyB) return familyA.localeCompare(familyB);
|
|
236
|
-
return a.name.localeCompare(b.name);
|
|
237
|
-
});
|
|
238
|
-
|
|
239
|
-
const tableRows = sortedModels.map(generateReadmeRow).join('\n');
|
|
240
|
-
const newTable = `| Model | Type | Context | Max Tokens | Input Cost | Output Cost |
|
|
241
|
-
|-------|------|---------|------------|------------|-------------|
|
|
242
|
-
${tableRows}`;
|
|
243
|
-
|
|
244
|
-
const tableRegex = /\| Model \| Type \| Context \| Max Tokens \| Input Cost \| Output Cost \|[\s\S]*?(?=\n\*Costs are per million)/;
|
|
245
|
-
readme = readme.replace(tableRegex, newTable);
|
|
246
|
-
|
|
247
|
-
readme = readme.replace(/\*\*\d+\+ AI Models\*\*/, `**${models.length}+ AI Models**`);
|
|
248
|
-
|
|
249
|
-
fs.writeFileSync(readmePath, readme);
|
|
250
|
-
console.log(`✓ Updated README.md with ${models.length} models`);
|
|
251
|
-
}
|
|
252
|
-
|
|
253
|
-
// ─── Main ────────────────────────────────────────────────────────────────────
|
|
254
|
-
|
|
255
|
-
async function main() {
|
|
256
|
-
const apiKey = process.env.FIREWORKS_API_KEY;
|
|
257
|
-
if (!apiKey) {
|
|
258
|
-
console.error('Error: FIREWORKS_API_KEY environment variable is required');
|
|
259
|
-
console.error('Usage: FIREWORKS_API_KEY=your-key node scripts/update-models.js');
|
|
260
|
-
process.exit(1);
|
|
261
|
-
}
|
|
262
|
-
|
|
263
|
-
console.log('Fetching models from Fireworks API...\n');
|
|
264
|
-
|
|
265
|
-
try {
|
|
266
|
-
// 1. Fetch all models from Fireworks API
|
|
267
|
-
const apiModels = await fetchAllFireworksModels(apiKey);
|
|
268
|
-
console.log(`\nTotal models from API: ${apiModels.length}`);
|
|
269
|
-
|
|
270
|
-
// 2. Load existing models.json and patch.json
|
|
271
|
-
const existingModels = Array.isArray(loadJSON(MODELS_PATH)) ? loadJSON(MODELS_PATH) : [];
|
|
272
|
-
const patchData = loadJSON(PATCH_PATH);
|
|
273
|
-
const existingIds = new Set(existingModels.map((m) => m.id));
|
|
274
|
-
|
|
275
|
-
// 3. Filter to relevant LLMs (serverless + previously curated)
|
|
276
|
-
const relevantApiModels = apiModels.filter((m) => isRelevantModel(m, existingIds));
|
|
277
|
-
console.log(`Relevant LLM models: ${relevantApiModels.length}`);
|
|
278
|
-
|
|
279
|
-
// 4. Convert API models to models.json format (no pricing — that comes from patch.json)
|
|
280
|
-
const newModels = relevantApiModels.map((apiModel) => convertModel(apiModel));
|
|
281
|
-
|
|
282
|
-
// Log new models (not in patch.json)
|
|
283
|
-
for (const m of newModels) {
|
|
284
|
-
if (!patchData[m.id]) {
|
|
285
|
-
console.log(` 🆕 New model: ${m.id} (${m.name}) — add to patch.json for pricing/output limits`);
|
|
286
|
-
}
|
|
287
|
-
}
|
|
288
|
-
|
|
289
|
-
// Live API is authoritative — models absent from API are removed
|
|
290
|
-
const allUpstreamModels = [...newModels];
|
|
291
|
-
|
|
292
|
-
// 5. Save upstream models (API-derived, no pricing)
|
|
293
|
-
saveJSON(MODELS_PATH, allUpstreamModels);
|
|
294
|
-
|
|
295
|
-
// 6. Load and process custom models
|
|
296
|
-
const customModels = Array.isArray(loadJSON(CUSTOM_MODELS_PATH)) ? loadJSON(CUSTOM_MODELS_PATH) : [];
|
|
297
|
-
|
|
298
|
-
// Find custom models that now appear in upstream (remove from custom)
|
|
299
|
-
const upstreamIds = new Set(allUpstreamModels.map((m) => m.id));
|
|
300
|
-
const duplicates = customModels.filter((m) => upstreamIds.has(m.id));
|
|
301
|
-
if (duplicates.length > 0) {
|
|
302
|
-
console.log(`\nFound ${duplicates.length} custom model(s) now available upstream:`);
|
|
303
|
-
for (const dup of duplicates) {
|
|
304
|
-
console.log(` - ${dup.id} (${dup.name})`);
|
|
305
|
-
}
|
|
306
|
-
const cleaned = customModels.filter((m) => !upstreamIds.has(m.id));
|
|
307
|
-
saveJSON(CUSTOM_MODELS_PATH, cleaned);
|
|
308
|
-
console.log(`✓ Removed ${duplicates.length} duplicate(s) from custom-models.json`);
|
|
309
|
-
customModels.length = 0;
|
|
310
|
-
customModels.push(...cleaned);
|
|
311
|
-
}
|
|
312
|
-
|
|
313
|
-
// 7. Build merged models with patches applied (for README)
|
|
314
|
-
const mergedMap = new Map();
|
|
315
|
-
|
|
316
|
-
// Start with upstream models
|
|
317
|
-
for (const m of allUpstreamModels) mergedMap.set(m.id, m);
|
|
318
|
-
|
|
319
|
-
// Apply patches (enrichment: pricing, reasoning, limits, etc.)
|
|
320
|
-
for (const [id, patch] of Object.entries(patchData)) {
|
|
321
|
-
const existing = mergedMap.get(id);
|
|
322
|
-
if (existing) {
|
|
323
|
-
mergedMap.set(id, applyPatch(existing, patch));
|
|
324
|
-
}
|
|
325
|
-
}
|
|
326
|
-
|
|
327
|
-
// Add/override with custom models, also applying their patches
|
|
328
|
-
for (const m of customModels) {
|
|
329
|
-
const patch = patchData[m.id];
|
|
330
|
-
mergedMap.set(m.id, patch ? applyPatch(m, patch) : m);
|
|
331
|
-
}
|
|
332
|
-
|
|
333
|
-
const allModels = Array.from(mergedMap.values());
|
|
334
|
-
|
|
335
|
-
console.log(
|
|
336
|
-
`\nTotal: ${allModels.length} models (${allUpstreamModels.length} upstream + ${customModels.length} custom, ${Object.keys(patchData).length} patches)`
|
|
337
|
-
);
|
|
338
|
-
|
|
339
|
-
// 8. Update README
|
|
340
|
-
updateReadme(allModels);
|
|
341
|
-
|
|
342
|
-
console.log('\nDone!');
|
|
343
|
-
} catch (error) {
|
|
344
|
-
console.error('Error:', error.message);
|
|
345
|
-
process.exit(1);
|
|
346
|
-
}
|
|
347
|
-
}
|
|
348
|
-
|
|
349
|
-
main();
|