pi-fireworks-provider 1.1.0 → 1.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -15,9 +15,11 @@ _Kimi, MiniMax, GLM, DeepSeek, GPT-OSS — via Fireworks AI's Anthropic Messages
15
15
 
16
16
  ## Features
17
17
 
18
- - **35+ AI Models** including Kimi K2.5, MiniMax M2.5, GLM 4.5/4.7/5, DeepSeek V3.1/V3.2, DeepSeek V4 Flash, and GPT-OSS
18
+ - **39+ AI Models** including Kimi K2.5, MiniMax M2.5, GLM 4.5/4.7/5, DeepSeek V3.1/V3.2, DeepSeek V4 Flash, and GPT-OSS
19
19
  - **Dual API support** via Fireworks AI's Anthropic Messages and OpenAI-compatible completions endpoints (per-model routing, matching pi core's Fireworks provider)
20
20
  - **Service tiers** — toggle Fireworks `priority` vs `standard` per request on supported models (with priority pricing reflected in cost tracking), via a keybinding, `/fireworks-tier`, and a footer status area
21
+ - **Preserved thinking** — toggle Fireworks' `reasoning_history: "preserved"` so prior assistant reasoning is retained across turns (better multi-turn recall; uses more tokens), via the `/fireworks-settings` panel, with a model-select notification. Matches neuralwatt/makora's settings-only UX, adapted to Fireworks' single global `reasoning_history` knob
22
+ - **Settings panel** — `/fireworks-settings` (TUI) to configure preserved thinking, service tier, and display preferences; persisted to `~/.pi/agent/extensions/fireworks.json`
21
23
  - **Cost Tracking** with per-model pricing for budget management
22
24
  - **Reasoning Models** support for advanced reasoning capabilities
23
25
  - **Vision Support** for image-capable models
@@ -71,8 +73,8 @@ pi
71
73
  |-------|------|---------|------------|------------|-------------|
72
74
  | DeepSeek V3.1 | Text | 164K | 164K | $0.56 | $1.68 |
73
75
  | DeepSeek V3.2 | Text | 164K | 160K | $0.56 | $1.68 |
74
- | DeepSeek V4 Flash | Text | 1.0M | 1.0M | $0.14 | $0.28 |
75
- | DeepSeek V4 Pro | Text | 1.0M | 1.0M | Free | Free |
76
+ | DeepSeek V4 Flash | Text | 1.0M | 384K | $0.14 | $0.28 |
77
+ | DeepSeek V4 Pro | Text | 1.0M | 384K | $1.74 | $3.48 |
76
78
  | DeepSeek V4 Pro (router) | Text | 1.0M | 1.0M | $1.74 | $3.48 |
77
79
  | Gemma 4 26B A4B IT | Text + Image | 262K | 0 | Free | Free |
78
80
  | Gemma 4 31B IT | Text + Image | 262K | 0 | Free | Free |
@@ -80,27 +82,31 @@ pi
80
82
  | GLM 4.5 Air | Text | 131K | 131K | $0.22 | $0.88 |
81
83
  | GLM 4.7 | Text | 203K | 198K | $0.60 | $2.20 |
82
84
  | GLM 5 | Text | 203K | 131K | $1.00 | $3.20 |
83
- | GLM 5 Fast (router) | Text | 203K | 131K | $1.00 | $3.20 |
85
+ | GLM 5 Fast | Text | 203K | 131K | $1.00 | $3.20 |
84
86
  | GLM 5.1 | Text | 203K | 131K | $1.40 | $4.40 |
85
- | GLM 5.1 Fast (router) | Text | 203K | 131K | $1.40 | $4.40 |
86
- | GLM 5.2 | Text | 1.0M | 0 | Free | Free |
87
+ | GLM 5.1 Fast | Text | 203K | 131K | $2.80 | $8.80 |
88
+ | GLM 5.2 | Text | 1.0M | 131K | $1.40 | $4.40 |
89
+ | GLM 5.2 Fast | Text | 1.0M | 131K | $2.10 | $6.60 |
87
90
  | GPT OSS 120B | Text | 131K | 33K | $0.15 | $0.60 |
88
- | GPT OSS 20B | Text | 131K | 33K | $0.05 | $0.20 |
91
+ | GPT OSS 20B | Text | 131K | 33K | $0.07 | $0.30 |
89
92
  | Kimi K2 Instruct | Text | 131K | 16K | $1.00 | $3.00 |
90
93
  | Kimi K2 Thinking | Text | 262K | 256K | $0.60 | $2.50 |
91
94
  | Kimi K2.5 | Text + Image | 262K | 256K | $0.60 | $3.00 |
92
95
  | Kimi K2.5 Fast (router) | Text + Image | 262K | 256K | $0.60 | $3.00 |
93
96
  | Kimi K2.6 | Text + Image | 262K | 262K | $0.95 | $4.00 |
94
97
  | Kimi K2.6 (router) | Text + Image | 262K | 262K | $0.95 | $4.00 |
95
- | Kimi K2.6 Turbo (router) | Text + Image | 262K | 262K | $0.95 | $4.00 |
96
- | Kimi K2.7 Code | Text + Image | 262K | 0 | Free | Free |
98
+ | Kimi K2.6 Fast | Text + Image | 262K | 262K | $2.00 | $8.00 |
99
+ | Kimi K2.6 Turbo | Text + Image | 262K | 262K | $2.00 | $8.00 |
100
+ | Kimi K2.7 Code | Text + Image | 262K | 262K | $0.95 | $4.00 |
101
+ | Kimi K2.7 Code Fast | Text + Image | 262K | 262K | $1.90 | $8.00 |
97
102
  | Llama 3.3 70B Instruct | Text | 131K | 0 | Free | Free |
98
103
  | MiniMax M2.7 (router) | Text | 204K | 0 | $0.30 | $1.20 |
99
- | Minimax M3 | Text + Image | 512K | 0 | Free | Free |
100
104
  | MiniMax-M2.1 | Text | 197K | 200K | $0.30 | $1.20 |
101
105
  | MiniMax-M2.5 | Text | 197K | 197K | $0.30 | $1.20 |
102
106
  | MiniMax-M2.7 | Text | 197K | 197K | $0.30 | $1.20 |
107
+ | MiniMax-M3 | Text | 512K | 512K | $0.30 | $1.20 |
103
108
  | NVIDIA Nemotron 3 Ultra NVFP4 | Text | 262K | 0 | Free | Free |
109
+ | Qwen 3.7 Plus | Text + Image | 262K | 66K | $0.40 | $1.60 |
104
110
  | Qwen3 8B | Text | 41K | 0 | Free | Free |
105
111
  | Qwen3 VL 30B A3B Instruct | Text + Image | 262K | 0 | Free | Free |
106
112
  | Qwen3 VL 30B A3B Thinking | Text + Image | 262K | 0 | Free | Free |
@@ -140,16 +146,35 @@ The selection is persisted per session (survives `/reload` and resume). When `pr
140
146
  "default": "standard",
141
147
  "keybinding": "ctrl+shift+l",
142
148
  "display": "statusbar"
149
+ },
150
+ "preserveThinking": {
151
+ "default": false
143
152
  }
144
153
  }
145
154
  ```
146
155
 
147
- - `default` — tier used until you toggle (`standard` | `priority`).
148
- - `keybinding` — any [pi key format](https://github.com/earendil-works/pi-coding-agent/blob/main/docs/keybindings.md) (e.g. `ctrl+shift+l`, `ctrl+shift+k`). Requires `/reload` after changing. On macOS browser terminals (localterm), avoid `alt`/`ctrl+alt` (Option produces special chars) and `ctrl+shift+t/w/n/c/v` (browser/localterm tab + copy/paste shortcuts).
149
- - `display` — `statusbar` (footer status area) or `off` (hide the tier indicator).
156
+ - `serviceTier.default` — tier used until you toggle (`standard` | `priority`).
157
+ - `serviceTier.keybinding` — any [pi key format](https://github.com/earendil-works/pi-coding-agent/blob/main/docs/keybindings.md) (e.g. `ctrl+shift+l`, `ctrl+shift+k`). Requires `/reload` after changing. On macOS browser terminals (localterm), avoid `alt`/`ctrl+alt` (Option produces special chars) and `ctrl+shift+t/w/n/c/v` (browser/localterm tab + copy/paste shortcuts).
158
+ - `serviceTier.display` — `statusbar` (footer status area) or `off` (hide the tier indicator).
159
+ - `preserveThinking.default` — whether `reasoning_history: "preserved"` is injected (`true` | `false`, default `false`). Also settable via `/fireworks-settings`.
150
160
 
151
161
  > **Note:** The OpenAI completions endpoint accepts `service_tier` directly (per Fireworks' API). The Anthropic Messages endpoint passes the top-level field through as an extra. If a supported Anthropic-routed model rejects it, file an issue so we can gate injection by API.
152
162
 
163
+ ## Preserved Thinking
164
+
165
+ Fireworks exposes a top-level `reasoning_history` request parameter. The only accepted value is `"preserved"`; omitting it (the default) means prior assistant reasoning is **stripped** from the model's context each turn. Setting `reasoning_history: "preserved"` makes Fireworks render prior assistant reasoning into the model's context, improving multi-turn recall at the cost of extra tokens. See [the Fireworks reasoning guide](https://docs.fireworks.ai/guides/reasoning#preserved-thinking).
166
+
167
+ This works on **both** transports — the OpenAI completions endpoint (assistant `reasoning_content` field) and the Anthropic Messages endpoint (assistant `thinking` content blocks, for which Fireworks returns a `signature` so pi-ai replays them). pi-ai already replays the reasoning field/block on prior assistant turns; this extension's only job is injecting the top-level `reasoning_history: "preserved"` flag that makes Fireworks honor it.
168
+
169
+ Unlike neuralwatt/makora (which use per-model vLLM `chat_template_kwargs` flags like `preserve_thinking`/`clear_thinking`), Fireworks' knob is a single global parameter that applies to every reasoning model, so we expose it as one on/off toggle rather than a per-model submenu.
170
+
171
+ **Toggle it:**
172
+
173
+ - **`/fireworks-settings`** (TUI) → *Preserved thinking* → `on` / `off`. Takes effect immediately and persists as the default for future sessions.
174
+ - A dim **model-select notification** tells you the current state when you switch to a Fireworks reasoning model (`Preserved thinking ON for …` / `… OFF for …`).
175
+
176
+ Preserved thinking is **off by default** to match pi core and Fireworks' default (stripped). There is intentionally no `/fireworks-preserve` command or keybinding — it's settings-panel-only, mirroring neuralwatt/makora.
177
+
153
178
  ## Usage
154
179
 
155
180
  After loading the extension, use the `/model` command in pi to select your preferred model:
@@ -190,6 +215,18 @@ Add to your pi configuration for automatic loading:
190
215
  }
191
216
  ```
192
217
 
218
+ ## Development
219
+
220
+ ```bash
221
+ pnpm install # install dev tooling (vitest, knip, typescript)
222
+ pnpm test # run the test suite (vitest)
223
+ pnpm run test:watch # watch mode
224
+ pnpm run lint:dead # dead-code / unused-export scan (knip)
225
+ pnpm run check # typecheck (tsc) + tests + knip, all green or exit non-zero
226
+ ```
227
+
228
+ Tests live in `tests/` and stub the `@earendil-works/pi-coding-agent` / `@earendil-works/pi-tui` peer dependencies (see `tests/__mocks__/`) so they run without the real pi packages installed. A per-run temp dir is used for `~/.pi/agent` (via `PI_CODING_AGENT_DIR` in `tests/vitest.setup.ts`) so config/cache reads and writes never touch your real environment.
229
+
193
230
  ## License
194
231
 
195
232
  MIT
package/index.ts CHANGED
@@ -366,14 +366,21 @@ const PRIORITY_PRICING: Record<string, { input: number; output: number; cacheRea
366
366
 
367
367
  type ServiceTier = "standard" | "priority";
368
368
 
369
+ type PreserveMode = boolean; // true = inject reasoning_history:"preserved"; false = default (stripped)
370
+
369
371
  interface ServiceTierConfig {
370
372
  default: ServiceTier;
371
373
  keybinding: string;
372
374
  display: "statusbar" | "off";
373
375
  }
374
376
 
377
+ interface PreserveThinkingConfig {
378
+ default: PreserveMode;
379
+ }
380
+
375
381
  interface FireworksConfig {
376
382
  serviceTier: ServiceTierConfig;
383
+ preserveThinking: PreserveThinkingConfig;
377
384
  }
378
385
 
379
386
  const FIREWORKS_CONFIG_PATH = path.join(getAgentDir(), "extensions", "fireworks.json");
@@ -384,22 +391,40 @@ const DEFAULT_SERVICE_TIER_CONFIG: ServiceTierConfig = {
384
391
  keybinding: "ctrl+shift+l",
385
392
  display: "statusbar",
386
393
  };
387
- const DEFAULT_FIREWORKS_CONFIG: FireworksConfig = { serviceTier: DEFAULT_SERVICE_TIER_CONFIG };
394
+ // Preserved thinking is OFF by default to match pi core and Fireworks' default
395
+ // (`reasoning_history` omitted = prior reasoning stripped). Opt in via the
396
+ // /fireworks-settings panel (mirrors neuralwatt/makora: settings-only, no
397
+ // command or keybinding).
398
+ const DEFAULT_PRESERVE_CONFIG: PreserveThinkingConfig = {
399
+ default: false,
400
+ };
401
+ const DEFAULT_FIREWORKS_CONFIG: FireworksConfig = {
402
+ serviceTier: DEFAULT_SERVICE_TIER_CONFIG,
403
+ preserveThinking: DEFAULT_PRESERVE_CONFIG,
404
+ };
388
405
 
389
406
  function isValidTier(v: unknown): v is ServiceTier {
390
407
  return v === "standard" || v === "priority";
391
408
  }
392
409
 
410
+ function isValidKeybinding(v: unknown): v is string {
411
+ return typeof v === "string" && v.length > 0;
412
+ }
413
+
393
414
  function loadFireworksConfig(): FireworksConfig {
394
415
  try {
395
416
  const raw = JSON.parse(fs.readFileSync(FIREWORKS_CONFIG_PATH, "utf8"));
396
417
  const st = raw?.serviceTier ?? {};
418
+ const pt = raw?.preserveThinking ?? {};
397
419
  return {
398
420
  serviceTier: {
399
421
  default: isValidTier(st.default) ? st.default : DEFAULT_SERVICE_TIER_CONFIG.default,
400
- keybinding: typeof st.keybinding === "string" && st.keybinding.length > 0 ? st.keybinding : DEFAULT_SERVICE_TIER_CONFIG.keybinding,
422
+ keybinding: isValidKeybinding(st.keybinding) ? st.keybinding : DEFAULT_SERVICE_TIER_CONFIG.keybinding,
401
423
  display: st.display === "off" ? "off" : "statusbar",
402
424
  },
425
+ preserveThinking: {
426
+ default: typeof pt.default === "boolean" ? pt.default : DEFAULT_PRESERVE_CONFIG.default,
427
+ },
403
428
  };
404
429
  } catch {
405
430
  // Config missing or invalid — write defaults so the user can discover it.
@@ -409,7 +434,27 @@ function loadFireworksConfig(): FireworksConfig {
409
434
  } catch {
410
435
  // Write failure is non-fatal — defaults still work in memory.
411
436
  }
412
- return { serviceTier: { ...DEFAULT_SERVICE_TIER_CONFIG } };
437
+ return { serviceTier: { ...DEFAULT_SERVICE_TIER_CONFIG }, preserveThinking: { ...DEFAULT_PRESERVE_CONFIG } };
438
+ }
439
+ }
440
+
441
+ // Read-modify-write the raw config JSON without re-validating, so unrelated
442
+ // fields a user added survive a settings-UI write. `loadFireworksConfig()`
443
+ // (validated) is still called after writing to refresh the in-memory config.
444
+ function readRawFireworksConfig(): Record<string, any> {
445
+ try {
446
+ return JSON.parse(fs.readFileSync(FIREWORKS_CONFIG_PATH, "utf8"));
447
+ } catch {
448
+ return JSON.parse(JSON.stringify(DEFAULT_FIREWORKS_CONFIG));
449
+ }
450
+ }
451
+
452
+ function writeRawFireworksConfig(raw: Record<string, any>): void {
453
+ try {
454
+ fs.mkdirSync(path.dirname(FIREWORKS_CONFIG_PATH), { recursive: true });
455
+ fs.writeFileSync(FIREWORKS_CONFIG_PATH, JSON.stringify(raw, null, 2) + "\n");
456
+ } catch {
457
+ // Write failure is non-fatal — the in-memory refresh still applies.
413
458
  }
414
459
  }
415
460
 
@@ -508,14 +553,108 @@ function recomputePriorityCost(message: any): any {
508
553
  return { ...message, usage: { ...usage, cost } };
509
554
  }
510
555
 
556
+ // ─── Preserved Thinking (reasoning_history) ──────────────────────────────────
557
+
558
+ // Fireworks exposes a top-level `reasoning_history` request parameter. The
559
+ // only accepted value is `"preserved"`; omitting it (or any other value)
560
+ // means prior assistant reasoning is STRIPPED from the model's context
561
+ // (verified by e2e: reasoning_content on an assistant message yields 0%
562
+ // multi-turn recall without the flag, 100% with it). This works on BOTH
563
+ // transports — the OpenAI completions endpoint (assistant `reasoning_content`
564
+ // field) and the Anthropic Messages endpoint (assistant `thinking` content
565
+ // block, which Firebooks returns a `signature` for so pi-ai replays it).
566
+ //
567
+ // Unlike neuralwatt/makora (which use per-model vLLM `chat_template_kwargs`
568
+ // flags like preserve_thinking/clear_thinking), Fireworks' knob is a single
569
+ // global top-level param that applies to every reasoning model. We expose it
570
+ // as one on/off toggle. See https://docs.fireworks.ai/guides/reasoning#preserved-thinking.
571
+
572
+ // A Fireworks model is preserve-eligible if it's a reasoning model. We read
573
+ // `reasoning` off the registered model when available, but also accept a
574
+ // truthy `reasoning` flag on ctx.model (pi-ai attaches it).
575
+ function isPreserveEligible(model: any): boolean {
576
+ if (!model || model.provider !== "fireworks") return false;
577
+ return model.reasoning === true;
578
+ }
579
+
580
+ // Runtime state: whether preserved thinking is active. Initialized from the
581
+ // config-file default at session_start (mirrors neuralwatt/makora, which drive
582
+ // preserve state from the config file, not session entries) and updated by the
583
+ // /fireworks-settings panel, which also persists the new default.
584
+ let preserveOn: boolean = fireworksConfig.preserveThinking.default;
585
+
586
+ function setPreserve(on: boolean): void {
587
+ preserveOn = on;
588
+ }
589
+
511
590
  // ─── Extension Entry Point ────────────────────────────────────────────────────
512
591
 
592
+ export {
593
+ applyPatch,
594
+ buildModels,
595
+ isChatModel,
596
+ transformApiModel,
597
+ mergeWithEmbedded,
598
+ loadStaleModels,
599
+ isFireworksKimiModel,
600
+ sanitizePattern,
601
+ sanitizeSchemaForKimi,
602
+ stripAnchorBleedInPlace,
603
+ isValidTier,
604
+ isValidKeybinding,
605
+ loadFireworksConfig,
606
+ readRawFireworksConfig,
607
+ writeRawFireworksConfig,
608
+ isPriorityApplicable,
609
+ PRIORITY_PRICING,
610
+ recomputePriorityCost,
611
+ replayTierState,
612
+ setTier,
613
+ updateTierStatus,
614
+ isPreserveEligible,
615
+ setPreserve,
616
+ };
617
+
618
+ export type {
619
+ JsonModel,
620
+ PatchEntry,
621
+ PatchData,
622
+ FireworksApi,
623
+ ServiceTier,
624
+ ServiceTierConfig,
625
+ PreserveThinkingConfig,
626
+ FireworksConfig,
627
+ };
628
+
513
629
  export default function (pi: ExtensionAPI) {
514
630
  piRef = pi;
515
631
  const embeddedModels = modelsData as JsonModel[];
516
632
  const customModels = customModelsData as JsonModel[];
517
633
  const patches = patchData as PatchData;
518
634
 
635
+ // Deferred model_select notify timer — cleared on rapid re-switch and on
636
+ // session_shutdown so only the latest switch notifies. Mirrors neuralwatt's
637
+ // pattern so pi core's (and other extensions') notifications land first.
638
+ let modelSelectNotifyTimer: ReturnType<typeof setTimeout> | null = null;
639
+ const MODEL_SELECT_NOTIFY_DELAY_MS = 250;
640
+
641
+ // Notify preserved-thinking state for a Fireworks reasoning model. Deferred
642
+ // so pi core's notifications land first; cancelled on re-switch/shutdown so
643
+ // only the latest shows. Always level "info" — the text conveys the
644
+ // preserve/strip tradeoff, not a warning.
645
+ function notifyPreserveOnSelect(model: any, ctx: any): void {
646
+ if (!model || model.provider !== "fireworks") return;
647
+ if (!isPreserveEligible(model)) return;
648
+ const msg = preserveOn
649
+ ? `Preserved thinking ON for ${model.name || model.id} — full reasoning history retained across turns (better multi-turn recall; uses more tokens). Open /fireworks-settings to change.`
650
+ : `Preserved thinking OFF for ${model.name || model.id} — reasoning stripped each turn (Fireworks default; lighter, weaker multi-turn recall). Open /fireworks-settings to change.`;
651
+ if (modelSelectNotifyTimer) clearTimeout(modelSelectNotifyTimer);
652
+ modelSelectNotifyTimer = setTimeout(() => {
653
+ modelSelectNotifyTimer = null;
654
+ try { ctx.ui.notify(msg, "info"); } catch { /* notify is a no-op without a UI runner */ }
655
+ }, MODEL_SELECT_NOTIFY_DELAY_MS);
656
+ }
657
+
519
658
  const staleBase = loadStaleModels(embeddedModels);
520
659
  const staleModels = buildModels(staleBase, customModels, patches);
521
660
 
@@ -532,7 +671,9 @@ export default function (pi: ExtensionAPI) {
532
671
  const signal = revalidateAbort.signal;
533
672
  fireworksConfig = loadFireworksConfig();
534
673
  replayTierState(ctx, fireworksConfig.serviceTier.default);
674
+ preserveOn = fireworksConfig.preserveThinking.default;
535
675
  updateTierStatus(ctx);
676
+ notifyPreserveOnSelect(ctx.model, ctx);
536
677
  resolveApiKey(ctx.modelRegistry).then(() => {
537
678
  revalidateModels(cachedApiKey, embeddedModels, signal).then((freshBase) => {
538
679
  if (freshBase && !signal.aborted) {
@@ -550,6 +691,7 @@ export default function (pi: ExtensionAPI) {
550
691
  pi.on("session_shutdown", (_event, ctx) => {
551
692
  revalidateAbort?.abort();
552
693
  try { ctx.ui.setStatus(TIER_STATUS_KEY, undefined); } catch {}
694
+ if (modelSelectNotifyTimer) { clearTimeout(modelSelectNotifyTimer); modelSelectNotifyTimer = null; }
553
695
  });
554
696
 
555
697
  // Sanitize JSON Schema patterns for Kimi models before sending to Fireworks.
@@ -576,6 +718,19 @@ export default function (pi: ExtensionAPI) {
576
718
  modified = true;
577
719
  }
578
720
 
721
+ // Preserved thinking: inject top-level `reasoning_history: "preserved"`
722
+ // so Fireworks renders prior assistant reasoning (reasoning_content on the
723
+ // OpenAI endpoint, thinking blocks on the Anthropic endpoint) into the
724
+ // model's context instead of stripping it. The only accepted value is
725
+ // "preserved"; omitted = stripped (Fireworks default / pi core). Applies to
726
+ // any Fireworks reasoning model on both transports. pi-ai already replays
727
+ // the reasoning field/block on prior assistant turns; this flag is what
728
+ // makes Fireworks honor it. See https://docs.fireworks.ai/guides/reasoning.
729
+ if (preserveOn && isPreserveEligible(model)) {
730
+ payload.reasoning_history = "preserved";
731
+ modified = true;
732
+ }
733
+
579
734
  // Kimi anchor-bleed sanitization (Kimi K2.x pattern bug). Only applies to
580
735
  // Kimi models, but a single request can be both Kimi and priority-tiered.
581
736
  if (isFireworksKimiModel(model)) {
@@ -623,9 +778,12 @@ export default function (pi: ExtensionAPI) {
623
778
  }
624
779
  });
625
780
 
626
- // Refresh the tier status area when the active model changes.
627
- pi.on("model_select", async (_event, ctx) => {
781
+ // Refresh the tier status area when the active model changes, and notify
782
+ // preserved-thinking state for the selected reasoning model (mirrors
783
+ // neuralwatt/makora: notification only, no preserve status area).
784
+ pi.on("model_select", async (event, ctx) => {
628
785
  updateTierStatus(ctx);
786
+ notifyPreserveOnSelect(event.model ?? ctx.model, ctx);
629
787
  });
630
788
 
631
789
  // Recompute finalized assistant-message cost against priority pricing so
@@ -672,4 +830,107 @@ export default function (pi: ExtensionAPI) {
672
830
  }
673
831
  },
674
832
  });
833
+
834
+ // /fireworks-settings: TUI settings panel (mirrors /neuralwatt-settings &
835
+ // /makora-settings). Opens a SettingsList via ctx.ui.custom(). Toggles write
836
+ // to ~/.pi/agent/extensions/fireworks.json (raw read-modify-write so unknown
837
+ // fields survive) and refresh the in-memory config. Preserved thinking is
838
+ // settings-only (no command/keybinding), exactly like the siblings; the
839
+ // service-tier keybinding is load-time only, so keybinding changes need /reload.
840
+ pi.registerCommand("fireworks-settings", {
841
+ description: "Configure Fireworks: preserved thinking + service tier + display",
842
+ async handler(_args, ctx) {
843
+ if (ctx.mode !== "tui") {
844
+ ctx.ui.notify("/fireworks-settings requires TUI mode.", "error");
845
+ return;
846
+ }
847
+ const { SettingsList, Container } = await import("@earendil-works/pi-tui");
848
+ const { getSettingsListTheme, DynamicBorder } = await import("@earendil-works/pi-coding-agent");
849
+
850
+ await ctx.ui.custom((_tui, theme, _kb, done) => {
851
+ const border = () => new DynamicBorder((s: string) => theme.fg("border", s));
852
+
853
+ const items: any[] = [
854
+ {
855
+ id: "preserveThinking",
856
+ label: "Preserved thinking",
857
+ description: "Inject reasoning_history:\"preserved\" so Fireworks retains prior assistant reasoning across turns (better multi-turn recall; uses more tokens). Off = Fireworks default (stripped). Applies to every Fireworks reasoning model on both endpoints.",
858
+ currentValue: preserveOn ? "on" : "off",
859
+ values: ["on", "off"],
860
+ },
861
+ {
862
+ id: "serviceTier",
863
+ label: "Service tier",
864
+ description: "Fireworks service tier for supported models. priority = higher throughput / latency at ~1.2\u20131.5\u00d7 cost (priority pricing reflected in cost tracking). standard = default.",
865
+ currentValue: currentTier,
866
+ values: ["standard", "priority"],
867
+ },
868
+ {
869
+ id: "serviceTier.display",
870
+ label: "Service tier display",
871
+ description: "Where the \u201ctier: standard / \u26a1priority\u201d indicator is shown: footer status area or hidden",
872
+ currentValue: fireworksConfig.serviceTier.display,
873
+ values: ["statusbar", "off"],
874
+ },
875
+ ];
876
+
877
+ const container = new Container();
878
+ container.addChild(border());
879
+
880
+ const settingsList = new SettingsList(
881
+ items,
882
+ Math.min(items.length + 2, 15),
883
+ getSettingsListTheme(),
884
+ (id: string, newValue: string) => {
885
+ if (id === "preserveThinking") {
886
+ const on = newValue === "on";
887
+ setPreserve(on);
888
+ // Persist as the default too, so it survives new sessions.
889
+ const raw = readRawFireworksConfig();
890
+ raw.preserveThinking = { ...(raw.preserveThinking ?? {}), default: on };
891
+ writeRawFireworksConfig(raw);
892
+ fireworksConfig = loadFireworksConfig();
893
+ ctx.ui.notify(`Preserved thinking ${on ? "on" : "off"} \u2014 takes effect now.`, "info");
894
+ } else if (id === "serviceTier") {
895
+ let model: any;
896
+ try { model = ctx.model; } catch { model = undefined; }
897
+ if (!model || model.provider !== "fireworks" || !isPriorityApplicable(model.id)) {
898
+ ctx.ui.notify("Service tier only applies to supported Fireworks models.", "info");
899
+ return;
900
+ }
901
+ const tier = newValue as ServiceTier;
902
+ setTier(ctx, tier);
903
+ const raw = readRawFireworksConfig();
904
+ raw.serviceTier = { ...(raw.serviceTier ?? {}), default: tier };
905
+ writeRawFireworksConfig(raw);
906
+ fireworksConfig = loadFireworksConfig();
907
+ ctx.ui.notify(`Fireworks service tier: ${tier}`, "info");
908
+ } else if (id === "serviceTier.display") {
909
+ const raw = readRawFireworksConfig();
910
+ raw.serviceTier = { ...(raw.serviceTier ?? {}), display: newValue };
911
+ writeRawFireworksConfig(raw);
912
+ fireworksConfig = loadFireworksConfig();
913
+ updateTierStatus(ctx);
914
+ }
915
+ },
916
+ () => done(undefined),
917
+ { enableSearch: true },
918
+ );
919
+ container.addChild(settingsList);
920
+ container.addChild(border());
921
+
922
+ return {
923
+ render(width: number) {
924
+ return container.render(width);
925
+ },
926
+ invalidate() {
927
+ container.invalidate();
928
+ },
929
+ handleInput(data: string) {
930
+ settingsList.handleInput?.(data);
931
+ },
932
+ };
933
+ });
934
+ },
935
+ });
675
936
  }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "pi-fireworks-provider",
3
- "version": "1.1.0",
3
+ "version": "1.2.0",
4
4
  "description": "Fireworks AI provider extension for pi - Access Kimi, MiniMax, GLM, DeepSeek, and GPT-OSS models through the Fireworks AI API",
5
5
  "type": "module",
6
6
  "main": "index.ts",
@@ -19,6 +19,20 @@
19
19
  ],
20
20
  "author": "",
21
21
  "license": "MIT",
22
+ "files": [
23
+ "index.ts",
24
+ "models.json",
25
+ "custom-models.json",
26
+ "patch.json",
27
+ "README.md",
28
+ "LICENSE"
29
+ ],
30
+ "devDependencies": {
31
+ "@types/node": "^22.20.0",
32
+ "knip": "6.14.1",
33
+ "typescript": "6.0.3",
34
+ "vitest": "4.1.7"
35
+ },
22
36
  "pi": {
23
37
  "extensions": [
24
38
  "./index.ts"
@@ -27,7 +41,10 @@
27
41
  "scripts": {
28
42
  "clean": "echo 'nothing to clean'",
29
43
  "build": "echo 'nothing to build'",
30
- "check": "echo 'nothing to check'",
44
+ "check": "tsc --noEmit && vitest run && knip --no-gitignore",
45
+ "lint:dead": "knip --no-gitignore",
46
+ "test": "vitest run",
47
+ "test:watch": "vitest",
31
48
  "update-models": "node scripts/update-models.js"
32
49
  }
33
50
  }
package/patch.json CHANGED
@@ -417,22 +417,6 @@
417
417
  "supportsLongCacheRetention": false
418
418
  }
419
419
  },
420
- "accounts/fireworks/models/qwen3p6-plus": {
421
- "name": "Qwen 3.6 Plus",
422
- "reasoning": true,
423
- "input": ["text", "image"],
424
- "cost": {
425
- "input": 0.5,
426
- "output": 3,
427
- "cacheRead": 0.1,
428
- "cacheWrite": 0
429
- },
430
- "contextWindow": 262144,
431
- "maxTokens": 8192,
432
- "compat": {
433
- "supportsReasoningEffort": true
434
- }
435
- },
436
420
  "accounts/fireworks/models/qwen3p7-plus": {
437
421
  "name": "Qwen 3.7 Plus",
438
422
  "api": "anthropic-messages",
@@ -544,20 +528,6 @@
544
528
  "supportsReasoningEffort": true
545
529
  }
546
530
  },
547
- "accounts/fireworks/routers/kimi-k2p5-turbo": {
548
- "name": "Kimi K2.5 Turbo (router)",
549
- "reasoning": true,
550
- "cost": {
551
- "input": 0.6,
552
- "output": 3,
553
- "cacheRead": 0.1,
554
- "cacheWrite": 0
555
- },
556
- "maxTokens": 256000,
557
- "compat": {
558
- "supportsReasoningEffort": true
559
- }
560
- },
561
531
  "accounts/fireworks/routers/kimi-k2p6": {
562
532
  "name": "Kimi K2.6 (router)",
563
533
  "reasoning": true,
@@ -1,4 +0,0 @@
1
- github: monotykamary
2
- ko_fi: monotykamary
3
- buy_me_a_coffee: monotykamary
4
- polar: monotykamary
@@ -1 +0,0 @@
1
- {"_meta":true,"v":1,"id":"memory","type":"named","createdAt":"2026-05-25T08:03:16.155Z","description":"Cross-session knowledge and insights"}
@@ -1 +0,0 @@
1
- 019e98a1-b474-7d6d-b80e-b1dedf22aa71
package/AGENTS.md DELETED
@@ -1,56 +0,0 @@
1
- # AGENTS.md
2
-
3
- ## DO NOT EDIT — Auto-generated Files
4
-
5
- The following files are **idempotent** and regenerated by `scripts/update-models.js`. Never edit them directly — your changes will be overwritten on the next model sync.
6
-
7
- | File | Why it's auto-generated |
8
- |------|------------------------|
9
- | `models.json` | Built from the provider API. `update-models.js` fetches models, preserves curated data for known IDs, and writes this file. |
10
- | `README.md` (model table) | The table under `## Available Models` is replaced in-place by `update-models.js` after merging base models → patch → custom models. |
11
-
12
- ## Correct Files to Edit
13
-
14
- When a model needs overrides, new properties, or corrections, edit the appropriate source file below. These are the **source of truth** that the update script reads but never writes.
15
-
16
- | File | Purpose |
17
- |------|---------|
18
- | `patch.json` | Per-model overrides keyed by model ID. Add reasoning flags, compat settings, pricing corrections, thinking level maps, etc. Applied on top of `models.json` at runtime and for README generation. |
19
- | `custom-models.json` | Models that don't exist in the provider API (hidden models, router endpoints, cross-provider aliases). Merged after patch. Format: array of full model objects (same schema as `models.json` entries). |
20
- | `index.ts` | Provider extension code. |
21
- | `scripts/update-models.js` | The sync script itself (edit only if changing how models are fetched/transformed). |
22
-
23
- ## Data Flow
24
-
25
- ```
26
- Provider API ──fetch──► models.json ──apply──► patch.json ──merge──► custom-models.json
27
- │ │ │
28
- └────────────────────────────┴──────────────────────┘
29
- │
30
- README model table
31
- ```
32
-
33
- 1. `models.json` — base data from the provider API (auto-generated, DO NOT EDIT)
34
- 2. `patch.json` — overrides applied on top (EDIT THIS for corrections/enrichments)
35
- 3. `custom-models.json` — additional models not in the API (EDIT THIS for new models)
36
- 4. README table — rendered from the merged result of all three (auto-generated, DO NOT EDIT)
37
-
38
- ## Common Tasks
39
-
40
- ### Add a compat setting or override pricing for an existing model
41
- → Edit `patch.json`. Add an entry keyed by the model's `id`.
42
-
43
- ### Add a model not available in the provider API
44
- → Edit `custom-models.json`. Add a full model object to the array.
45
-
46
- ### Update models from the provider API
47
- → Run `node scripts/update-models.js` (may require an API key env var).
48
-
49
- ### Regenerate the README model table
50
- → Run `node scripts/update-models.js` — it updates both `models.json` and the README table.
51
-
52
- ## TL;DR
53
-
54
- - **Never edit `models.json`** — edit `patch.json` instead.
55
- - **Never edit the README model table** — run the update script instead.
56
- - `patch.json` and `custom-models.json` are the source files you should modify.
@@ -1,349 +0,0 @@
1
- #!/usr/bin/env node
2
-
3
- /**
4
- * Script to update fireworks models from the Fireworks API
5
- *
6
- * Uses the official Fireworks Gateway REST API to discover available models:
7
- * GET /v1/accounts/fireworks/models
8
- *
9
- * Requires FIREWORKS_API_KEY environment variable.
10
- * Usage: FIREWORKS_API_KEY=your-key node scripts/update-models.js
11
- *
12
- * Data flow:
13
- * models.json → auto-generated from Fireworks API (model discovery)
14
- * patch.json → manual overrides (pricing, reasoning, limits, etc.)
15
- * custom-models.json → hidden/router models not in the API
16
- *
17
- * The API provides: id, displayName, contextLength, supportsImageInput,
18
- * supportsTools, supportsServerless, state, kind, moe, parameterCount, etc.
19
- *
20
- * It does NOT provide: pricing, max output tokens, reasoning mode, or
21
- * interleaved thinking details. Those come from patch.json.
22
- *
23
- * Merge order for README: models.json → apply patch.json → merge custom-models.json
24
- */
25
-
26
- import https from 'https';
27
- import fs from 'fs';
28
- import path from 'path';
29
- import { fileURLToPath } from 'url';
30
-
31
- const __filename = fileURLToPath(import.meta.url);
32
- const __dirname = path.dirname(__filename);
33
-
34
- const FIREWORKS_API_BASE = 'https://api.fireworks.ai';
35
- const ACCOUNT_ID = 'fireworks';
36
- const MODELS_PATH = path.join(process.cwd(), 'models.json');
37
- const CUSTOM_MODELS_PATH = path.join(process.cwd(), 'custom-models.json');
38
- const PATCH_PATH = path.join(process.cwd(), 'patch.json');
39
-
40
- // ─── HTTP helpers ───────────────────────────────────────────────────────────
41
-
42
- function fetchJSON(url, headers = {}) {
43
- return new Promise((resolve, reject) => {
44
- const req = https.get(url, { headers }, (res) => {
45
- let data = '';
46
- res.on('data', (chunk) => (data += chunk));
47
- res.on('end', () => {
48
- try {
49
- resolve(JSON.parse(data));
50
- } catch (e) {
51
- reject(new Error(`Failed to parse JSON from ${url}: ${e.message}`));
52
- }
53
- });
54
- });
55
- req.on('error', reject);
56
- });
57
- }
58
-
59
- /**
60
- * Paginate through the Fireworks account models API.
61
- * Returns all models across all pages.
62
- */
63
- async function fetchAllFireworksModels(apiKey) {
64
- const headers = {};
65
- if (apiKey) headers.Authorization = `Bearer ${apiKey}`;
66
-
67
- const allModels = [];
68
- let pageToken = undefined;
69
- let page = 0;
70
-
71
- do {
72
- let url = `${FIREWORKS_API_BASE}/v1/accounts/${ACCOUNT_ID}/models?pageSize=200`;
73
- if (pageToken) url += `&pageToken=${pageToken}`;
74
-
75
- const data = await fetchJSON(url, headers);
76
- const models = data.models || [];
77
- allModels.push(...models);
78
-
79
- pageToken = data.nextPageToken || undefined;
80
- page++;
81
- console.log(` Page ${page}: fetched ${models.length} models (total so far: ${allModels.length})`);
82
- } while (pageToken);
83
-
84
- return allModels;
85
- }
86
-
87
- // ─── File I/O ───────────────────────────────────────────────────────────────
88
-
89
- function loadJSON(filePath) {
90
- try {
91
- if (!fs.existsSync(filePath)) return [];
92
- const data = fs.readFileSync(filePath, 'utf8');
93
- const parsed = JSON.parse(data);
94
- console.log(`✓ Loaded ${Array.isArray(parsed) ? parsed.length : Object.keys(parsed).length} entries from ${path.basename(filePath)}`);
95
- return parsed;
96
- } catch (e) {
97
- console.warn(`Warning: Could not load ${path.basename(filePath)}: ${e.message}`);
98
- return {};
99
- }
100
- }
101
-
102
- function saveJSON(filePath, data) {
103
- fs.writeFileSync(filePath, JSON.stringify(data, null, 2) + '\n');
104
- const count = Array.isArray(data) ? data.length : Object.keys(data).length;
105
- console.log(`✓ Saved ${count} entries to ${path.basename(filePath)}`);
106
- }
107
-
108
- // ─── Model filtering & mapping ──────────────────────────────────────────────
109
-
110
- /**
111
- * Filter: only serverless chat-capable LLM base models that are READY.
112
- *
113
- * We only include models that are serverless (pay-per-token) because those
114
- * are the ones relevant for the pi provider. Non-serverless models can only
115
- * be used via on-demand deployments, which isn't what this provider targets.
116
- *
117
- * Exceptions: models present in the existing models.json are kept even if
118
- * they lose serverless status (they may still work via routers/firepass).
119
- */
120
- function isRelevantModel(m, existingIds = new Set()) {
121
- const kind = m.kind || '';
122
- // Only HuggingFace base models
123
- if (kind !== 'HF_BASE_MODEL') return false;
124
- // Must be READY
125
- if (m.state !== 'READY') return false;
126
- // Must have a context length
127
- if (!m.contextLength || m.contextLength === 0) return false;
128
- // Must be serverless, OR already exist in our curated list
129
- if (!m.supportsServerless && !existingIds.has(m.name)) return false;
130
- return true;
131
- }
132
-
133
- /**
134
- * Build a display name from the API displayName, falling back to the model id.
135
- */
136
- function buildDisplayName(m) {
137
- let name = m.displayName || m.name || '';
138
- if (!name || name === m.name) {
139
- name = m.name.split('/').pop() || m.name;
140
- }
141
- return name;
142
- }
143
-
144
- /**
145
- * Convert a Fireworks API model to Pi-native models.json format.
146
- * Only includes data the API provides — no pricing, reasoning, or output limits.
147
- * Those come from patch.json.
148
- */
149
- function convertModel(apiModel) {
150
- const id = apiModel.name;
151
- const name = buildDisplayName(apiModel);
152
- const input = ['text'];
153
- if (apiModel.supportsImageInput) input.push('image');
154
-
155
- return {
156
- id,
157
- name,
158
- reasoning: false,
159
- input,
160
- cost: {
161
- input: 0,
162
- output: 0,
163
- cacheRead: 0,
164
- cacheWrite: 0,
165
- },
166
- contextWindow: apiModel.contextLength || 0,
167
- maxTokens: 0,
168
- };
169
- }
170
-
171
- /**
172
- * Deep merge a patch into a model. Nested objects (cost) are merged
173
- * field-by-field; scalar fields are replaced.
174
- */
175
- function applyPatch(model, patch) {
176
- const result = { ...model };
177
-
178
- if (patch.name !== undefined) result.name = patch.name;
179
- if (patch.family !== undefined) result.family = patch.family;
180
- if (patch.reasoning !== undefined) result.reasoning = patch.reasoning;
181
- if (patch.interleaved !== undefined) result.interleaved = patch.interleaved;
182
-
183
- if (patch.input !== undefined) result.input = patch.input;
184
- if (patch.contextWindow !== undefined) result.contextWindow = patch.contextWindow;
185
- if (patch.maxTokens !== undefined) result.maxTokens = patch.maxTokens;
186
-
187
- if (patch.cost) {
188
- result.cost = {
189
- input: patch.cost.input ?? result.cost?.input ?? 0,
190
- output: patch.cost.output ?? result.cost?.output ?? 0,
191
- cacheRead: patch.cost.cacheRead ?? result.cost?.cacheRead ?? 0,
192
- cacheWrite: patch.cost.cacheWrite ?? result.cost?.cacheWrite ?? 0,
193
- };
194
- }
195
-
196
- return result;
197
- }
198
-
199
- // ─── README generation ──────────────────────────────────────────────────────
200
-
201
- function formatCost(cost) {
202
- if (cost === null || cost === undefined) return '-';
203
- if (cost === 0) return 'Free';
204
- return `$${cost.toFixed(2)}`;
205
- }
206
-
207
- function formatNumber(num) {
208
- if (num === null || num === undefined) return '-';
209
- if (num >= 1000000) return `${(num / 1000000).toFixed(1)}M`;
210
- if (num >= 1000) return `${(num / 1000).toFixed(0)}K`;
211
- return num.toString();
212
- }
213
-
214
- function getInputTypes(inputTypes) {
215
- const types = inputTypes || ['text'];
216
- const hasImage = types.includes('image');
217
- const hasText = types.includes('text');
218
- if (hasImage && hasText) return 'Text + Image';
219
- if (hasImage) return 'Image';
220
- return 'Text';
221
- }
222
-
223
- function generateReadmeRow(model) {
224
- const cost = model.cost || {};
225
- return `| ${model.name} | ${getInputTypes(model.input)} | ${formatNumber(model.contextWindow)} | ${formatNumber(model.maxTokens)} | ${formatCost(cost.input)} | ${formatCost(cost.output)} |`;
226
- }
227
-
228
- function updateReadme(models) {
229
- const readmePath = path.join(process.cwd(), 'README.md');
230
- let readme = fs.readFileSync(readmePath, 'utf8');
231
-
232
- const sortedModels = [...models].sort((a, b) => {
233
- const familyA = a.family || '';
234
- const familyB = b.family || '';
235
- if (familyA !== familyB) return familyA.localeCompare(familyB);
236
- return a.name.localeCompare(b.name);
237
- });
238
-
239
- const tableRows = sortedModels.map(generateReadmeRow).join('\n');
240
- const newTable = `| Model | Type | Context | Max Tokens | Input Cost | Output Cost |
241
- |-------|------|---------|------------|------------|-------------|
242
- ${tableRows}`;
243
-
244
- const tableRegex = /\| Model \| Type \| Context \| Max Tokens \| Input Cost \| Output Cost \|[\s\S]*?(?=\n\*Costs are per million)/;
245
- readme = readme.replace(tableRegex, newTable);
246
-
247
- readme = readme.replace(/\*\*\d+\+ AI Models\*\*/, `**${models.length}+ AI Models**`);
248
-
249
- fs.writeFileSync(readmePath, readme);
250
- console.log(`✓ Updated README.md with ${models.length} models`);
251
- }
252
-
253
- // ─── Main ────────────────────────────────────────────────────────────────────
254
-
255
- async function main() {
256
- const apiKey = process.env.FIREWORKS_API_KEY;
257
- if (!apiKey) {
258
- console.error('Error: FIREWORKS_API_KEY environment variable is required');
259
- console.error('Usage: FIREWORKS_API_KEY=your-key node scripts/update-models.js');
260
- process.exit(1);
261
- }
262
-
263
- console.log('Fetching models from Fireworks API...\n');
264
-
265
- try {
266
- // 1. Fetch all models from Fireworks API
267
- const apiModels = await fetchAllFireworksModels(apiKey);
268
- console.log(`\nTotal models from API: ${apiModels.length}`);
269
-
270
- // 2. Load existing models.json and patch.json
271
- const existingModels = Array.isArray(loadJSON(MODELS_PATH)) ? loadJSON(MODELS_PATH) : [];
272
- const patchData = loadJSON(PATCH_PATH);
273
- const existingIds = new Set(existingModels.map((m) => m.id));
274
-
275
- // 3. Filter to relevant LLMs (serverless + previously curated)
276
- const relevantApiModels = apiModels.filter((m) => isRelevantModel(m, existingIds));
277
- console.log(`Relevant LLM models: ${relevantApiModels.length}`);
278
-
279
- // 4. Convert API models to models.json format (no pricing — that comes from patch.json)
280
- const newModels = relevantApiModels.map((apiModel) => convertModel(apiModel));
281
-
282
- // Log new models (not in patch.json)
283
- for (const m of newModels) {
284
- if (!patchData[m.id]) {
285
- console.log(` 🆕 New model: ${m.id} (${m.name}) — add to patch.json for pricing/output limits`);
286
- }
287
- }
288
-
289
- // Live API is authoritative — models absent from API are removed
290
- const allUpstreamModels = [...newModels];
291
-
292
- // 5. Save upstream models (API-derived, no pricing)
293
- saveJSON(MODELS_PATH, allUpstreamModels);
294
-
295
- // 6. Load and process custom models
296
- const customModels = Array.isArray(loadJSON(CUSTOM_MODELS_PATH)) ? loadJSON(CUSTOM_MODELS_PATH) : [];
297
-
298
- // Find custom models that now appear in upstream (remove from custom)
299
- const upstreamIds = new Set(allUpstreamModels.map((m) => m.id));
300
- const duplicates = customModels.filter((m) => upstreamIds.has(m.id));
301
- if (duplicates.length > 0) {
302
- console.log(`\nFound ${duplicates.length} custom model(s) now available upstream:`);
303
- for (const dup of duplicates) {
304
- console.log(` - ${dup.id} (${dup.name})`);
305
- }
306
- const cleaned = customModels.filter((m) => !upstreamIds.has(m.id));
307
- saveJSON(CUSTOM_MODELS_PATH, cleaned);
308
- console.log(`✓ Removed ${duplicates.length} duplicate(s) from custom-models.json`);
309
- customModels.length = 0;
310
- customModels.push(...cleaned);
311
- }
312
-
313
- // 7. Build merged models with patches applied (for README)
314
- const mergedMap = new Map();
315
-
316
- // Start with upstream models
317
- for (const m of allUpstreamModels) mergedMap.set(m.id, m);
318
-
319
- // Apply patches (enrichment: pricing, reasoning, limits, etc.)
320
- for (const [id, patch] of Object.entries(patchData)) {
321
- const existing = mergedMap.get(id);
322
- if (existing) {
323
- mergedMap.set(id, applyPatch(existing, patch));
324
- }
325
- }
326
-
327
- // Add/override with custom models, also applying their patches
328
- for (const m of customModels) {
329
- const patch = patchData[m.id];
330
- mergedMap.set(m.id, patch ? applyPatch(m, patch) : m);
331
- }
332
-
333
- const allModels = Array.from(mergedMap.values());
334
-
335
- console.log(
336
- `\nTotal: ${allModels.length} models (${allUpstreamModels.length} upstream + ${customModels.length} custom, ${Object.keys(patchData).length} patches)`
337
- );
338
-
339
- // 8. Update README
340
- updateReadme(allModels);
341
-
342
- console.log('\nDone!');
343
- } catch (error) {
344
- console.error('Error:', error.message);
345
- process.exit(1);
346
- }
347
- }
348
-
349
- main();