@aliou/pi-neuralwatt 0.15.2 → 0.15.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -144,5 +144,5 @@ This repository uses [Changesets](https://github.com/changesets/changesets) for
144
144
  ## Links
145
145
 
146
146
  - [Neuralwatt](https://portal.neuralwatt.com/auth/register?ref=NW-ALIOU-Q7MF)
147
- - [Neuralwatt API Docs](https://neuralwatt.com/docs)
147
+ - [Neuralwatt API Docs](https://docs.neuralwatt.com/quickstart.md)
148
148
  - [Pi Documentation](https://buildwithpi.ai/)
@@ -8,15 +8,6 @@ export type ThinkingLevelMap = NonNullable<
8
8
  ProviderModelConfig["thinkingLevelMap"]
9
9
  >;
10
10
 
11
- /**
12
- * Flex tier is billed at 65% of standard pricing (35% off) when the request
13
- * streams. A non-streaming request to a `-flex` model silently falls back to
14
- * the standard tier and the standard price.
15
- *
16
- * https://portal.neuralwatt.com/docs/guides/flex-tier
17
- */
18
- export const FLEX_COST_MULTIPLIER = 0.65;
19
-
20
11
  export interface NeuralwattCost {
21
12
  input: number;
22
13
  output: number;
@@ -107,12 +98,14 @@ export function buildThinkingLevelMap(
107
98
 
108
99
  /**
109
100
  * Neuralwatt reports `max_output_tokens: null` for models whose output is only
110
- * bounded by the context window. Mirror the API instead of inventing a cap.
101
+ * bounded by the context window. Some models incorrectly report 0; treat 0
102
+ * like null so we never emit maxTokens: 0.
111
103
  */
112
104
  export function resolveMaxTokens(
113
105
  maxOutputTokens: number | null | undefined,
114
106
  contextWindow: number,
115
107
  ): number {
108
+ if (maxOutputTokens === 0) return contextWindow;
116
109
  return maxOutputTokens ?? contextWindow;
117
110
  }
118
111
 
@@ -2,7 +2,6 @@ import type { ProviderModelConfig } from "@earendil-works/pi-coding-agent";
2
2
  import type { NeuralwattApiModel } from "../../../src/types/models-api";
3
3
  import {
4
4
  buildThinkingLevelMap,
5
- FLEX_COST_MULTIPLIER,
6
5
  resolveMaxTokens,
7
6
  type ThinkingLevelMap,
8
7
  } from "./build";
@@ -10,12 +9,6 @@ import { NEURALWATT_MODELS } from "./public-models";
10
9
 
11
10
  export type NeuralwattModel = ProviderModelConfig;
12
11
 
13
- const CONTEXT_WINDOW_OVERRIDES: ReadonlyMap<string, number> = new Map([
14
- ["kimi-k3", 327_680],
15
- ["kimi-k3-fast", 327_680],
16
- ["kimi-k3-flex", 327_680],
17
- ]);
18
-
19
12
  // Chat-template thinking: the API exposes a `reasoning` block, but the
20
13
  // underlying mechanism is chat_template_kwargs, so Pi needs the mapping.
21
14
  const COMPAT_OVERRIDES: Partial<
@@ -30,16 +23,11 @@ const COMPAT_OVERRIDES: Partial<
30
23
  };
31
24
 
32
25
  const HARDCODED_ALIASES: Record<string, string> = {
33
- "zai-org/GLM-5.2-FP8": "glm-5.2",
34
26
  "moonshotai/Kimi-K2.7-Code": "kimi-k2.7-code",
35
27
  "Qwen/Qwen3.6-35B-A3B": "qwen3.6-35b",
36
28
  "deepseek-ai/DeepSeek-V4-Flash": "deepseek-v4-flash",
37
29
  };
38
30
 
39
- function isFlexModelId(id: string): boolean {
40
- return id.endsWith("-flex");
41
- }
42
-
43
31
  function isVariantId(id: string): boolean {
44
32
  return id.includes("-fast") || id.includes("-flex") || id.includes("-short");
45
33
  }
@@ -52,8 +40,6 @@ function apiModelToProviderModel(model: NeuralwattApiModel): NeuralwattModel {
52
40
  );
53
41
 
54
42
  const reasoning = meta.capabilities.reasoning;
55
- // Flex variants are billed at 0.65x when streaming (35% off).
56
- const multiplier = isFlexModelId(model.id) ? FLEX_COST_MULTIPLIER : 1;
57
43
 
58
44
  const compat: NonNullable<ProviderModelConfig["compat"]> = {
59
45
  supportsDeveloperRole: meta.capabilities.developer_role,
@@ -62,8 +48,7 @@ function apiModelToProviderModel(model: NeuralwattApiModel): NeuralwattModel {
62
48
  if (reasoning) compat.requiresReasoningContentOnAssistantMessages = true;
63
49
  Object.assign(compat, COMPAT_OVERRIDES[model.id]);
64
50
 
65
- const contextWindow =
66
- CONTEXT_WINDOW_OVERRIDES.get(model.id) ?? model.max_model_len;
51
+ const contextWindow = model.max_model_len;
67
52
 
68
53
  const result: NeuralwattModel = {
69
54
  id: model.id,
@@ -73,10 +58,10 @@ function apiModelToProviderModel(model: NeuralwattApiModel): NeuralwattModel {
73
58
  ? (["text", "image"] as const)
74
59
  : (["text"] as const),
75
60
  cost: {
76
- input: meta.pricing.input_per_million * multiplier,
77
- output: meta.pricing.output_per_million * multiplier,
78
- cacheRead: (meta.pricing.cached_input_per_million ?? 0) * multiplier,
79
- cacheWrite: (meta.pricing.cached_output_per_million ?? 0) * multiplier,
61
+ input: meta.pricing.input_per_million,
62
+ output: meta.pricing.output_per_million,
63
+ cacheRead: meta.pricing.cached_input_per_million ?? 0,
64
+ cacheWrite: meta.pricing.cached_output_per_million ?? 0,
80
65
  },
81
66
  contextWindow,
82
67
  maxTokens: resolveMaxTokens(meta.limits.max_output_tokens, contextWindow),
@@ -135,7 +120,14 @@ export function buildNeuralwattProviderModelsFromApi(
135
120
  const models = apiModels
136
121
  .filter(
137
122
  (m) =>
138
- m.metadata && !m.metadata.deprecated && !m.metadata.pricing.pricing_tbd,
123
+ m.metadata &&
124
+ !m.metadata.deprecated &&
125
+ !m.metadata.pricing.pricing_tbd &&
126
+ // Exclude non-chat models (e.g. embeddings) by task
127
+ !(
128
+ m.metadata.capabilities.task &&
129
+ !["chat", "completions"].includes(m.metadata.capabilities.task)
130
+ ),
139
131
  )
140
132
  .map(apiModelToProviderModel);
141
133
  return [...models, ...buildAliases(models, apiModels)];
@@ -1,7 +1,6 @@
1
1
  import type { ProviderModelConfig } from "@earendil-works/pi-coding-agent";
2
2
  import {
3
3
  buildNeuralwattFamily,
4
- FLEX_COST_MULTIPLIER,
5
4
  type NeuralwattModelFamily,
6
5
  type NeuralwattVariantSpec,
7
6
  } from "./build";
@@ -31,7 +30,7 @@ const DEEPSEEK_V4_FLASH: NeuralwattModelFamily = {
31
30
  // `max` and `none`; every non-`none` request resolves to `max` upstream.
32
31
  // It does not reason by default (`default_enabled: false`), but the model
33
32
  // can produce reasoning traces when asked. See
34
- // https://portal.neuralwatt.com/docs/api/chat-completions#reasoning-effort
33
+ // https://docs.neuralwatt.com/api/chat-completions.md
35
34
  const GEMMA_4: NeuralwattModelFamily = {
36
35
  cost: { input: 0.144, output: 0.42, cacheRead: 0.0144 },
37
36
  vision: true,
@@ -41,25 +40,23 @@ const GEMMA_4: NeuralwattModelFamily = {
41
40
  },
42
41
  };
43
42
 
44
- // ZhipuAI. GLM-5.2 natively supports `high` and `max` reasoning efforts;
45
- // `xhigh` is an unsupported hole between them. Pi's `max` level (0.80.6) maps
46
- // to GLM's top tier.
47
- const GLM_5_2: NeuralwattModelFamily = {
43
+ // ZhipuAI. GLM-5.3 has mandatory reasoning and `none` is not offered:
44
+ // efforts are max/high/low (default max).
45
+ const GLM_5_3: NeuralwattModelFamily = {
48
46
  cost: { input: 1.45, output: 4.5, cacheRead: 0.145 },
49
47
  vision: false,
50
48
  reasoningMetadata: {
51
- supported_efforts: ["max", "high", "none"],
52
- mandatory: false,
49
+ supported_efforts: ["max", "high", "low"],
50
+ mandatory: true,
53
51
  },
54
52
  };
55
53
 
56
- // ZhipuAI. GLM-5.3 ships as a GLM-5.2 weight swap in gated preview, with
57
- // GLM-5.2 pricing parity (per the API metadata; review at launch). Unlike
58
- // 5.2, reasoning is mandatory and `none` is not offered: efforts are
59
- // max/high/low (default max).
60
- const GLM_5_3: NeuralwattModelFamily = {
61
- cost: { input: 1.45, output: 4.5, cacheRead: 0.145 },
62
- vision: false,
54
+ // ZhipuAI. GLM-5.3 Flash is the small GLM-5.3 tier: vision-capable, much
55
+ // cheaper than the flagship, with the same mandatory max/high/low reasoning
56
+ // contract as GLM-5.3.
57
+ const GLM_5_3_FLASH: NeuralwattModelFamily = {
58
+ cost: { input: 0.15, output: 0.5, cacheRead: 0.03 },
59
+ vision: true,
63
60
  reasoningMetadata: {
64
61
  supported_efforts: ["max", "high", "low"],
65
62
  mandatory: true,
@@ -99,6 +96,18 @@ const QWEN_3_6_35B: NeuralwattModelFamily = {
99
96
  },
100
97
  };
101
98
 
99
+ // Qwen. Qwen 3.8 27B tops out at `xhigh` (its default) and also supports
100
+ // `medium`, `low`, and `none`; there is no `max` effort. Reasoning is on by
101
+ // default but can be disabled.
102
+ const QWEN_3_8_27B: NeuralwattModelFamily = {
103
+ cost: { input: 0.45, output: 3.2, cacheRead: 0.25 },
104
+ vision: true,
105
+ reasoningMetadata: {
106
+ supported_efforts: ["xhigh", "medium", "low", "none"],
107
+ mandatory: false,
108
+ },
109
+ };
110
+
102
111
  const FAMILIES: [NeuralwattModelFamily, NeuralwattVariantSpec[]][] = [
103
112
  [
104
113
  DEEPSEEK_V4_FLASH,
@@ -116,7 +125,7 @@ const FAMILIES: [NeuralwattModelFamily, NeuralwattVariantSpec[]][] = [
116
125
  contextWindow: 1048560,
117
126
  maxOutputTokens: 65536,
118
127
  reasoning: true,
119
- costMultiplier: FLEX_COST_MULTIPLIER,
128
+ costMultiplier: 0.65,
120
129
  },
121
130
  ],
122
131
  ],
@@ -133,114 +142,69 @@ const FAMILIES: [NeuralwattModelFamily, NeuralwattVariantSpec[]][] = [
133
142
  ],
134
143
  ],
135
144
  [
136
- GLM_5_2,
145
+ GLM_5_3,
137
146
  [
138
147
  {
139
- id: "glm-5.2",
140
- name: "GLM-5.2",
148
+ id: "glm-5.3",
149
+ name: "GLM 5.3",
141
150
  contextWindow: 1048560,
142
151
  maxOutputTokens: null,
143
152
  reasoning: true,
144
153
  },
145
154
  {
146
- // GLM-5.2 Fast pins thinking off by default, but keeps the parent's
147
- // full reasoning contract (`high`/`max`/`none`): sending
148
- // `reasoning_effort` re-enables thinking for that request.
149
- id: "glm-5.2-fast",
150
- name: "GLM-5.2 (fast)",
155
+ id: "glm-5.3-flex",
156
+ name: "GLM 5.3 (flex)",
151
157
  contextWindow: 1048560,
152
158
  maxOutputTokens: null,
153
159
  reasoning: true,
160
+ costMultiplier: 0.65,
154
161
  },
162
+ ],
163
+ ],
164
+ [
165
+ GLM_5_3_FLASH,
166
+ [
155
167
  {
156
- id: "glm-5.2-flex",
157
- name: "GLM-5.2 (flex)",
168
+ id: "glm-5.3-flash",
169
+ name: "GLM-5.3 Flash",
158
170
  contextWindow: 1048560,
159
171
  maxOutputTokens: null,
160
172
  reasoning: true,
161
- costMultiplier: FLEX_COST_MULTIPLIER,
162
- },
163
- {
164
- id: "glm-5.2-short",
165
- name: "GLM-5.2 Short",
166
- contextWindow: 199984,
167
- maxOutputTokens: 32000,
168
- reasoning: true,
169
- },
170
- {
171
- // Short/fast: pins thinking off but keeps the parent reasoning
172
- // contract, like glm-5.2-fast.
173
- id: "glm-5.2-short-fast",
174
- name: "GLM-5.2 (short, fast)",
175
- contextWindow: 199984,
176
- maxOutputTokens: 32000,
177
- reasoning: true,
178
- },
179
- {
180
- id: "glm-5.2-short-flex",
181
- name: "GLM-5.2 (short, flex)",
182
- contextWindow: 199984,
183
- maxOutputTokens: 32000,
184
- reasoning: true,
185
- costMultiplier: FLEX_COST_MULTIPLIER,
186
- },
187
- {
188
- // Short/fast/flex: pins thinking off but keeps the parent reasoning
189
- // contract, like glm-5.2-fast.
190
- id: "glm-5.2-short-fast-flex",
191
- name: "GLM-5.2 (short, fast, flex)",
192
- contextWindow: 199984,
193
- maxOutputTokens: 32000,
194
- reasoning: true,
195
- costMultiplier: FLEX_COST_MULTIPLIER,
196
173
  },
197
- ],
198
- ],
199
- [
200
- GLM_5_3,
201
- [
202
174
  {
203
- id: "glm-5.3",
204
- name: "GLM-5.3",
175
+ id: "glm-5.3-flash-flex",
176
+ name: "GLM-5.3 Flash (flex)",
205
177
  contextWindow: 1048560,
206
178
  maxOutputTokens: null,
207
179
  reasoning: true,
180
+ costMultiplier: 0.65,
208
181
  },
209
182
  ],
210
183
  ],
211
- // The kimi-k3 endpoint rejects anything above 327,680 total tokens with
212
- // `400: max_completion_tokens is too large … supports at most 327680
213
- // completion tokens` (verified at runtime), even though the API advertises
214
- // `max_model_len: 1048560` with a null output cap for the whole family.
215
- // The -fast/-flex endpoints don't enforce any cap server-side yet (they
216
- // accept max_completion_tokens beyond the advertised window), but they are
217
- // the same K3 deployment and are expected to share the 327,680 limit, so
218
- // all three variants are pinned to it. The drift check in models.test.ts
219
- // whitelists this divergence via CONTEXT_WINDOW_OVERRIDES.
220
184
  [
221
185
  KIMI_K3,
222
186
  [
223
187
  {
224
188
  id: "kimi-k3",
225
189
  name: "Kimi K3",
226
- contextWindow: 327680,
227
- maxOutputTokens: 327680,
190
+ contextWindow: 1048560,
191
+ maxOutputTokens: null,
228
192
  reasoning: true,
229
193
  },
230
194
  {
231
195
  id: "kimi-k3-fast",
232
196
  name: "Kimi K3 Fast",
233
- contextWindow: 327680,
234
- maxOutputTokens: 327680,
197
+ contextWindow: 1048560,
198
+ maxOutputTokens: null,
235
199
  reasoning: false,
236
200
  },
237
201
  {
238
202
  id: "kimi-k3-flex",
239
203
  name: "Kimi K3 (flex)",
240
- contextWindow: 327680,
241
- maxOutputTokens: 327680,
204
+ contextWindow: 1048560,
205
+ maxOutputTokens: null,
242
206
  reasoning: true,
243
- costMultiplier: FLEX_COST_MULTIPLIER,
207
+ costMultiplier: 0.65,
244
208
  },
245
209
  ],
246
210
  ],
@@ -269,7 +233,7 @@ const FAMILIES: [NeuralwattModelFamily, NeuralwattVariantSpec[]][] = [
269
233
  contextWindow: 262128,
270
234
  maxOutputTokens: null,
271
235
  reasoning: true,
272
- costMultiplier: FLEX_COST_MULTIPLIER,
236
+ costMultiplier: 0.65,
273
237
  },
274
238
  ],
275
239
  ],
@@ -290,16 +254,43 @@ const FAMILIES: [NeuralwattModelFamily, NeuralwattVariantSpec[]][] = [
290
254
  maxOutputTokens: null,
291
255
  reasoning: false,
292
256
  },
257
+ {
258
+ id: "qwen3.6-35b-flex",
259
+ name: "Qwen3.6 35B (flex)",
260
+ contextWindow: 131056,
261
+ maxOutputTokens: null,
262
+ reasoning: true,
263
+ costMultiplier: 0.65,
264
+ },
265
+ ],
266
+ ],
267
+ [
268
+ QWEN_3_8_27B,
269
+ [
270
+ {
271
+ id: "qwen-3.8-27b",
272
+ name: "Qwen 3.8 27B",
273
+ contextWindow: 262128,
274
+ maxOutputTokens: 131072,
275
+ reasoning: true,
276
+ },
277
+ {
278
+ id: "qwen-3.8-27b-flex",
279
+ name: "Qwen 3.8 27B (flex)",
280
+ contextWindow: 262128,
281
+ maxOutputTokens: 131072,
282
+ reasoning: true,
283
+ costMultiplier: 0.65,
284
+ },
293
285
  ],
294
286
  ],
295
287
  ];
296
288
 
297
289
  // `-flex` variants are the Flex tier: same model, context window, output cap,
298
- // and prompt cache as the standard variant, admitted on spare capacity.
299
- // The API now advertises flex variants but lists them at standard pricing;
300
- // the 35% Flex discount is a billing-time concept applied here via
301
- // `costMultiplier` rather than reflected in the catalog metadata.
302
- // https://portal.neuralwatt.com/docs/guides/flex-tier
290
+ // and prompt cache as the standard variant, admitted on spare capacity. The
291
+ // API lists them at discounted prices; the fallback mirrors that with
292
+ // `costMultiplier: 0.65` per variant.
293
+ // https://docs.neuralwatt.com/guides/flex-tier.md
303
294
 
304
295
  export const NEURALWATT_MODELS: ProviderModelConfig[] = FAMILIES.flatMap(
305
296
  ([family, variants]) => buildNeuralwattFamily(family, variants),
@@ -12,6 +12,38 @@ import type {
12
12
 
13
13
  export const MODEL_STORE_TTL_MS = 4 * 60 * 60 * 1000;
14
14
 
15
+ /**
16
+ * Scope a store entry applies to. The public catalog is a subset of any
17
+ * key-scoped catalog (preview, grant-gated, private models), so an entry
18
+ * stamped "public" must not shadow a keyed refresh — and a key-scoped entry
19
+ * must not be replayed for an anonymous user. Matches the anonymous-key
20
+ * convention in src/lib/neuralwatt-api.ts (authHeaders).
21
+ */
22
+ const CATALOG_SCOPE_VERSION = "v1";
23
+ type CatalogScope = "public" | "key";
24
+
25
+ type ScopedModelsStoreEntry = ModelsStoreEntry & { catalogKey?: string };
26
+
27
+ function catalogScope(apiKey: string | undefined): CatalogScope {
28
+ return apiKey !== undefined && apiKey !== "" && apiKey !== "-"
29
+ ? "key"
30
+ : "public";
31
+ }
32
+
33
+ function storedCatalogKey(entry: ScopedModelsStoreEntry): string | undefined {
34
+ return entry.catalogKey;
35
+ }
36
+
37
+ function catalogKeyMatches(
38
+ entry: ScopedModelsStoreEntry | undefined,
39
+ scope: CatalogScope,
40
+ ): boolean {
41
+ return (
42
+ entry !== undefined &&
43
+ storedCatalogKey(entry) === `${scope} ${CATALOG_SCOPE_VERSION}`
44
+ );
45
+ }
46
+
15
47
  export type FetchNeuralwattApiModels = (
16
48
  apiKey: string | undefined,
17
49
  signal?: AbortSignal,
@@ -25,6 +57,13 @@ function isFreshStoreEntry(
25
57
  return Date.now() - checkedAt < MODEL_STORE_TTL_MS;
26
58
  }
27
59
 
60
+ function isUsableStoreEntry(
61
+ entry: Readonly<ModelsStoreEntry> | undefined,
62
+ scope: CatalogScope,
63
+ ): entry is ModelsStoreEntry {
64
+ return isFreshStoreEntry(entry) && catalogKeyMatches(entry, scope);
65
+ }
66
+
28
67
  export function createNeuralwattRefreshModels(
29
68
  staticModels: ReturnType<typeof buildNeuralwattProviderModels>,
30
69
  fetchApiModels: FetchNeuralwattApiModels,
@@ -35,29 +74,29 @@ export function createNeuralwattRefreshModels(
35
74
  context.signal.throwIfAborted();
36
75
  const fallback = buildFromStore(staticModels);
37
76
  try {
38
- if (!context.allowNetwork) {
39
- return context.stored
40
- ? buildFromStore(context.stored.models)
41
- : fallback;
42
- }
43
- if (!context.force && isFreshStoreEntry(context.stored)) {
44
- return buildFromStore(context.stored.models);
45
- }
46
77
  const apiKey =
47
78
  context.credential?.type === "api_key"
48
79
  ? context.credential.key
49
80
  : undefined;
81
+ const scope = catalogScope(apiKey);
82
+ const stored = context.stored as ScopedModelsStoreEntry | undefined;
83
+ if (!context.allowNetwork) {
84
+ return stored !== undefined && catalogKeyMatches(stored, scope)
85
+ ? buildFromStore(stored.models)
86
+ : fallback;
87
+ }
88
+ if (!context.force && isUsableStoreEntry(stored, scope)) {
89
+ return buildFromStore(stored.models);
90
+ }
50
91
  const apiModels = await fetchApiModels(apiKey, context.signal);
51
92
  context.signal.throwIfAborted();
52
93
  const models = buildFromApi(apiModels);
53
- await context
54
- .publish({
55
- persist: {
56
- models: models as unknown as ModelsStoreEntry["models"],
57
- checkedAt: Date.now(),
58
- },
59
- })
60
- .catch(() => undefined);
94
+ const entry: ScopedModelsStoreEntry = {
95
+ models: models as unknown as ModelsStoreEntry["models"],
96
+ checkedAt: Date.now(),
97
+ catalogKey: `${scope} ${CATALOG_SCOPE_VERSION}`,
98
+ };
99
+ await context.publish({ persist: entry }).catch(() => undefined);
61
100
  context.signal.throwIfAborted();
62
101
  return models;
63
102
  } catch (error) {
@@ -12,7 +12,7 @@ interface AssistantErrorLike {
12
12
  * layer. Each sets unique headers so the client can tell which layer
13
13
  * triggered the rejection.
14
14
  *
15
- * @see https://portal.neuralwatt.com/docs/guides/rate-limits
15
+ * @see https://docs.neuralwatt.com/guides/rate-limits.md
16
16
  */
17
17
  export interface NeuralwattRateLimitInfo {
18
18
  /** Which rate-limit layer triggered the 429 */
@@ -8,10 +8,37 @@ const COOLDOWN_MS = 60 * 60 * 1000; // 60 minutes
8
8
  const LOW_PCT = 25;
9
9
  const CRITICAL_PCT = 10;
10
10
 
11
- /** Per-kWh price once a subscription's included kWh are exhausted. */
12
- const OVERAGE_RATE_PER_KWH_SUBSCRIBED = 5;
13
- /** Per-kWh price when there is no active subscription (no included kWh). */
11
+ /** $/kWh by plan on a monthly interval. docs.neuralwatt.com/billing/faq */
12
+ const OVERAGE_RATES_MONTHLY = {
13
+ basic: 8.5,
14
+ standard: 8.0,
15
+ pro: 7.5,
16
+ max: 7.0,
17
+ } as const;
18
+ /** $/kWh by plan on an annual interval. */
19
+ const OVERAGE_RATES_ANNUAL = {
20
+ basic: 7.08,
21
+ standard: 6.67,
22
+ pro: 6.25,
23
+ max: 5.83,
24
+ } as const;
25
+ /** Pay-as-you-go, verified on portal.neuralwatt.com/pricing. */
14
26
  const OVERAGE_RATE_PER_KWH_UNSUBSCRIBED = 10;
27
+ /** Unknown plan on a subscription: fall back to the Standard monthly rate. */
28
+ const OVERAGE_RATE_PER_KWH_UNKNOWN_SUBSCRIBED = 8.0;
29
+
30
+ function resolveOverageRate(sub: NeuralwattQuotas["subscription"]): number {
31
+ if (!sub) return OVERAGE_RATE_PER_KWH_UNSUBSCRIBED;
32
+ const plan = sub.plan.toLowerCase();
33
+ const table =
34
+ sub.billing_interval === "year"
35
+ ? OVERAGE_RATES_ANNUAL
36
+ : OVERAGE_RATES_MONTHLY;
37
+ return (
38
+ (table as Record<string, number>)[plan] ??
39
+ OVERAGE_RATE_PER_KWH_UNKNOWN_SUBSCRIBED
40
+ );
41
+ }
15
42
 
16
43
  interface AlertState {
17
44
  lastSeverity: WarningSeverity;
@@ -112,9 +139,7 @@ export function computeOverageProgress(
112
139
  )
113
140
  : quotas.usage.current_month.energy_kwh;
114
141
 
115
- const rate = hasSub
116
- ? OVERAGE_RATE_PER_KWH_SUBSCRIBED
117
- : OVERAGE_RATE_PER_KWH_UNSUBSCRIBED;
142
+ const rate = resolveOverageRate(quotas.subscription);
118
143
  const costUsd = overageKwh * rate;
119
144
  const remainingUsd = Math.max(0, capUsd - costUsd);
120
145
  const pctRemaining = capUsd > 0 ? (remainingUsd / capUsd) * 100 : 0;
@@ -153,9 +178,12 @@ function overageWarning(progress: OverageProgress): PendingWarning {
153
178
  * no subscription, cap set → overage cap progress (all kWh billable)
154
179
  * no subscription, no cap → balance credits
155
180
  *
156
- * Overage cost is derived from kWh usage: subscribed pays $5/kWh for kWh
157
- * beyond the included quota; unsubscribed pays $10/kWh for all usage. There is
158
- * no overage-spent counter in the API, so progress is computed.
181
+ * Overage cost is derived from kWh usage: subscribed pays a per-plan rate
182
+ * ($7.00–$8.50/kWh by plan and billing interval) for kWh beyond the included
183
+ * quota; unsubscribed pays $10/kWh for all usage. There is no overage-spent
184
+ * counter in the API, so progress is computed. Note `kwh_used` is the
185
+ * *charged* energy — flex usage bills at 0.65× kWh — so the "kWh over" figure
186
+ * is billed kWh, not physical consumption.
159
187
  *
160
188
  * Usage totals (monthly/lifetime cost in USD) are deliberately not used as a
161
189
  * threshold basis — they are not directly tied to the subscription's kWh quota.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@aliou/pi-neuralwatt",
3
- "version": "0.15.2",
3
+ "version": "0.15.4",
4
4
  "license": "MIT",
5
5
  "type": "module",
6
6
  "private": false,
@@ -5,6 +5,10 @@ export interface NeuralwattApiModelPricing {
5
5
  cached_output_per_million: number | null;
6
6
  currency: string;
7
7
  pricing_tbd: boolean;
8
+ /** Service tier the pricing applies to (e.g. "standard", "flex"). */
9
+ service_tier?: string;
10
+ /** Flex tier cost multiplier (e.g. 0.65); null/absent on standard pricing. */
11
+ flex_discount_multiplier?: number | null;
8
12
  }
9
13
 
10
14
  export interface NeuralwattApiModelCapabilities {
@@ -16,6 +20,8 @@ export interface NeuralwattApiModelCapabilities {
16
20
  streaming: boolean;
17
21
  system_role: boolean;
18
22
  developer_role: boolean;
23
+ task?: string;
24
+ embedding_dimensions?: number;
19
25
  }
20
26
 
21
27
  /**
@@ -28,6 +28,10 @@ export interface NeuralwattQuotas {
28
28
  total_credits_usd: number;
29
29
  credits_used_usd: number;
30
30
  accounting_method: string;
31
+ /** Legacy credit pool split (USD). Typed for fidelity, not consumed yet. */
32
+ legacy_credits_usd?: number;
33
+ /** New credit pool split (USD). Typed for fidelity, not consumed yet. */
34
+ new_credits_usd?: number;
31
35
  };
32
36
  usage: {
33
37
  lifetime: {