@aliou/pi-neuralwatt 0.15.3 → 0.15.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -8,15 +8,6 @@ export type ThinkingLevelMap = NonNullable<
8
8
  ProviderModelConfig["thinkingLevelMap"]
9
9
  >;
10
10
 
11
- /**
12
- * Flex tier is billed at 65% of standard pricing (35% off) when the request
13
- * streams. A non-streaming request to a `-flex` model silently falls back to
14
- * the standard tier and the standard price.
15
- *
16
- * https://docs.neuralwatt.com/guides/flex-tier.md
17
- */
18
- export const FLEX_COST_MULTIPLIER = 0.65;
19
-
20
11
  export interface NeuralwattCost {
21
12
  input: number;
22
13
  output: number;
@@ -2,7 +2,6 @@ import type { ProviderModelConfig } from "@earendil-works/pi-coding-agent";
2
2
  import type { NeuralwattApiModel } from "../../../src/types/models-api";
3
3
  import {
4
4
  buildThinkingLevelMap,
5
- FLEX_COST_MULTIPLIER,
6
5
  resolveMaxTokens,
7
6
  type ThinkingLevelMap,
8
7
  } from "./build";
@@ -24,16 +23,11 @@ const COMPAT_OVERRIDES: Partial<
24
23
  };
25
24
 
26
25
  const HARDCODED_ALIASES: Record<string, string> = {
27
- "zai-org/GLM-5.2-FP8": "glm-5.2",
28
26
  "moonshotai/Kimi-K2.7-Code": "kimi-k2.7-code",
29
27
  "Qwen/Qwen3.6-35B-A3B": "qwen3.6-35b",
30
28
  "deepseek-ai/DeepSeek-V4-Flash": "deepseek-v4-flash",
31
29
  };
32
30
 
33
- function isFlexModelId(id: string): boolean {
34
- return id.endsWith("-flex");
35
- }
36
-
37
31
  function isVariantId(id: string): boolean {
38
32
  return id.includes("-fast") || id.includes("-flex") || id.includes("-short");
39
33
  }
@@ -46,8 +40,6 @@ function apiModelToProviderModel(model: NeuralwattApiModel): NeuralwattModel {
46
40
  );
47
41
 
48
42
  const reasoning = meta.capabilities.reasoning;
49
- // Flex variants are billed at 0.65x when streaming (35% off).
50
- const multiplier = isFlexModelId(model.id) ? FLEX_COST_MULTIPLIER : 1;
51
43
 
52
44
  const compat: NonNullable<ProviderModelConfig["compat"]> = {
53
45
  supportsDeveloperRole: meta.capabilities.developer_role,
@@ -66,10 +58,10 @@ function apiModelToProviderModel(model: NeuralwattApiModel): NeuralwattModel {
66
58
  ? (["text", "image"] as const)
67
59
  : (["text"] as const),
68
60
  cost: {
69
- input: meta.pricing.input_per_million * multiplier,
70
- output: meta.pricing.output_per_million * multiplier,
71
- cacheRead: (meta.pricing.cached_input_per_million ?? 0) * multiplier,
72
- cacheWrite: (meta.pricing.cached_output_per_million ?? 0) * multiplier,
61
+ input: meta.pricing.input_per_million,
62
+ output: meta.pricing.output_per_million,
63
+ cacheRead: meta.pricing.cached_input_per_million ?? 0,
64
+ cacheWrite: meta.pricing.cached_output_per_million ?? 0,
73
65
  },
74
66
  contextWindow,
75
67
  maxTokens: resolveMaxTokens(meta.limits.max_output_tokens, contextWindow),
@@ -1,7 +1,6 @@
1
1
  import type { ProviderModelConfig } from "@earendil-works/pi-coding-agent";
2
2
  import {
3
3
  buildNeuralwattFamily,
4
- FLEX_COST_MULTIPLIER,
5
4
  type NeuralwattModelFamily,
6
5
  type NeuralwattVariantSpec,
7
6
  } from "./build";
@@ -41,25 +40,23 @@ const GEMMA_4: NeuralwattModelFamily = {
41
40
  },
42
41
  };
43
42
 
44
- // ZhipuAI. GLM-5.2 natively supports `high` and `max` reasoning efforts;
45
- // `xhigh` is an unsupported hole between them. Pi's `max` level (0.80.6) maps
46
- // to GLM's top tier.
47
- const GLM_5_2: NeuralwattModelFamily = {
43
+ // ZhipuAI. GLM-5.3 has mandatory reasoning and `none` is not offered:
44
+ // efforts are max/high/low (default max).
45
+ const GLM_5_3: NeuralwattModelFamily = {
48
46
  cost: { input: 1.45, output: 4.5, cacheRead: 0.145 },
49
47
  vision: false,
50
48
  reasoningMetadata: {
51
- supported_efforts: ["max", "high", "none"],
52
- mandatory: false,
49
+ supported_efforts: ["max", "high", "low"],
50
+ mandatory: true,
53
51
  },
54
52
  };
55
53
 
56
- // ZhipuAI. GLM-5.3 ships as a GLM-5.2 weight swap in gated preview, with
57
- // GLM-5.2 pricing parity (per the API metadata; review at launch). Unlike
58
- // 5.2, reasoning is mandatory and `none` is not offered: efforts are
59
- // max/high/low (default max).
60
- const GLM_5_3: NeuralwattModelFamily = {
61
- cost: { input: 1.45, output: 4.5, cacheRead: 0.145 },
62
- vision: false,
54
+ // ZhipuAI. GLM-5.3 Flash is the small GLM-5.3 tier: vision-capable, much
55
+ // cheaper than the flagship, with the same mandatory max/high/low reasoning
56
+ // contract as GLM-5.3.
57
+ const GLM_5_3_FLASH: NeuralwattModelFamily = {
58
+ cost: { input: 0.15, output: 0.5, cacheRead: 0.03 },
59
+ vision: true,
63
60
  reasoningMetadata: {
64
61
  supported_efforts: ["max", "high", "low"],
65
62
  mandatory: true,
@@ -99,6 +96,18 @@ const QWEN_3_6_35B: NeuralwattModelFamily = {
99
96
  },
100
97
  };
101
98
 
99
+ // Qwen. Qwen 3.8 27B tops out at `xhigh` (its default) and also supports
100
+ // `medium`, `low`, and `none`; there is no `max` effort. Reasoning is on by
101
+ // default but can be disabled.
102
+ const QWEN_3_8_27B: NeuralwattModelFamily = {
103
+ cost: { input: 0.45, output: 3.2, cacheRead: 0.25 },
104
+ vision: true,
105
+ reasoningMetadata: {
106
+ supported_efforts: ["xhigh", "medium", "low", "none"],
107
+ mandatory: false,
108
+ },
109
+ };
110
+
102
111
  const FAMILIES: [NeuralwattModelFamily, NeuralwattVariantSpec[]][] = [
103
112
  [
104
113
  DEEPSEEK_V4_FLASH,
@@ -116,7 +125,7 @@ const FAMILIES: [NeuralwattModelFamily, NeuralwattVariantSpec[]][] = [
116
125
  contextWindow: 1048560,
117
126
  maxOutputTokens: 65536,
118
127
  reasoning: true,
119
- costMultiplier: FLEX_COST_MULTIPLIER,
128
+ costMultiplier: 0.65,
120
129
  },
121
130
  ],
122
131
  ],
@@ -133,78 +142,42 @@ const FAMILIES: [NeuralwattModelFamily, NeuralwattVariantSpec[]][] = [
133
142
  ],
134
143
  ],
135
144
  [
136
- GLM_5_2,
145
+ GLM_5_3,
137
146
  [
138
147
  {
139
- id: "glm-5.2",
140
- name: "GLM-5.2",
148
+ id: "glm-5.3",
149
+ name: "GLM 5.3",
141
150
  contextWindow: 1048560,
142
151
  maxOutputTokens: null,
143
152
  reasoning: true,
144
153
  },
145
154
  {
146
- // GLM-5.2 Fast pins thinking off by default, but keeps the parent's
147
- // full reasoning contract (`high`/`max`/`none`): sending
148
- // `reasoning_effort` re-enables thinking for that request.
149
- id: "glm-5.2-fast",
150
- name: "GLM-5.2 (fast)",
155
+ id: "glm-5.3-flex",
156
+ name: "GLM 5.3 (flex)",
151
157
  contextWindow: 1048560,
152
158
  maxOutputTokens: null,
153
159
  reasoning: true,
160
+ costMultiplier: 0.65,
154
161
  },
162
+ ],
163
+ ],
164
+ [
165
+ GLM_5_3_FLASH,
166
+ [
155
167
  {
156
- id: "glm-5.2-flex",
157
- name: "GLM-5.2 (flex)",
168
+ id: "glm-5.3-flash",
169
+ name: "GLM-5.3 Flash",
158
170
  contextWindow: 1048560,
159
171
  maxOutputTokens: null,
160
172
  reasoning: true,
161
- costMultiplier: FLEX_COST_MULTIPLIER,
162
- },
163
- {
164
- id: "glm-5.2-short",
165
- name: "GLM-5.2 Short",
166
- contextWindow: 199984,
167
- maxOutputTokens: 32000,
168
- reasoning: true,
169
- },
170
- {
171
- // Short/fast: pins thinking off but keeps the parent reasoning
172
- // contract, like glm-5.2-fast.
173
- id: "glm-5.2-short-fast",
174
- name: "GLM-5.2 (short, fast)",
175
- contextWindow: 199984,
176
- maxOutputTokens: 32000,
177
- reasoning: true,
178
- },
179
- {
180
- id: "glm-5.2-short-flex",
181
- name: "GLM-5.2 (short, flex)",
182
- contextWindow: 199984,
183
- maxOutputTokens: 32000,
184
- reasoning: true,
185
- costMultiplier: FLEX_COST_MULTIPLIER,
186
173
  },
187
174
  {
188
- // Short/fast/flex: pins thinking off but keeps the parent reasoning
189
- // contract, like glm-5.2-fast.
190
- id: "glm-5.2-short-fast-flex",
191
- name: "GLM-5.2 (short, fast, flex)",
192
- contextWindow: 199984,
193
- maxOutputTokens: 32000,
194
- reasoning: true,
195
- costMultiplier: FLEX_COST_MULTIPLIER,
196
- },
197
- ],
198
- ],
199
- [
200
- GLM_5_3,
201
- [
202
- {
203
- id: "glm-5.3",
204
- name: "GLM-5.3",
175
+ id: "glm-5.3-flash-flex",
176
+ name: "GLM-5.3 Flash (flex)",
205
177
  contextWindow: 1048560,
206
178
  maxOutputTokens: null,
207
179
  reasoning: true,
180
+ costMultiplier: 0.65,
208
181
  },
209
182
  ],
210
183
  ],
@@ -231,7 +204,7 @@ const FAMILIES: [NeuralwattModelFamily, NeuralwattVariantSpec[]][] = [
231
204
  contextWindow: 1048560,
232
205
  maxOutputTokens: null,
233
206
  reasoning: true,
234
- costMultiplier: FLEX_COST_MULTIPLIER,
207
+ costMultiplier: 0.65,
235
208
  },
236
209
  ],
237
210
  ],
@@ -260,7 +233,7 @@ const FAMILIES: [NeuralwattModelFamily, NeuralwattVariantSpec[]][] = [
260
233
  contextWindow: 262128,
261
234
  maxOutputTokens: null,
262
235
  reasoning: true,
263
- costMultiplier: FLEX_COST_MULTIPLIER,
236
+ costMultiplier: 0.65,
264
237
  },
265
238
  ],
266
239
  ],
@@ -281,15 +254,42 @@ const FAMILIES: [NeuralwattModelFamily, NeuralwattVariantSpec[]][] = [
281
254
  maxOutputTokens: null,
282
255
  reasoning: false,
283
256
  },
257
+ {
258
+ id: "qwen3.6-35b-flex",
259
+ name: "Qwen3.6 35B (flex)",
260
+ contextWindow: 131056,
261
+ maxOutputTokens: null,
262
+ reasoning: true,
263
+ costMultiplier: 0.65,
264
+ },
265
+ ],
266
+ ],
267
+ [
268
+ QWEN_3_8_27B,
269
+ [
270
+ {
271
+ id: "qwen-3.8-27b",
272
+ name: "Qwen 3.8 27B",
273
+ contextWindow: 262128,
274
+ maxOutputTokens: 131072,
275
+ reasoning: true,
276
+ },
277
+ {
278
+ id: "qwen-3.8-27b-flex",
279
+ name: "Qwen 3.8 27B (flex)",
280
+ contextWindow: 262128,
281
+ maxOutputTokens: 131072,
282
+ reasoning: true,
283
+ costMultiplier: 0.65,
284
+ },
284
285
  ],
285
286
  ],
286
287
  ];
287
288
 
288
289
  // `-flex` variants are the Flex tier: same model, context window, output cap,
289
- // and prompt cache as the standard variant, admitted on spare capacity.
290
- // The API now advertises flex variants but lists them at standard pricing;
291
- // the 35% Flex discount is a billing-time concept applied here via
292
- // `costMultiplier` rather than reflected in the catalog metadata.
290
+ // and prompt cache as the standard variant, admitted on spare capacity. The
291
+ // API lists them at discounted prices; the fallback mirrors that with
292
+ // `costMultiplier: 0.65` per variant.
293
293
  // https://docs.neuralwatt.com/guides/flex-tier.md
294
294
 
295
295
  export const NEURALWATT_MODELS: ProviderModelConfig[] = FAMILIES.flatMap(
@@ -8,10 +8,37 @@ const COOLDOWN_MS = 60 * 60 * 1000; // 60 minutes
8
8
  const LOW_PCT = 25;
9
9
  const CRITICAL_PCT = 10;
10
10
 
11
- /** Per-kWh price once a subscription's included kWh are exhausted. */
12
- const OVERAGE_RATE_PER_KWH_SUBSCRIBED = 5;
13
- /** Per-kWh price when there is no active subscription (no included kWh). */
11
+ /** $/kWh by plan on a monthly interval. docs.neuralwatt.com/billing/faq */
12
+ const OVERAGE_RATES_MONTHLY = {
13
+ basic: 8.5,
14
+ standard: 8.0,
15
+ pro: 7.5,
16
+ max: 7.0,
17
+ } as const;
18
+ /** $/kWh by plan on an annual interval. */
19
+ const OVERAGE_RATES_ANNUAL = {
20
+ basic: 7.08,
21
+ standard: 6.67,
22
+ pro: 6.25,
23
+ max: 5.83,
24
+ } as const;
25
+ /** Pay-as-you-go, verified on portal.neuralwatt.com/pricing. */
14
26
  const OVERAGE_RATE_PER_KWH_UNSUBSCRIBED = 10;
27
+ /** Unknown plan on a subscription: fall back to the Standard monthly rate. */
28
+ const OVERAGE_RATE_PER_KWH_UNKNOWN_SUBSCRIBED = 8.0;
29
+
30
+ function resolveOverageRate(sub: NeuralwattQuotas["subscription"]): number {
31
+ if (!sub) return OVERAGE_RATE_PER_KWH_UNSUBSCRIBED;
32
+ const plan = sub.plan.toLowerCase();
33
+ const table =
34
+ sub.billing_interval === "year"
35
+ ? OVERAGE_RATES_ANNUAL
36
+ : OVERAGE_RATES_MONTHLY;
37
+ return (
38
+ (table as Record<string, number>)[plan] ??
39
+ OVERAGE_RATE_PER_KWH_UNKNOWN_SUBSCRIBED
40
+ );
41
+ }
15
42
 
16
43
  interface AlertState {
17
44
  lastSeverity: WarningSeverity;
@@ -112,9 +139,7 @@ export function computeOverageProgress(
112
139
  )
113
140
  : quotas.usage.current_month.energy_kwh;
114
141
 
115
- const rate = hasSub
116
- ? OVERAGE_RATE_PER_KWH_SUBSCRIBED
117
- : OVERAGE_RATE_PER_KWH_UNSUBSCRIBED;
142
+ const rate = resolveOverageRate(quotas.subscription);
118
143
  const costUsd = overageKwh * rate;
119
144
  const remainingUsd = Math.max(0, capUsd - costUsd);
120
145
  const pctRemaining = capUsd > 0 ? (remainingUsd / capUsd) * 100 : 0;
@@ -153,9 +178,12 @@ function overageWarning(progress: OverageProgress): PendingWarning {
153
178
  * no subscription, cap set → overage cap progress (all kWh billable)
154
179
  * no subscription, no cap → balance credits
155
180
  *
156
- * Overage cost is derived from kWh usage: subscribed pays $5/kWh for kWh
157
- * beyond the included quota; unsubscribed pays $10/kWh for all usage. There is
158
- * no overage-spent counter in the API, so progress is computed.
181
+ * Overage cost is derived from kWh usage: subscribed pays a per-plan rate
182
+ * ($7.00–$8.50/kWh by plan and billing interval) for kWh beyond the included
183
+ * quota; unsubscribed pays $10/kWh for all usage. There is no overage-spent
184
+ * counter in the API, so progress is computed. Note `kwh_used` is the
185
+ * *charged* energy — flex usage bills at 0.65× kWh — so the "kWh over" figure
186
+ * is billed kWh, not physical consumption.
159
187
  *
160
188
  * Usage totals (monthly/lifetime cost in USD) are deliberately not used as a
161
189
  * threshold basis — they are not directly tied to the subscription's kWh quota.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@aliou/pi-neuralwatt",
3
- "version": "0.15.3",
3
+ "version": "0.15.4",
4
4
  "license": "MIT",
5
5
  "type": "module",
6
6
  "private": false,
@@ -5,6 +5,10 @@ export interface NeuralwattApiModelPricing {
5
5
  cached_output_per_million: number | null;
6
6
  currency: string;
7
7
  pricing_tbd: boolean;
8
+ /** Service tier the pricing applies to (e.g. "standard", "flex"). */
9
+ service_tier?: string;
10
+ /** Flex tier cost multiplier (e.g. 0.65); null/absent on standard pricing. */
11
+ flex_discount_multiplier?: number | null;
8
12
  }
9
13
 
10
14
  export interface NeuralwattApiModelCapabilities {
@@ -28,6 +28,10 @@ export interface NeuralwattQuotas {
28
28
  total_credits_usd: number;
29
29
  credits_used_usd: number;
30
30
  accounting_method: string;
31
+ /** Legacy credit pool split (USD). Typed for fidelity, not consumed yet. */
32
+ legacy_credits_usd?: number;
33
+ /** New credit pool split (USD). Typed for fidelity, not consumed yet. */
34
+ new_credits_usd?: number;
31
35
  };
32
36
  usage: {
33
37
  lifetime: {