@aliou/pi-neuralwatt 0.15.3 → 0.15.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/extensions/provider/models/build.ts +0 -9
- package/extensions/provider/models/catalog.ts +4 -12
- package/extensions/provider/models/public-models.ts +73 -73
- package/extensions/quota-warnings/notifier.ts +37 -9
- package/package.json +1 -1
- package/src/types/models-api.ts +4 -0
- package/src/types/quota-api.ts +4 -0
|
@@ -8,15 +8,6 @@ export type ThinkingLevelMap = NonNullable<
|
|
|
8
8
|
ProviderModelConfig["thinkingLevelMap"]
|
|
9
9
|
>;
|
|
10
10
|
|
|
11
|
-
/**
|
|
12
|
-
* Flex tier is billed at 65% of standard pricing (35% off) when the request
|
|
13
|
-
* streams. A non-streaming request to a `-flex` model silently falls back to
|
|
14
|
-
* the standard tier and the standard price.
|
|
15
|
-
*
|
|
16
|
-
* https://docs.neuralwatt.com/guides/flex-tier.md
|
|
17
|
-
*/
|
|
18
|
-
export const FLEX_COST_MULTIPLIER = 0.65;
|
|
19
|
-
|
|
20
11
|
export interface NeuralwattCost {
|
|
21
12
|
input: number;
|
|
22
13
|
output: number;
|
|
@@ -2,7 +2,6 @@ import type { ProviderModelConfig } from "@earendil-works/pi-coding-agent";
|
|
|
2
2
|
import type { NeuralwattApiModel } from "../../../src/types/models-api";
|
|
3
3
|
import {
|
|
4
4
|
buildThinkingLevelMap,
|
|
5
|
-
FLEX_COST_MULTIPLIER,
|
|
6
5
|
resolveMaxTokens,
|
|
7
6
|
type ThinkingLevelMap,
|
|
8
7
|
} from "./build";
|
|
@@ -24,16 +23,11 @@ const COMPAT_OVERRIDES: Partial<
|
|
|
24
23
|
};
|
|
25
24
|
|
|
26
25
|
const HARDCODED_ALIASES: Record<string, string> = {
|
|
27
|
-
"zai-org/GLM-5.2-FP8": "glm-5.2",
|
|
28
26
|
"moonshotai/Kimi-K2.7-Code": "kimi-k2.7-code",
|
|
29
27
|
"Qwen/Qwen3.6-35B-A3B": "qwen3.6-35b",
|
|
30
28
|
"deepseek-ai/DeepSeek-V4-Flash": "deepseek-v4-flash",
|
|
31
29
|
};
|
|
32
30
|
|
|
33
|
-
function isFlexModelId(id: string): boolean {
|
|
34
|
-
return id.endsWith("-flex");
|
|
35
|
-
}
|
|
36
|
-
|
|
37
31
|
function isVariantId(id: string): boolean {
|
|
38
32
|
return id.includes("-fast") || id.includes("-flex") || id.includes("-short");
|
|
39
33
|
}
|
|
@@ -46,8 +40,6 @@ function apiModelToProviderModel(model: NeuralwattApiModel): NeuralwattModel {
|
|
|
46
40
|
);
|
|
47
41
|
|
|
48
42
|
const reasoning = meta.capabilities.reasoning;
|
|
49
|
-
// Flex variants are billed at 0.65x when streaming (35% off).
|
|
50
|
-
const multiplier = isFlexModelId(model.id) ? FLEX_COST_MULTIPLIER : 1;
|
|
51
43
|
|
|
52
44
|
const compat: NonNullable<ProviderModelConfig["compat"]> = {
|
|
53
45
|
supportsDeveloperRole: meta.capabilities.developer_role,
|
|
@@ -66,10 +58,10 @@ function apiModelToProviderModel(model: NeuralwattApiModel): NeuralwattModel {
|
|
|
66
58
|
? (["text", "image"] as const)
|
|
67
59
|
: (["text"] as const),
|
|
68
60
|
cost: {
|
|
69
|
-
input: meta.pricing.input_per_million
|
|
70
|
-
output: meta.pricing.output_per_million
|
|
71
|
-
cacheRead:
|
|
72
|
-
cacheWrite:
|
|
61
|
+
input: meta.pricing.input_per_million,
|
|
62
|
+
output: meta.pricing.output_per_million,
|
|
63
|
+
cacheRead: meta.pricing.cached_input_per_million ?? 0,
|
|
64
|
+
cacheWrite: meta.pricing.cached_output_per_million ?? 0,
|
|
73
65
|
},
|
|
74
66
|
contextWindow,
|
|
75
67
|
maxTokens: resolveMaxTokens(meta.limits.max_output_tokens, contextWindow),
|
|
@@ -1,7 +1,6 @@
|
|
|
1
1
|
import type { ProviderModelConfig } from "@earendil-works/pi-coding-agent";
|
|
2
2
|
import {
|
|
3
3
|
buildNeuralwattFamily,
|
|
4
|
-
FLEX_COST_MULTIPLIER,
|
|
5
4
|
type NeuralwattModelFamily,
|
|
6
5
|
type NeuralwattVariantSpec,
|
|
7
6
|
} from "./build";
|
|
@@ -41,25 +40,23 @@ const GEMMA_4: NeuralwattModelFamily = {
|
|
|
41
40
|
},
|
|
42
41
|
};
|
|
43
42
|
|
|
44
|
-
// ZhipuAI. GLM-5.
|
|
45
|
-
//
|
|
46
|
-
|
|
47
|
-
const GLM_5_2: NeuralwattModelFamily = {
|
|
43
|
+
// ZhipuAI. GLM-5.3 has mandatory reasoning and `none` is not offered:
|
|
44
|
+
// efforts are max/high/low (default max).
|
|
45
|
+
const GLM_5_3: NeuralwattModelFamily = {
|
|
48
46
|
cost: { input: 1.45, output: 4.5, cacheRead: 0.145 },
|
|
49
47
|
vision: false,
|
|
50
48
|
reasoningMetadata: {
|
|
51
|
-
supported_efforts: ["max", "high", "
|
|
52
|
-
mandatory:
|
|
49
|
+
supported_efforts: ["max", "high", "low"],
|
|
50
|
+
mandatory: true,
|
|
53
51
|
},
|
|
54
52
|
};
|
|
55
53
|
|
|
56
|
-
// ZhipuAI. GLM-5.3
|
|
57
|
-
//
|
|
58
|
-
// 5.
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
vision: false,
|
|
54
|
+
// ZhipuAI. GLM-5.3 Flash is the small GLM-5.3 tier: vision-capable, much
|
|
55
|
+
// cheaper than the flagship, with the same mandatory max/high/low reasoning
|
|
56
|
+
// contract as GLM-5.3.
|
|
57
|
+
const GLM_5_3_FLASH: NeuralwattModelFamily = {
|
|
58
|
+
cost: { input: 0.15, output: 0.5, cacheRead: 0.03 },
|
|
59
|
+
vision: true,
|
|
63
60
|
reasoningMetadata: {
|
|
64
61
|
supported_efforts: ["max", "high", "low"],
|
|
65
62
|
mandatory: true,
|
|
@@ -99,6 +96,18 @@ const QWEN_3_6_35B: NeuralwattModelFamily = {
|
|
|
99
96
|
},
|
|
100
97
|
};
|
|
101
98
|
|
|
99
|
+
// Qwen. Qwen 3.8 27B tops out at `xhigh` (its default) and also supports
|
|
100
|
+
// `medium`, `low`, and `none`; there is no `max` effort. Reasoning is on by
|
|
101
|
+
// default but can be disabled.
|
|
102
|
+
const QWEN_3_8_27B: NeuralwattModelFamily = {
|
|
103
|
+
cost: { input: 0.45, output: 3.2, cacheRead: 0.25 },
|
|
104
|
+
vision: true,
|
|
105
|
+
reasoningMetadata: {
|
|
106
|
+
supported_efforts: ["xhigh", "medium", "low", "none"],
|
|
107
|
+
mandatory: false,
|
|
108
|
+
},
|
|
109
|
+
};
|
|
110
|
+
|
|
102
111
|
const FAMILIES: [NeuralwattModelFamily, NeuralwattVariantSpec[]][] = [
|
|
103
112
|
[
|
|
104
113
|
DEEPSEEK_V4_FLASH,
|
|
@@ -116,7 +125,7 @@ const FAMILIES: [NeuralwattModelFamily, NeuralwattVariantSpec[]][] = [
|
|
|
116
125
|
contextWindow: 1048560,
|
|
117
126
|
maxOutputTokens: 65536,
|
|
118
127
|
reasoning: true,
|
|
119
|
-
costMultiplier:
|
|
128
|
+
costMultiplier: 0.65,
|
|
120
129
|
},
|
|
121
130
|
],
|
|
122
131
|
],
|
|
@@ -133,78 +142,42 @@ const FAMILIES: [NeuralwattModelFamily, NeuralwattVariantSpec[]][] = [
|
|
|
133
142
|
],
|
|
134
143
|
],
|
|
135
144
|
[
|
|
136
|
-
|
|
145
|
+
GLM_5_3,
|
|
137
146
|
[
|
|
138
147
|
{
|
|
139
|
-
id: "glm-5.
|
|
140
|
-
name: "GLM
|
|
148
|
+
id: "glm-5.3",
|
|
149
|
+
name: "GLM 5.3",
|
|
141
150
|
contextWindow: 1048560,
|
|
142
151
|
maxOutputTokens: null,
|
|
143
152
|
reasoning: true,
|
|
144
153
|
},
|
|
145
154
|
{
|
|
146
|
-
|
|
147
|
-
|
|
148
|
-
// `reasoning_effort` re-enables thinking for that request.
|
|
149
|
-
id: "glm-5.2-fast",
|
|
150
|
-
name: "GLM-5.2 (fast)",
|
|
155
|
+
id: "glm-5.3-flex",
|
|
156
|
+
name: "GLM 5.3 (flex)",
|
|
151
157
|
contextWindow: 1048560,
|
|
152
158
|
maxOutputTokens: null,
|
|
153
159
|
reasoning: true,
|
|
160
|
+
costMultiplier: 0.65,
|
|
154
161
|
},
|
|
162
|
+
],
|
|
163
|
+
],
|
|
164
|
+
[
|
|
165
|
+
GLM_5_3_FLASH,
|
|
166
|
+
[
|
|
155
167
|
{
|
|
156
|
-
id: "glm-5.
|
|
157
|
-
name: "GLM-5.
|
|
168
|
+
id: "glm-5.3-flash",
|
|
169
|
+
name: "GLM-5.3 Flash",
|
|
158
170
|
contextWindow: 1048560,
|
|
159
171
|
maxOutputTokens: null,
|
|
160
172
|
reasoning: true,
|
|
161
|
-
costMultiplier: FLEX_COST_MULTIPLIER,
|
|
162
|
-
},
|
|
163
|
-
{
|
|
164
|
-
id: "glm-5.2-short",
|
|
165
|
-
name: "GLM-5.2 Short",
|
|
166
|
-
contextWindow: 199984,
|
|
167
|
-
maxOutputTokens: 32000,
|
|
168
|
-
reasoning: true,
|
|
169
|
-
},
|
|
170
|
-
{
|
|
171
|
-
// Short/fast: pins thinking off but keeps the parent reasoning
|
|
172
|
-
// contract, like glm-5.2-fast.
|
|
173
|
-
id: "glm-5.2-short-fast",
|
|
174
|
-
name: "GLM-5.2 (short, fast)",
|
|
175
|
-
contextWindow: 199984,
|
|
176
|
-
maxOutputTokens: 32000,
|
|
177
|
-
reasoning: true,
|
|
178
|
-
},
|
|
179
|
-
{
|
|
180
|
-
id: "glm-5.2-short-flex",
|
|
181
|
-
name: "GLM-5.2 (short, flex)",
|
|
182
|
-
contextWindow: 199984,
|
|
183
|
-
maxOutputTokens: 32000,
|
|
184
|
-
reasoning: true,
|
|
185
|
-
costMultiplier: FLEX_COST_MULTIPLIER,
|
|
186
173
|
},
|
|
187
174
|
{
|
|
188
|
-
|
|
189
|
-
|
|
190
|
-
id: "glm-5.2-short-fast-flex",
|
|
191
|
-
name: "GLM-5.2 (short, fast, flex)",
|
|
192
|
-
contextWindow: 199984,
|
|
193
|
-
maxOutputTokens: 32000,
|
|
194
|
-
reasoning: true,
|
|
195
|
-
costMultiplier: FLEX_COST_MULTIPLIER,
|
|
196
|
-
},
|
|
197
|
-
],
|
|
198
|
-
],
|
|
199
|
-
[
|
|
200
|
-
GLM_5_3,
|
|
201
|
-
[
|
|
202
|
-
{
|
|
203
|
-
id: "glm-5.3",
|
|
204
|
-
name: "GLM-5.3",
|
|
175
|
+
id: "glm-5.3-flash-flex",
|
|
176
|
+
name: "GLM-5.3 Flash (flex)",
|
|
205
177
|
contextWindow: 1048560,
|
|
206
178
|
maxOutputTokens: null,
|
|
207
179
|
reasoning: true,
|
|
180
|
+
costMultiplier: 0.65,
|
|
208
181
|
},
|
|
209
182
|
],
|
|
210
183
|
],
|
|
@@ -231,7 +204,7 @@ const FAMILIES: [NeuralwattModelFamily, NeuralwattVariantSpec[]][] = [
|
|
|
231
204
|
contextWindow: 1048560,
|
|
232
205
|
maxOutputTokens: null,
|
|
233
206
|
reasoning: true,
|
|
234
|
-
costMultiplier:
|
|
207
|
+
costMultiplier: 0.65,
|
|
235
208
|
},
|
|
236
209
|
],
|
|
237
210
|
],
|
|
@@ -260,7 +233,7 @@ const FAMILIES: [NeuralwattModelFamily, NeuralwattVariantSpec[]][] = [
|
|
|
260
233
|
contextWindow: 262128,
|
|
261
234
|
maxOutputTokens: null,
|
|
262
235
|
reasoning: true,
|
|
263
|
-
costMultiplier:
|
|
236
|
+
costMultiplier: 0.65,
|
|
264
237
|
},
|
|
265
238
|
],
|
|
266
239
|
],
|
|
@@ -281,15 +254,42 @@ const FAMILIES: [NeuralwattModelFamily, NeuralwattVariantSpec[]][] = [
|
|
|
281
254
|
maxOutputTokens: null,
|
|
282
255
|
reasoning: false,
|
|
283
256
|
},
|
|
257
|
+
{
|
|
258
|
+
id: "qwen3.6-35b-flex",
|
|
259
|
+
name: "Qwen3.6 35B (flex)",
|
|
260
|
+
contextWindow: 131056,
|
|
261
|
+
maxOutputTokens: null,
|
|
262
|
+
reasoning: true,
|
|
263
|
+
costMultiplier: 0.65,
|
|
264
|
+
},
|
|
265
|
+
],
|
|
266
|
+
],
|
|
267
|
+
[
|
|
268
|
+
QWEN_3_8_27B,
|
|
269
|
+
[
|
|
270
|
+
{
|
|
271
|
+
id: "qwen-3.8-27b",
|
|
272
|
+
name: "Qwen 3.8 27B",
|
|
273
|
+
contextWindow: 262128,
|
|
274
|
+
maxOutputTokens: 131072,
|
|
275
|
+
reasoning: true,
|
|
276
|
+
},
|
|
277
|
+
{
|
|
278
|
+
id: "qwen-3.8-27b-flex",
|
|
279
|
+
name: "Qwen 3.8 27B (flex)",
|
|
280
|
+
contextWindow: 262128,
|
|
281
|
+
maxOutputTokens: 131072,
|
|
282
|
+
reasoning: true,
|
|
283
|
+
costMultiplier: 0.65,
|
|
284
|
+
},
|
|
284
285
|
],
|
|
285
286
|
],
|
|
286
287
|
];
|
|
287
288
|
|
|
288
289
|
// `-flex` variants are the Flex tier: same model, context window, output cap,
|
|
289
|
-
// and prompt cache as the standard variant, admitted on spare capacity.
|
|
290
|
-
//
|
|
291
|
-
//
|
|
292
|
-
// `costMultiplier` rather than reflected in the catalog metadata.
|
|
290
|
+
// and prompt cache as the standard variant, admitted on spare capacity. The
|
|
291
|
+
// API lists them at discounted prices; the fallback mirrors that with
|
|
292
|
+
// `costMultiplier: 0.65` per variant.
|
|
293
293
|
// https://docs.neuralwatt.com/guides/flex-tier.md
|
|
294
294
|
|
|
295
295
|
export const NEURALWATT_MODELS: ProviderModelConfig[] = FAMILIES.flatMap(
|
|
@@ -8,10 +8,37 @@ const COOLDOWN_MS = 60 * 60 * 1000; // 60 minutes
|
|
|
8
8
|
const LOW_PCT = 25;
|
|
9
9
|
const CRITICAL_PCT = 10;
|
|
10
10
|
|
|
11
|
-
/**
|
|
12
|
-
const
|
|
13
|
-
|
|
11
|
+
/** $/kWh by plan on a monthly interval. docs.neuralwatt.com/billing/faq */
|
|
12
|
+
const OVERAGE_RATES_MONTHLY = {
|
|
13
|
+
basic: 8.5,
|
|
14
|
+
standard: 8.0,
|
|
15
|
+
pro: 7.5,
|
|
16
|
+
max: 7.0,
|
|
17
|
+
} as const;
|
|
18
|
+
/** $/kWh by plan on an annual interval. */
|
|
19
|
+
const OVERAGE_RATES_ANNUAL = {
|
|
20
|
+
basic: 7.08,
|
|
21
|
+
standard: 6.67,
|
|
22
|
+
pro: 6.25,
|
|
23
|
+
max: 5.83,
|
|
24
|
+
} as const;
|
|
25
|
+
/** Pay-as-you-go, verified on portal.neuralwatt.com/pricing. */
|
|
14
26
|
const OVERAGE_RATE_PER_KWH_UNSUBSCRIBED = 10;
|
|
27
|
+
/** Unknown plan on a subscription: fall back to the Standard monthly rate. */
|
|
28
|
+
const OVERAGE_RATE_PER_KWH_UNKNOWN_SUBSCRIBED = 8.0;
|
|
29
|
+
|
|
30
|
+
function resolveOverageRate(sub: NeuralwattQuotas["subscription"]): number {
|
|
31
|
+
if (!sub) return OVERAGE_RATE_PER_KWH_UNSUBSCRIBED;
|
|
32
|
+
const plan = sub.plan.toLowerCase();
|
|
33
|
+
const table =
|
|
34
|
+
sub.billing_interval === "year"
|
|
35
|
+
? OVERAGE_RATES_ANNUAL
|
|
36
|
+
: OVERAGE_RATES_MONTHLY;
|
|
37
|
+
return (
|
|
38
|
+
(table as Record<string, number>)[plan] ??
|
|
39
|
+
OVERAGE_RATE_PER_KWH_UNKNOWN_SUBSCRIBED
|
|
40
|
+
);
|
|
41
|
+
}
|
|
15
42
|
|
|
16
43
|
interface AlertState {
|
|
17
44
|
lastSeverity: WarningSeverity;
|
|
@@ -112,9 +139,7 @@ export function computeOverageProgress(
|
|
|
112
139
|
)
|
|
113
140
|
: quotas.usage.current_month.energy_kwh;
|
|
114
141
|
|
|
115
|
-
const rate =
|
|
116
|
-
? OVERAGE_RATE_PER_KWH_SUBSCRIBED
|
|
117
|
-
: OVERAGE_RATE_PER_KWH_UNSUBSCRIBED;
|
|
142
|
+
const rate = resolveOverageRate(quotas.subscription);
|
|
118
143
|
const costUsd = overageKwh * rate;
|
|
119
144
|
const remainingUsd = Math.max(0, capUsd - costUsd);
|
|
120
145
|
const pctRemaining = capUsd > 0 ? (remainingUsd / capUsd) * 100 : 0;
|
|
@@ -153,9 +178,12 @@ function overageWarning(progress: OverageProgress): PendingWarning {
|
|
|
153
178
|
* no subscription, cap set → overage cap progress (all kWh billable)
|
|
154
179
|
* no subscription, no cap → balance credits
|
|
155
180
|
*
|
|
156
|
-
* Overage cost is derived from kWh usage: subscribed pays
|
|
157
|
-
*
|
|
158
|
-
*
|
|
181
|
+
* Overage cost is derived from kWh usage: subscribed pays a per-plan rate
|
|
182
|
+
* ($7.00–$8.50/kWh by plan and billing interval) for kWh beyond the included
|
|
183
|
+
* quota; unsubscribed pays $10/kWh for all usage. There is no overage-spent
|
|
184
|
+
* counter in the API, so progress is computed. Note `kwh_used` is the
|
|
185
|
+
* *charged* energy — flex usage bills at 0.65× kWh — so the "kWh over" figure
|
|
186
|
+
* is billed kWh, not physical consumption.
|
|
159
187
|
*
|
|
160
188
|
* Usage totals (monthly/lifetime cost in USD) are deliberately not used as a
|
|
161
189
|
* threshold basis — they are not directly tied to the subscription's kWh quota.
|
package/package.json
CHANGED
package/src/types/models-api.ts
CHANGED
|
@@ -5,6 +5,10 @@ export interface NeuralwattApiModelPricing {
|
|
|
5
5
|
cached_output_per_million: number | null;
|
|
6
6
|
currency: string;
|
|
7
7
|
pricing_tbd: boolean;
|
|
8
|
+
/** Service tier the pricing applies to (e.g. "standard", "flex"). */
|
|
9
|
+
service_tier?: string;
|
|
10
|
+
/** Flex tier cost multiplier (e.g. 0.65); null/absent on standard pricing. */
|
|
11
|
+
flex_discount_multiplier?: number | null;
|
|
8
12
|
}
|
|
9
13
|
|
|
10
14
|
export interface NeuralwattApiModelCapabilities {
|
package/src/types/quota-api.ts
CHANGED
|
@@ -28,6 +28,10 @@ export interface NeuralwattQuotas {
|
|
|
28
28
|
total_credits_usd: number;
|
|
29
29
|
credits_used_usd: number;
|
|
30
30
|
accounting_method: string;
|
|
31
|
+
/** Legacy credit pool split (USD). Typed for fidelity, not consumed yet. */
|
|
32
|
+
legacy_credits_usd?: number;
|
|
33
|
+
/** New credit pool split (USD). Typed for fidelity, not consumed yet. */
|
|
34
|
+
new_credits_usd?: number;
|
|
31
35
|
};
|
|
32
36
|
usage: {
|
|
33
37
|
lifetime: {
|