@aliou/pi-neuralwatt 0.15.2 → 0.15.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -1
- package/extensions/provider/models/build.ts +3 -10
- package/extensions/provider/models/catalog.ts +13 -21
- package/extensions/provider/models/public-models.ts +81 -90
- package/extensions/provider/models/refresh.ts +55 -16
- package/extensions/provider/rate-limit-error.ts +1 -1
- package/extensions/quota-warnings/notifier.ts +37 -9
- package/package.json +1 -1
- package/src/types/models-api.ts +6 -0
- package/src/types/quota-api.ts +4 -0
package/README.md
CHANGED
|
@@ -144,5 +144,5 @@ This repository uses [Changesets](https://github.com/changesets/changesets) for
|
|
|
144
144
|
## Links
|
|
145
145
|
|
|
146
146
|
- [Neuralwatt](https://portal.neuralwatt.com/auth/register?ref=NW-ALIOU-Q7MF)
|
|
147
|
-
- [Neuralwatt API Docs](https://neuralwatt.com/
|
|
147
|
+
- [Neuralwatt API Docs](https://docs.neuralwatt.com/quickstart.md)
|
|
148
148
|
- [Pi Documentation](https://buildwithpi.ai/)
|
|
@@ -8,15 +8,6 @@ export type ThinkingLevelMap = NonNullable<
|
|
|
8
8
|
ProviderModelConfig["thinkingLevelMap"]
|
|
9
9
|
>;
|
|
10
10
|
|
|
11
|
-
/**
|
|
12
|
-
* Flex tier is billed at 65% of standard pricing (35% off) when the request
|
|
13
|
-
* streams. A non-streaming request to a `-flex` model silently falls back to
|
|
14
|
-
* the standard tier and the standard price.
|
|
15
|
-
*
|
|
16
|
-
* https://portal.neuralwatt.com/docs/guides/flex-tier
|
|
17
|
-
*/
|
|
18
|
-
export const FLEX_COST_MULTIPLIER = 0.65;
|
|
19
|
-
|
|
20
11
|
export interface NeuralwattCost {
|
|
21
12
|
input: number;
|
|
22
13
|
output: number;
|
|
@@ -107,12 +98,14 @@ export function buildThinkingLevelMap(
|
|
|
107
98
|
|
|
108
99
|
/**
|
|
109
100
|
* Neuralwatt reports `max_output_tokens: null` for models whose output is only
|
|
110
|
-
* bounded by the context window.
|
|
101
|
+
* bounded by the context window. Some models incorrectly report 0; treat 0
|
|
102
|
+
* like null so we never emit maxTokens: 0.
|
|
111
103
|
*/
|
|
112
104
|
export function resolveMaxTokens(
|
|
113
105
|
maxOutputTokens: number | null | undefined,
|
|
114
106
|
contextWindow: number,
|
|
115
107
|
): number {
|
|
108
|
+
if (maxOutputTokens === 0) return contextWindow;
|
|
116
109
|
return maxOutputTokens ?? contextWindow;
|
|
117
110
|
}
|
|
118
111
|
|
|
@@ -2,7 +2,6 @@ import type { ProviderModelConfig } from "@earendil-works/pi-coding-agent";
|
|
|
2
2
|
import type { NeuralwattApiModel } from "../../../src/types/models-api";
|
|
3
3
|
import {
|
|
4
4
|
buildThinkingLevelMap,
|
|
5
|
-
FLEX_COST_MULTIPLIER,
|
|
6
5
|
resolveMaxTokens,
|
|
7
6
|
type ThinkingLevelMap,
|
|
8
7
|
} from "./build";
|
|
@@ -10,12 +9,6 @@ import { NEURALWATT_MODELS } from "./public-models";
|
|
|
10
9
|
|
|
11
10
|
export type NeuralwattModel = ProviderModelConfig;
|
|
12
11
|
|
|
13
|
-
const CONTEXT_WINDOW_OVERRIDES: ReadonlyMap<string, number> = new Map([
|
|
14
|
-
["kimi-k3", 327_680],
|
|
15
|
-
["kimi-k3-fast", 327_680],
|
|
16
|
-
["kimi-k3-flex", 327_680],
|
|
17
|
-
]);
|
|
18
|
-
|
|
19
12
|
// Chat-template thinking: the API exposes a `reasoning` block, but the
|
|
20
13
|
// underlying mechanism is chat_template_kwargs, so Pi needs the mapping.
|
|
21
14
|
const COMPAT_OVERRIDES: Partial<
|
|
@@ -30,16 +23,11 @@ const COMPAT_OVERRIDES: Partial<
|
|
|
30
23
|
};
|
|
31
24
|
|
|
32
25
|
const HARDCODED_ALIASES: Record<string, string> = {
|
|
33
|
-
"zai-org/GLM-5.2-FP8": "glm-5.2",
|
|
34
26
|
"moonshotai/Kimi-K2.7-Code": "kimi-k2.7-code",
|
|
35
27
|
"Qwen/Qwen3.6-35B-A3B": "qwen3.6-35b",
|
|
36
28
|
"deepseek-ai/DeepSeek-V4-Flash": "deepseek-v4-flash",
|
|
37
29
|
};
|
|
38
30
|
|
|
39
|
-
function isFlexModelId(id: string): boolean {
|
|
40
|
-
return id.endsWith("-flex");
|
|
41
|
-
}
|
|
42
|
-
|
|
43
31
|
function isVariantId(id: string): boolean {
|
|
44
32
|
return id.includes("-fast") || id.includes("-flex") || id.includes("-short");
|
|
45
33
|
}
|
|
@@ -52,8 +40,6 @@ function apiModelToProviderModel(model: NeuralwattApiModel): NeuralwattModel {
|
|
|
52
40
|
);
|
|
53
41
|
|
|
54
42
|
const reasoning = meta.capabilities.reasoning;
|
|
55
|
-
// Flex variants are billed at 0.65x when streaming (35% off).
|
|
56
|
-
const multiplier = isFlexModelId(model.id) ? FLEX_COST_MULTIPLIER : 1;
|
|
57
43
|
|
|
58
44
|
const compat: NonNullable<ProviderModelConfig["compat"]> = {
|
|
59
45
|
supportsDeveloperRole: meta.capabilities.developer_role,
|
|
@@ -62,8 +48,7 @@ function apiModelToProviderModel(model: NeuralwattApiModel): NeuralwattModel {
|
|
|
62
48
|
if (reasoning) compat.requiresReasoningContentOnAssistantMessages = true;
|
|
63
49
|
Object.assign(compat, COMPAT_OVERRIDES[model.id]);
|
|
64
50
|
|
|
65
|
-
const contextWindow =
|
|
66
|
-
CONTEXT_WINDOW_OVERRIDES.get(model.id) ?? model.max_model_len;
|
|
51
|
+
const contextWindow = model.max_model_len;
|
|
67
52
|
|
|
68
53
|
const result: NeuralwattModel = {
|
|
69
54
|
id: model.id,
|
|
@@ -73,10 +58,10 @@ function apiModelToProviderModel(model: NeuralwattApiModel): NeuralwattModel {
|
|
|
73
58
|
? (["text", "image"] as const)
|
|
74
59
|
: (["text"] as const),
|
|
75
60
|
cost: {
|
|
76
|
-
input: meta.pricing.input_per_million
|
|
77
|
-
output: meta.pricing.output_per_million
|
|
78
|
-
cacheRead:
|
|
79
|
-
cacheWrite:
|
|
61
|
+
input: meta.pricing.input_per_million,
|
|
62
|
+
output: meta.pricing.output_per_million,
|
|
63
|
+
cacheRead: meta.pricing.cached_input_per_million ?? 0,
|
|
64
|
+
cacheWrite: meta.pricing.cached_output_per_million ?? 0,
|
|
80
65
|
},
|
|
81
66
|
contextWindow,
|
|
82
67
|
maxTokens: resolveMaxTokens(meta.limits.max_output_tokens, contextWindow),
|
|
@@ -135,7 +120,14 @@ export function buildNeuralwattProviderModelsFromApi(
|
|
|
135
120
|
const models = apiModels
|
|
136
121
|
.filter(
|
|
137
122
|
(m) =>
|
|
138
|
-
m.metadata &&
|
|
123
|
+
m.metadata &&
|
|
124
|
+
!m.metadata.deprecated &&
|
|
125
|
+
!m.metadata.pricing.pricing_tbd &&
|
|
126
|
+
// Exclude non-chat models (e.g. embeddings) by task
|
|
127
|
+
!(
|
|
128
|
+
m.metadata.capabilities.task &&
|
|
129
|
+
!["chat", "completions"].includes(m.metadata.capabilities.task)
|
|
130
|
+
),
|
|
139
131
|
)
|
|
140
132
|
.map(apiModelToProviderModel);
|
|
141
133
|
return [...models, ...buildAliases(models, apiModels)];
|
|
@@ -1,7 +1,6 @@
|
|
|
1
1
|
import type { ProviderModelConfig } from "@earendil-works/pi-coding-agent";
|
|
2
2
|
import {
|
|
3
3
|
buildNeuralwattFamily,
|
|
4
|
-
FLEX_COST_MULTIPLIER,
|
|
5
4
|
type NeuralwattModelFamily,
|
|
6
5
|
type NeuralwattVariantSpec,
|
|
7
6
|
} from "./build";
|
|
@@ -31,7 +30,7 @@ const DEEPSEEK_V4_FLASH: NeuralwattModelFamily = {
|
|
|
31
30
|
// `max` and `none`; every non-`none` request resolves to `max` upstream.
|
|
32
31
|
// It does not reason by default (`default_enabled: false`), but the model
|
|
33
32
|
// can produce reasoning traces when asked. See
|
|
34
|
-
// https://
|
|
33
|
+
// https://docs.neuralwatt.com/api/chat-completions.md
|
|
35
34
|
const GEMMA_4: NeuralwattModelFamily = {
|
|
36
35
|
cost: { input: 0.144, output: 0.42, cacheRead: 0.0144 },
|
|
37
36
|
vision: true,
|
|
@@ -41,25 +40,23 @@ const GEMMA_4: NeuralwattModelFamily = {
|
|
|
41
40
|
},
|
|
42
41
|
};
|
|
43
42
|
|
|
44
|
-
// ZhipuAI. GLM-5.
|
|
45
|
-
//
|
|
46
|
-
|
|
47
|
-
const GLM_5_2: NeuralwattModelFamily = {
|
|
43
|
+
// ZhipuAI. GLM-5.3 has mandatory reasoning and `none` is not offered:
|
|
44
|
+
// efforts are max/high/low (default max).
|
|
45
|
+
const GLM_5_3: NeuralwattModelFamily = {
|
|
48
46
|
cost: { input: 1.45, output: 4.5, cacheRead: 0.145 },
|
|
49
47
|
vision: false,
|
|
50
48
|
reasoningMetadata: {
|
|
51
|
-
supported_efforts: ["max", "high", "
|
|
52
|
-
mandatory:
|
|
49
|
+
supported_efforts: ["max", "high", "low"],
|
|
50
|
+
mandatory: true,
|
|
53
51
|
},
|
|
54
52
|
};
|
|
55
53
|
|
|
56
|
-
// ZhipuAI. GLM-5.3
|
|
57
|
-
//
|
|
58
|
-
// 5.
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
vision: false,
|
|
54
|
+
// ZhipuAI. GLM-5.3 Flash is the small GLM-5.3 tier: vision-capable, much
|
|
55
|
+
// cheaper than the flagship, with the same mandatory max/high/low reasoning
|
|
56
|
+
// contract as GLM-5.3.
|
|
57
|
+
const GLM_5_3_FLASH: NeuralwattModelFamily = {
|
|
58
|
+
cost: { input: 0.15, output: 0.5, cacheRead: 0.03 },
|
|
59
|
+
vision: true,
|
|
63
60
|
reasoningMetadata: {
|
|
64
61
|
supported_efforts: ["max", "high", "low"],
|
|
65
62
|
mandatory: true,
|
|
@@ -99,6 +96,18 @@ const QWEN_3_6_35B: NeuralwattModelFamily = {
|
|
|
99
96
|
},
|
|
100
97
|
};
|
|
101
98
|
|
|
99
|
+
// Qwen. Qwen 3.8 27B tops out at `xhigh` (its default) and also supports
|
|
100
|
+
// `medium`, `low`, and `none`; there is no `max` effort. Reasoning is on by
|
|
101
|
+
// default but can be disabled.
|
|
102
|
+
const QWEN_3_8_27B: NeuralwattModelFamily = {
|
|
103
|
+
cost: { input: 0.45, output: 3.2, cacheRead: 0.25 },
|
|
104
|
+
vision: true,
|
|
105
|
+
reasoningMetadata: {
|
|
106
|
+
supported_efforts: ["xhigh", "medium", "low", "none"],
|
|
107
|
+
mandatory: false,
|
|
108
|
+
},
|
|
109
|
+
};
|
|
110
|
+
|
|
102
111
|
const FAMILIES: [NeuralwattModelFamily, NeuralwattVariantSpec[]][] = [
|
|
103
112
|
[
|
|
104
113
|
DEEPSEEK_V4_FLASH,
|
|
@@ -116,7 +125,7 @@ const FAMILIES: [NeuralwattModelFamily, NeuralwattVariantSpec[]][] = [
|
|
|
116
125
|
contextWindow: 1048560,
|
|
117
126
|
maxOutputTokens: 65536,
|
|
118
127
|
reasoning: true,
|
|
119
|
-
costMultiplier:
|
|
128
|
+
costMultiplier: 0.65,
|
|
120
129
|
},
|
|
121
130
|
],
|
|
122
131
|
],
|
|
@@ -133,114 +142,69 @@ const FAMILIES: [NeuralwattModelFamily, NeuralwattVariantSpec[]][] = [
|
|
|
133
142
|
],
|
|
134
143
|
],
|
|
135
144
|
[
|
|
136
|
-
|
|
145
|
+
GLM_5_3,
|
|
137
146
|
[
|
|
138
147
|
{
|
|
139
|
-
id: "glm-5.
|
|
140
|
-
name: "GLM
|
|
148
|
+
id: "glm-5.3",
|
|
149
|
+
name: "GLM 5.3",
|
|
141
150
|
contextWindow: 1048560,
|
|
142
151
|
maxOutputTokens: null,
|
|
143
152
|
reasoning: true,
|
|
144
153
|
},
|
|
145
154
|
{
|
|
146
|
-
|
|
147
|
-
|
|
148
|
-
// `reasoning_effort` re-enables thinking for that request.
|
|
149
|
-
id: "glm-5.2-fast",
|
|
150
|
-
name: "GLM-5.2 (fast)",
|
|
155
|
+
id: "glm-5.3-flex",
|
|
156
|
+
name: "GLM 5.3 (flex)",
|
|
151
157
|
contextWindow: 1048560,
|
|
152
158
|
maxOutputTokens: null,
|
|
153
159
|
reasoning: true,
|
|
160
|
+
costMultiplier: 0.65,
|
|
154
161
|
},
|
|
162
|
+
],
|
|
163
|
+
],
|
|
164
|
+
[
|
|
165
|
+
GLM_5_3_FLASH,
|
|
166
|
+
[
|
|
155
167
|
{
|
|
156
|
-
id: "glm-5.
|
|
157
|
-
name: "GLM-5.
|
|
168
|
+
id: "glm-5.3-flash",
|
|
169
|
+
name: "GLM-5.3 Flash",
|
|
158
170
|
contextWindow: 1048560,
|
|
159
171
|
maxOutputTokens: null,
|
|
160
172
|
reasoning: true,
|
|
161
|
-
costMultiplier: FLEX_COST_MULTIPLIER,
|
|
162
|
-
},
|
|
163
|
-
{
|
|
164
|
-
id: "glm-5.2-short",
|
|
165
|
-
name: "GLM-5.2 Short",
|
|
166
|
-
contextWindow: 199984,
|
|
167
|
-
maxOutputTokens: 32000,
|
|
168
|
-
reasoning: true,
|
|
169
|
-
},
|
|
170
|
-
{
|
|
171
|
-
// Short/fast: pins thinking off but keeps the parent reasoning
|
|
172
|
-
// contract, like glm-5.2-fast.
|
|
173
|
-
id: "glm-5.2-short-fast",
|
|
174
|
-
name: "GLM-5.2 (short, fast)",
|
|
175
|
-
contextWindow: 199984,
|
|
176
|
-
maxOutputTokens: 32000,
|
|
177
|
-
reasoning: true,
|
|
178
|
-
},
|
|
179
|
-
{
|
|
180
|
-
id: "glm-5.2-short-flex",
|
|
181
|
-
name: "GLM-5.2 (short, flex)",
|
|
182
|
-
contextWindow: 199984,
|
|
183
|
-
maxOutputTokens: 32000,
|
|
184
|
-
reasoning: true,
|
|
185
|
-
costMultiplier: FLEX_COST_MULTIPLIER,
|
|
186
|
-
},
|
|
187
|
-
{
|
|
188
|
-
// Short/fast/flex: pins thinking off but keeps the parent reasoning
|
|
189
|
-
// contract, like glm-5.2-fast.
|
|
190
|
-
id: "glm-5.2-short-fast-flex",
|
|
191
|
-
name: "GLM-5.2 (short, fast, flex)",
|
|
192
|
-
contextWindow: 199984,
|
|
193
|
-
maxOutputTokens: 32000,
|
|
194
|
-
reasoning: true,
|
|
195
|
-
costMultiplier: FLEX_COST_MULTIPLIER,
|
|
196
173
|
},
|
|
197
|
-
],
|
|
198
|
-
],
|
|
199
|
-
[
|
|
200
|
-
GLM_5_3,
|
|
201
|
-
[
|
|
202
174
|
{
|
|
203
|
-
id: "glm-5.3",
|
|
204
|
-
name: "GLM-5.3",
|
|
175
|
+
id: "glm-5.3-flash-flex",
|
|
176
|
+
name: "GLM-5.3 Flash (flex)",
|
|
205
177
|
contextWindow: 1048560,
|
|
206
178
|
maxOutputTokens: null,
|
|
207
179
|
reasoning: true,
|
|
180
|
+
costMultiplier: 0.65,
|
|
208
181
|
},
|
|
209
182
|
],
|
|
210
183
|
],
|
|
211
|
-
// The kimi-k3 endpoint rejects anything above 327,680 total tokens with
|
|
212
|
-
// `400: max_completion_tokens is too large … supports at most 327680
|
|
213
|
-
// completion tokens` (verified at runtime), even though the API advertises
|
|
214
|
-
// `max_model_len: 1048560` with a null output cap for the whole family.
|
|
215
|
-
// The -fast/-flex endpoints don't enforce any cap server-side yet (they
|
|
216
|
-
// accept max_completion_tokens beyond the advertised window), but they are
|
|
217
|
-
// the same K3 deployment and are expected to share the 327,680 limit, so
|
|
218
|
-
// all three variants are pinned to it. The drift check in models.test.ts
|
|
219
|
-
// whitelists this divergence via CONTEXT_WINDOW_OVERRIDES.
|
|
220
184
|
[
|
|
221
185
|
KIMI_K3,
|
|
222
186
|
[
|
|
223
187
|
{
|
|
224
188
|
id: "kimi-k3",
|
|
225
189
|
name: "Kimi K3",
|
|
226
|
-
contextWindow:
|
|
227
|
-
maxOutputTokens:
|
|
190
|
+
contextWindow: 1048560,
|
|
191
|
+
maxOutputTokens: null,
|
|
228
192
|
reasoning: true,
|
|
229
193
|
},
|
|
230
194
|
{
|
|
231
195
|
id: "kimi-k3-fast",
|
|
232
196
|
name: "Kimi K3 Fast",
|
|
233
|
-
contextWindow:
|
|
234
|
-
maxOutputTokens:
|
|
197
|
+
contextWindow: 1048560,
|
|
198
|
+
maxOutputTokens: null,
|
|
235
199
|
reasoning: false,
|
|
236
200
|
},
|
|
237
201
|
{
|
|
238
202
|
id: "kimi-k3-flex",
|
|
239
203
|
name: "Kimi K3 (flex)",
|
|
240
|
-
contextWindow:
|
|
241
|
-
maxOutputTokens:
|
|
204
|
+
contextWindow: 1048560,
|
|
205
|
+
maxOutputTokens: null,
|
|
242
206
|
reasoning: true,
|
|
243
|
-
costMultiplier:
|
|
207
|
+
costMultiplier: 0.65,
|
|
244
208
|
},
|
|
245
209
|
],
|
|
246
210
|
],
|
|
@@ -269,7 +233,7 @@ const FAMILIES: [NeuralwattModelFamily, NeuralwattVariantSpec[]][] = [
|
|
|
269
233
|
contextWindow: 262128,
|
|
270
234
|
maxOutputTokens: null,
|
|
271
235
|
reasoning: true,
|
|
272
|
-
costMultiplier:
|
|
236
|
+
costMultiplier: 0.65,
|
|
273
237
|
},
|
|
274
238
|
],
|
|
275
239
|
],
|
|
@@ -290,16 +254,43 @@ const FAMILIES: [NeuralwattModelFamily, NeuralwattVariantSpec[]][] = [
|
|
|
290
254
|
maxOutputTokens: null,
|
|
291
255
|
reasoning: false,
|
|
292
256
|
},
|
|
257
|
+
{
|
|
258
|
+
id: "qwen3.6-35b-flex",
|
|
259
|
+
name: "Qwen3.6 35B (flex)",
|
|
260
|
+
contextWindow: 131056,
|
|
261
|
+
maxOutputTokens: null,
|
|
262
|
+
reasoning: true,
|
|
263
|
+
costMultiplier: 0.65,
|
|
264
|
+
},
|
|
265
|
+
],
|
|
266
|
+
],
|
|
267
|
+
[
|
|
268
|
+
QWEN_3_8_27B,
|
|
269
|
+
[
|
|
270
|
+
{
|
|
271
|
+
id: "qwen-3.8-27b",
|
|
272
|
+
name: "Qwen 3.8 27B",
|
|
273
|
+
contextWindow: 262128,
|
|
274
|
+
maxOutputTokens: 131072,
|
|
275
|
+
reasoning: true,
|
|
276
|
+
},
|
|
277
|
+
{
|
|
278
|
+
id: "qwen-3.8-27b-flex",
|
|
279
|
+
name: "Qwen 3.8 27B (flex)",
|
|
280
|
+
contextWindow: 262128,
|
|
281
|
+
maxOutputTokens: 131072,
|
|
282
|
+
reasoning: true,
|
|
283
|
+
costMultiplier: 0.65,
|
|
284
|
+
},
|
|
293
285
|
],
|
|
294
286
|
],
|
|
295
287
|
];
|
|
296
288
|
|
|
297
289
|
// `-flex` variants are the Flex tier: same model, context window, output cap,
|
|
298
|
-
// and prompt cache as the standard variant, admitted on spare capacity.
|
|
299
|
-
//
|
|
300
|
-
//
|
|
301
|
-
//
|
|
302
|
-
// https://portal.neuralwatt.com/docs/guides/flex-tier
|
|
290
|
+
// and prompt cache as the standard variant, admitted on spare capacity. The
|
|
291
|
+
// API lists them at discounted prices; the fallback mirrors that with
|
|
292
|
+
// `costMultiplier: 0.65` per variant.
|
|
293
|
+
// https://docs.neuralwatt.com/guides/flex-tier.md
|
|
303
294
|
|
|
304
295
|
export const NEURALWATT_MODELS: ProviderModelConfig[] = FAMILIES.flatMap(
|
|
305
296
|
([family, variants]) => buildNeuralwattFamily(family, variants),
|
|
@@ -12,6 +12,38 @@ import type {
|
|
|
12
12
|
|
|
13
13
|
export const MODEL_STORE_TTL_MS = 4 * 60 * 60 * 1000;
|
|
14
14
|
|
|
15
|
+
/**
|
|
16
|
+
* Scope a store entry applies to. The public catalog is a subset of any
|
|
17
|
+
* key-scoped catalog (preview, grant-gated, private models), so an entry
|
|
18
|
+
* stamped "public" must not shadow a keyed refresh — and a key-scoped entry
|
|
19
|
+
* must not be replayed for an anonymous user. Matches the anonymous-key
|
|
20
|
+
* convention in src/lib/neuralwatt-api.ts (authHeaders).
|
|
21
|
+
*/
|
|
22
|
+
const CATALOG_SCOPE_VERSION = "v1";
|
|
23
|
+
type CatalogScope = "public" | "key";
|
|
24
|
+
|
|
25
|
+
type ScopedModelsStoreEntry = ModelsStoreEntry & { catalogKey?: string };
|
|
26
|
+
|
|
27
|
+
function catalogScope(apiKey: string | undefined): CatalogScope {
|
|
28
|
+
return apiKey !== undefined && apiKey !== "" && apiKey !== "-"
|
|
29
|
+
? "key"
|
|
30
|
+
: "public";
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
function storedCatalogKey(entry: ScopedModelsStoreEntry): string | undefined {
|
|
34
|
+
return entry.catalogKey;
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
function catalogKeyMatches(
|
|
38
|
+
entry: ScopedModelsStoreEntry | undefined,
|
|
39
|
+
scope: CatalogScope,
|
|
40
|
+
): boolean {
|
|
41
|
+
return (
|
|
42
|
+
entry !== undefined &&
|
|
43
|
+
storedCatalogKey(entry) === `${scope} ${CATALOG_SCOPE_VERSION}`
|
|
44
|
+
);
|
|
45
|
+
}
|
|
46
|
+
|
|
15
47
|
export type FetchNeuralwattApiModels = (
|
|
16
48
|
apiKey: string | undefined,
|
|
17
49
|
signal?: AbortSignal,
|
|
@@ -25,6 +57,13 @@ function isFreshStoreEntry(
|
|
|
25
57
|
return Date.now() - checkedAt < MODEL_STORE_TTL_MS;
|
|
26
58
|
}
|
|
27
59
|
|
|
60
|
+
function isUsableStoreEntry(
|
|
61
|
+
entry: Readonly<ModelsStoreEntry> | undefined,
|
|
62
|
+
scope: CatalogScope,
|
|
63
|
+
): entry is ModelsStoreEntry {
|
|
64
|
+
return isFreshStoreEntry(entry) && catalogKeyMatches(entry, scope);
|
|
65
|
+
}
|
|
66
|
+
|
|
28
67
|
export function createNeuralwattRefreshModels(
|
|
29
68
|
staticModels: ReturnType<typeof buildNeuralwattProviderModels>,
|
|
30
69
|
fetchApiModels: FetchNeuralwattApiModels,
|
|
@@ -35,29 +74,29 @@ export function createNeuralwattRefreshModels(
|
|
|
35
74
|
context.signal.throwIfAborted();
|
|
36
75
|
const fallback = buildFromStore(staticModels);
|
|
37
76
|
try {
|
|
38
|
-
if (!context.allowNetwork) {
|
|
39
|
-
return context.stored
|
|
40
|
-
? buildFromStore(context.stored.models)
|
|
41
|
-
: fallback;
|
|
42
|
-
}
|
|
43
|
-
if (!context.force && isFreshStoreEntry(context.stored)) {
|
|
44
|
-
return buildFromStore(context.stored.models);
|
|
45
|
-
}
|
|
46
77
|
const apiKey =
|
|
47
78
|
context.credential?.type === "api_key"
|
|
48
79
|
? context.credential.key
|
|
49
80
|
: undefined;
|
|
81
|
+
const scope = catalogScope(apiKey);
|
|
82
|
+
const stored = context.stored as ScopedModelsStoreEntry | undefined;
|
|
83
|
+
if (!context.allowNetwork) {
|
|
84
|
+
return stored !== undefined && catalogKeyMatches(stored, scope)
|
|
85
|
+
? buildFromStore(stored.models)
|
|
86
|
+
: fallback;
|
|
87
|
+
}
|
|
88
|
+
if (!context.force && isUsableStoreEntry(stored, scope)) {
|
|
89
|
+
return buildFromStore(stored.models);
|
|
90
|
+
}
|
|
50
91
|
const apiModels = await fetchApiModels(apiKey, context.signal);
|
|
51
92
|
context.signal.throwIfAborted();
|
|
52
93
|
const models = buildFromApi(apiModels);
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
})
|
|
60
|
-
.catch(() => undefined);
|
|
94
|
+
const entry: ScopedModelsStoreEntry = {
|
|
95
|
+
models: models as unknown as ModelsStoreEntry["models"],
|
|
96
|
+
checkedAt: Date.now(),
|
|
97
|
+
catalogKey: `${scope} ${CATALOG_SCOPE_VERSION}`,
|
|
98
|
+
};
|
|
99
|
+
await context.publish({ persist: entry }).catch(() => undefined);
|
|
61
100
|
context.signal.throwIfAborted();
|
|
62
101
|
return models;
|
|
63
102
|
} catch (error) {
|
|
@@ -12,7 +12,7 @@ interface AssistantErrorLike {
|
|
|
12
12
|
* layer. Each sets unique headers so the client can tell which layer
|
|
13
13
|
* triggered the rejection.
|
|
14
14
|
*
|
|
15
|
-
* @see https://
|
|
15
|
+
* @see https://docs.neuralwatt.com/guides/rate-limits.md
|
|
16
16
|
*/
|
|
17
17
|
export interface NeuralwattRateLimitInfo {
|
|
18
18
|
/** Which rate-limit layer triggered the 429 */
|
|
@@ -8,10 +8,37 @@ const COOLDOWN_MS = 60 * 60 * 1000; // 60 minutes
|
|
|
8
8
|
const LOW_PCT = 25;
|
|
9
9
|
const CRITICAL_PCT = 10;
|
|
10
10
|
|
|
11
|
-
/**
|
|
12
|
-
const
|
|
13
|
-
|
|
11
|
+
/** $/kWh by plan on a monthly interval. docs.neuralwatt.com/billing/faq */
|
|
12
|
+
const OVERAGE_RATES_MONTHLY = {
|
|
13
|
+
basic: 8.5,
|
|
14
|
+
standard: 8.0,
|
|
15
|
+
pro: 7.5,
|
|
16
|
+
max: 7.0,
|
|
17
|
+
} as const;
|
|
18
|
+
/** $/kWh by plan on an annual interval. */
|
|
19
|
+
const OVERAGE_RATES_ANNUAL = {
|
|
20
|
+
basic: 7.08,
|
|
21
|
+
standard: 6.67,
|
|
22
|
+
pro: 6.25,
|
|
23
|
+
max: 5.83,
|
|
24
|
+
} as const;
|
|
25
|
+
/** Pay-as-you-go, verified on portal.neuralwatt.com/pricing. */
|
|
14
26
|
const OVERAGE_RATE_PER_KWH_UNSUBSCRIBED = 10;
|
|
27
|
+
/** Unknown plan on a subscription: fall back to the Standard monthly rate. */
|
|
28
|
+
const OVERAGE_RATE_PER_KWH_UNKNOWN_SUBSCRIBED = 8.0;
|
|
29
|
+
|
|
30
|
+
function resolveOverageRate(sub: NeuralwattQuotas["subscription"]): number {
|
|
31
|
+
if (!sub) return OVERAGE_RATE_PER_KWH_UNSUBSCRIBED;
|
|
32
|
+
const plan = sub.plan.toLowerCase();
|
|
33
|
+
const table =
|
|
34
|
+
sub.billing_interval === "year"
|
|
35
|
+
? OVERAGE_RATES_ANNUAL
|
|
36
|
+
: OVERAGE_RATES_MONTHLY;
|
|
37
|
+
return (
|
|
38
|
+
(table as Record<string, number>)[plan] ??
|
|
39
|
+
OVERAGE_RATE_PER_KWH_UNKNOWN_SUBSCRIBED
|
|
40
|
+
);
|
|
41
|
+
}
|
|
15
42
|
|
|
16
43
|
interface AlertState {
|
|
17
44
|
lastSeverity: WarningSeverity;
|
|
@@ -112,9 +139,7 @@ export function computeOverageProgress(
|
|
|
112
139
|
)
|
|
113
140
|
: quotas.usage.current_month.energy_kwh;
|
|
114
141
|
|
|
115
|
-
const rate =
|
|
116
|
-
? OVERAGE_RATE_PER_KWH_SUBSCRIBED
|
|
117
|
-
: OVERAGE_RATE_PER_KWH_UNSUBSCRIBED;
|
|
142
|
+
const rate = resolveOverageRate(quotas.subscription);
|
|
118
143
|
const costUsd = overageKwh * rate;
|
|
119
144
|
const remainingUsd = Math.max(0, capUsd - costUsd);
|
|
120
145
|
const pctRemaining = capUsd > 0 ? (remainingUsd / capUsd) * 100 : 0;
|
|
@@ -153,9 +178,12 @@ function overageWarning(progress: OverageProgress): PendingWarning {
|
|
|
153
178
|
* no subscription, cap set → overage cap progress (all kWh billable)
|
|
154
179
|
* no subscription, no cap → balance credits
|
|
155
180
|
*
|
|
156
|
-
* Overage cost is derived from kWh usage: subscribed pays
|
|
157
|
-
*
|
|
158
|
-
*
|
|
181
|
+
* Overage cost is derived from kWh usage: subscribed pays a per-plan rate
|
|
182
|
+
* ($7.00–$8.50/kWh by plan and billing interval) for kWh beyond the included
|
|
183
|
+
* quota; unsubscribed pays $10/kWh for all usage. There is no overage-spent
|
|
184
|
+
* counter in the API, so progress is computed. Note `kwh_used` is the
|
|
185
|
+
* *charged* energy — flex usage bills at 0.65× kWh — so the "kWh over" figure
|
|
186
|
+
* is billed kWh, not physical consumption.
|
|
159
187
|
*
|
|
160
188
|
* Usage totals (monthly/lifetime cost in USD) are deliberately not used as a
|
|
161
189
|
* threshold basis — they are not directly tied to the subscription's kWh quota.
|
package/package.json
CHANGED
package/src/types/models-api.ts
CHANGED
|
@@ -5,6 +5,10 @@ export interface NeuralwattApiModelPricing {
|
|
|
5
5
|
cached_output_per_million: number | null;
|
|
6
6
|
currency: string;
|
|
7
7
|
pricing_tbd: boolean;
|
|
8
|
+
/** Service tier the pricing applies to (e.g. "standard", "flex"). */
|
|
9
|
+
service_tier?: string;
|
|
10
|
+
/** Flex tier cost multiplier (e.g. 0.65); null/absent on standard pricing. */
|
|
11
|
+
flex_discount_multiplier?: number | null;
|
|
8
12
|
}
|
|
9
13
|
|
|
10
14
|
export interface NeuralwattApiModelCapabilities {
|
|
@@ -16,6 +20,8 @@ export interface NeuralwattApiModelCapabilities {
|
|
|
16
20
|
streaming: boolean;
|
|
17
21
|
system_role: boolean;
|
|
18
22
|
developer_role: boolean;
|
|
23
|
+
task?: string;
|
|
24
|
+
embedding_dimensions?: number;
|
|
19
25
|
}
|
|
20
26
|
|
|
21
27
|
/**
|
package/src/types/quota-api.ts
CHANGED
|
@@ -28,6 +28,10 @@ export interface NeuralwattQuotas {
|
|
|
28
28
|
total_credits_usd: number;
|
|
29
29
|
credits_used_usd: number;
|
|
30
30
|
accounting_method: string;
|
|
31
|
+
/** Legacy credit pool split (USD). Typed for fidelity, not consumed yet. */
|
|
32
|
+
legacy_credits_usd?: number;
|
|
33
|
+
/** New credit pool split (USD). Typed for fidelity, not consumed yet. */
|
|
34
|
+
new_credits_usd?: number;
|
|
31
35
|
};
|
|
32
36
|
usage: {
|
|
33
37
|
lifetime: {
|