@aliou/pi-neuralwatt 0.15.1 → 0.15.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -1
- package/extensions/provider/models/build.ts +4 -2
- package/extensions/provider/models/catalog.ts +9 -9
- package/extensions/provider/models/public-models.ts +33 -17
- package/extensions/provider/models/refresh.ts +55 -16
- package/extensions/provider/rate-limit-error.ts +1 -1
- package/package.json +3 -1
- package/src/types/models-api.ts +2 -0
package/README.md
CHANGED
|
@@ -144,5 +144,5 @@ This repository uses [Changesets](https://github.com/changesets/changesets) for
|
|
|
144
144
|
## Links
|
|
145
145
|
|
|
146
146
|
- [Neuralwatt](https://portal.neuralwatt.com/auth/register?ref=NW-ALIOU-Q7MF)
|
|
147
|
-
- [Neuralwatt API Docs](https://neuralwatt.com/
|
|
147
|
+
- [Neuralwatt API Docs](https://docs.neuralwatt.com/quickstart.md)
|
|
148
148
|
- [Pi Documentation](https://buildwithpi.ai/)
|
|
@@ -13,7 +13,7 @@ export type ThinkingLevelMap = NonNullable<
|
|
|
13
13
|
* streams. A non-streaming request to a `-flex` model silently falls back to
|
|
14
14
|
* the standard tier and the standard price.
|
|
15
15
|
*
|
|
16
|
-
* https://
|
|
16
|
+
* https://docs.neuralwatt.com/guides/flex-tier.md
|
|
17
17
|
*/
|
|
18
18
|
export const FLEX_COST_MULTIPLIER = 0.65;
|
|
19
19
|
|
|
@@ -107,12 +107,14 @@ export function buildThinkingLevelMap(
|
|
|
107
107
|
|
|
108
108
|
/**
|
|
109
109
|
* Neuralwatt reports `max_output_tokens: null` for models whose output is only
|
|
110
|
-
* bounded by the context window.
|
|
110
|
+
* bounded by the context window. Some models incorrectly report 0; treat 0
|
|
111
|
+
* like null so we never emit maxTokens: 0.
|
|
111
112
|
*/
|
|
112
113
|
export function resolveMaxTokens(
|
|
113
114
|
maxOutputTokens: number | null | undefined,
|
|
114
115
|
contextWindow: number,
|
|
115
116
|
): number {
|
|
117
|
+
if (maxOutputTokens === 0) return contextWindow;
|
|
116
118
|
return maxOutputTokens ?? contextWindow;
|
|
117
119
|
}
|
|
118
120
|
|
|
@@ -10,12 +10,6 @@ import { NEURALWATT_MODELS } from "./public-models";
|
|
|
10
10
|
|
|
11
11
|
export type NeuralwattModel = ProviderModelConfig;
|
|
12
12
|
|
|
13
|
-
const CONTEXT_WINDOW_OVERRIDES: ReadonlyMap<string, number> = new Map([
|
|
14
|
-
["kimi-k3", 327_680],
|
|
15
|
-
["kimi-k3-fast", 327_680],
|
|
16
|
-
["kimi-k3-flex", 327_680],
|
|
17
|
-
]);
|
|
18
|
-
|
|
19
13
|
// Chat-template thinking: the API exposes a `reasoning` block, but the
|
|
20
14
|
// underlying mechanism is chat_template_kwargs, so Pi needs the mapping.
|
|
21
15
|
const COMPAT_OVERRIDES: Partial<
|
|
@@ -62,8 +56,7 @@ function apiModelToProviderModel(model: NeuralwattApiModel): NeuralwattModel {
|
|
|
62
56
|
if (reasoning) compat.requiresReasoningContentOnAssistantMessages = true;
|
|
63
57
|
Object.assign(compat, COMPAT_OVERRIDES[model.id]);
|
|
64
58
|
|
|
65
|
-
const contextWindow =
|
|
66
|
-
CONTEXT_WINDOW_OVERRIDES.get(model.id) ?? model.max_model_len;
|
|
59
|
+
const contextWindow = model.max_model_len;
|
|
67
60
|
|
|
68
61
|
const result: NeuralwattModel = {
|
|
69
62
|
id: model.id,
|
|
@@ -135,7 +128,14 @@ export function buildNeuralwattProviderModelsFromApi(
|
|
|
135
128
|
const models = apiModels
|
|
136
129
|
.filter(
|
|
137
130
|
(m) =>
|
|
138
|
-
m.metadata &&
|
|
131
|
+
m.metadata &&
|
|
132
|
+
!m.metadata.deprecated &&
|
|
133
|
+
!m.metadata.pricing.pricing_tbd &&
|
|
134
|
+
// Exclude non-chat models (e.g. embeddings) by task
|
|
135
|
+
!(
|
|
136
|
+
m.metadata.capabilities.task &&
|
|
137
|
+
!["chat", "completions"].includes(m.metadata.capabilities.task)
|
|
138
|
+
),
|
|
139
139
|
)
|
|
140
140
|
.map(apiModelToProviderModel);
|
|
141
141
|
return [...models, ...buildAliases(models, apiModels)];
|
|
@@ -31,7 +31,7 @@ const DEEPSEEK_V4_FLASH: NeuralwattModelFamily = {
|
|
|
31
31
|
// `max` and `none`; every non-`none` request resolves to `max` upstream.
|
|
32
32
|
// It does not reason by default (`default_enabled: false`), but the model
|
|
33
33
|
// can produce reasoning traces when asked. See
|
|
34
|
-
// https://
|
|
34
|
+
// https://docs.neuralwatt.com/api/chat-completions.md
|
|
35
35
|
const GEMMA_4: NeuralwattModelFamily = {
|
|
36
36
|
cost: { input: 0.144, output: 0.42, cacheRead: 0.0144 },
|
|
37
37
|
vision: true,
|
|
@@ -53,6 +53,19 @@ const GLM_5_2: NeuralwattModelFamily = {
|
|
|
53
53
|
},
|
|
54
54
|
};
|
|
55
55
|
|
|
56
|
+
// ZhipuAI. GLM-5.3 ships as a GLM-5.2 weight swap in gated preview, with
|
|
57
|
+
// GLM-5.2 pricing parity (per the API metadata; review at launch). Unlike
|
|
58
|
+
// 5.2, reasoning is mandatory and `none` is not offered: efforts are
|
|
59
|
+
// max/high/low (default max).
|
|
60
|
+
const GLM_5_3: NeuralwattModelFamily = {
|
|
61
|
+
cost: { input: 1.45, output: 4.5, cacheRead: 0.145 },
|
|
62
|
+
vision: false,
|
|
63
|
+
reasoningMetadata: {
|
|
64
|
+
supported_efforts: ["max", "high", "low"],
|
|
65
|
+
mandatory: true,
|
|
66
|
+
},
|
|
67
|
+
};
|
|
68
|
+
|
|
56
69
|
// MoonshotAI. K3 supports reasoning efforts low/high/max (default max) and
|
|
57
70
|
// can be turned off (`mandatory: false`). The `-fast` endpoint is a shorthand
|
|
58
71
|
// to set thinking to off.
|
|
@@ -183,37 +196,40 @@ const FAMILIES: [NeuralwattModelFamily, NeuralwattVariantSpec[]][] = [
|
|
|
183
196
|
},
|
|
184
197
|
],
|
|
185
198
|
],
|
|
186
|
-
|
|
187
|
-
|
|
188
|
-
|
|
189
|
-
|
|
190
|
-
|
|
191
|
-
|
|
192
|
-
|
|
193
|
-
|
|
194
|
-
|
|
199
|
+
[
|
|
200
|
+
GLM_5_3,
|
|
201
|
+
[
|
|
202
|
+
{
|
|
203
|
+
id: "glm-5.3",
|
|
204
|
+
name: "GLM-5.3",
|
|
205
|
+
contextWindow: 1048560,
|
|
206
|
+
maxOutputTokens: null,
|
|
207
|
+
reasoning: true,
|
|
208
|
+
},
|
|
209
|
+
],
|
|
210
|
+
],
|
|
195
211
|
[
|
|
196
212
|
KIMI_K3,
|
|
197
213
|
[
|
|
198
214
|
{
|
|
199
215
|
id: "kimi-k3",
|
|
200
216
|
name: "Kimi K3",
|
|
201
|
-
contextWindow:
|
|
202
|
-
maxOutputTokens:
|
|
217
|
+
contextWindow: 1048560,
|
|
218
|
+
maxOutputTokens: null,
|
|
203
219
|
reasoning: true,
|
|
204
220
|
},
|
|
205
221
|
{
|
|
206
222
|
id: "kimi-k3-fast",
|
|
207
223
|
name: "Kimi K3 Fast",
|
|
208
|
-
contextWindow:
|
|
209
|
-
maxOutputTokens:
|
|
224
|
+
contextWindow: 1048560,
|
|
225
|
+
maxOutputTokens: null,
|
|
210
226
|
reasoning: false,
|
|
211
227
|
},
|
|
212
228
|
{
|
|
213
229
|
id: "kimi-k3-flex",
|
|
214
230
|
name: "Kimi K3 (flex)",
|
|
215
|
-
contextWindow:
|
|
216
|
-
maxOutputTokens:
|
|
231
|
+
contextWindow: 1048560,
|
|
232
|
+
maxOutputTokens: null,
|
|
217
233
|
reasoning: true,
|
|
218
234
|
costMultiplier: FLEX_COST_MULTIPLIER,
|
|
219
235
|
},
|
|
@@ -274,7 +290,7 @@ const FAMILIES: [NeuralwattModelFamily, NeuralwattVariantSpec[]][] = [
|
|
|
274
290
|
// The API now advertises flex variants but lists them at standard pricing;
|
|
275
291
|
// the 35% Flex discount is a billing-time concept applied here via
|
|
276
292
|
// `costMultiplier` rather than reflected in the catalog metadata.
|
|
277
|
-
// https://
|
|
293
|
+
// https://docs.neuralwatt.com/guides/flex-tier.md
|
|
278
294
|
|
|
279
295
|
export const NEURALWATT_MODELS: ProviderModelConfig[] = FAMILIES.flatMap(
|
|
280
296
|
([family, variants]) => buildNeuralwattFamily(family, variants),
|
|
@@ -12,6 +12,38 @@ import type {
|
|
|
12
12
|
|
|
13
13
|
export const MODEL_STORE_TTL_MS = 4 * 60 * 60 * 1000;
|
|
14
14
|
|
|
15
|
+
/**
|
|
16
|
+
* Scope a store entry applies to. The public catalog is a subset of any
|
|
17
|
+
* key-scoped catalog (preview, grant-gated, private models), so an entry
|
|
18
|
+
* stamped "public" must not shadow a keyed refresh — and a key-scoped entry
|
|
19
|
+
* must not be replayed for an anonymous user. Matches the anonymous-key
|
|
20
|
+
* convention in src/lib/neuralwatt-api.ts (authHeaders).
|
|
21
|
+
*/
|
|
22
|
+
const CATALOG_SCOPE_VERSION = "v1";
|
|
23
|
+
type CatalogScope = "public" | "key";
|
|
24
|
+
|
|
25
|
+
type ScopedModelsStoreEntry = ModelsStoreEntry & { catalogKey?: string };
|
|
26
|
+
|
|
27
|
+
function catalogScope(apiKey: string | undefined): CatalogScope {
|
|
28
|
+
return apiKey !== undefined && apiKey !== "" && apiKey !== "-"
|
|
29
|
+
? "key"
|
|
30
|
+
: "public";
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
function storedCatalogKey(entry: ScopedModelsStoreEntry): string | undefined {
|
|
34
|
+
return entry.catalogKey;
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
function catalogKeyMatches(
|
|
38
|
+
entry: ScopedModelsStoreEntry | undefined,
|
|
39
|
+
scope: CatalogScope,
|
|
40
|
+
): boolean {
|
|
41
|
+
return (
|
|
42
|
+
entry !== undefined &&
|
|
43
|
+
storedCatalogKey(entry) === `${scope} ${CATALOG_SCOPE_VERSION}`
|
|
44
|
+
);
|
|
45
|
+
}
|
|
46
|
+
|
|
15
47
|
export type FetchNeuralwattApiModels = (
|
|
16
48
|
apiKey: string | undefined,
|
|
17
49
|
signal?: AbortSignal,
|
|
@@ -25,6 +57,13 @@ function isFreshStoreEntry(
|
|
|
25
57
|
return Date.now() - checkedAt < MODEL_STORE_TTL_MS;
|
|
26
58
|
}
|
|
27
59
|
|
|
60
|
+
function isUsableStoreEntry(
|
|
61
|
+
entry: Readonly<ModelsStoreEntry> | undefined,
|
|
62
|
+
scope: CatalogScope,
|
|
63
|
+
): entry is ModelsStoreEntry {
|
|
64
|
+
return isFreshStoreEntry(entry) && catalogKeyMatches(entry, scope);
|
|
65
|
+
}
|
|
66
|
+
|
|
28
67
|
export function createNeuralwattRefreshModels(
|
|
29
68
|
staticModels: ReturnType<typeof buildNeuralwattProviderModels>,
|
|
30
69
|
fetchApiModels: FetchNeuralwattApiModels,
|
|
@@ -35,29 +74,29 @@ export function createNeuralwattRefreshModels(
|
|
|
35
74
|
context.signal.throwIfAborted();
|
|
36
75
|
const fallback = buildFromStore(staticModels);
|
|
37
76
|
try {
|
|
38
|
-
if (!context.allowNetwork) {
|
|
39
|
-
return context.stored
|
|
40
|
-
? buildFromStore(context.stored.models)
|
|
41
|
-
: fallback;
|
|
42
|
-
}
|
|
43
|
-
if (!context.force && isFreshStoreEntry(context.stored)) {
|
|
44
|
-
return buildFromStore(context.stored.models);
|
|
45
|
-
}
|
|
46
77
|
const apiKey =
|
|
47
78
|
context.credential?.type === "api_key"
|
|
48
79
|
? context.credential.key
|
|
49
80
|
: undefined;
|
|
81
|
+
const scope = catalogScope(apiKey);
|
|
82
|
+
const stored = context.stored as ScopedModelsStoreEntry | undefined;
|
|
83
|
+
if (!context.allowNetwork) {
|
|
84
|
+
return stored !== undefined && catalogKeyMatches(stored, scope)
|
|
85
|
+
? buildFromStore(stored.models)
|
|
86
|
+
: fallback;
|
|
87
|
+
}
|
|
88
|
+
if (!context.force && isUsableStoreEntry(stored, scope)) {
|
|
89
|
+
return buildFromStore(stored.models);
|
|
90
|
+
}
|
|
50
91
|
const apiModels = await fetchApiModels(apiKey, context.signal);
|
|
51
92
|
context.signal.throwIfAborted();
|
|
52
93
|
const models = buildFromApi(apiModels);
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
})
|
|
60
|
-
.catch(() => undefined);
|
|
94
|
+
const entry: ScopedModelsStoreEntry = {
|
|
95
|
+
models: models as unknown as ModelsStoreEntry["models"],
|
|
96
|
+
checkedAt: Date.now(),
|
|
97
|
+
catalogKey: `${scope} ${CATALOG_SCOPE_VERSION}`,
|
|
98
|
+
};
|
|
99
|
+
await context.publish({ persist: entry }).catch(() => undefined);
|
|
61
100
|
context.signal.throwIfAborted();
|
|
62
101
|
return models;
|
|
63
102
|
} catch (error) {
|
|
@@ -12,7 +12,7 @@ interface AssistantErrorLike {
|
|
|
12
12
|
* layer. Each sets unique headers so the client can tell which layer
|
|
13
13
|
* triggered the rejection.
|
|
14
14
|
*
|
|
15
|
-
* @see https://
|
|
15
|
+
* @see https://docs.neuralwatt.com/guides/rate-limits.md
|
|
16
16
|
*/
|
|
17
17
|
export interface NeuralwattRateLimitInfo {
|
|
18
18
|
/** Which rate-limit layer triggered the 429 */
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@aliou/pi-neuralwatt",
|
|
3
|
-
"version": "0.15.
|
|
3
|
+
"version": "0.15.3",
|
|
4
4
|
"license": "MIT",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"private": false,
|
|
@@ -52,6 +52,7 @@
|
|
|
52
52
|
"@types/node": "^25.0.10",
|
|
53
53
|
"husky": "^9.1.7",
|
|
54
54
|
"ts-json-schema-generator": "^2.4.0",
|
|
55
|
+
"tsx": "4.23.12",
|
|
55
56
|
"typescript": "^5.9.3",
|
|
56
57
|
"vitest": "^4.0.18"
|
|
57
58
|
},
|
|
@@ -67,6 +68,7 @@
|
|
|
67
68
|
"format": "biome check --write",
|
|
68
69
|
"test": "vitest run",
|
|
69
70
|
"test:watch": "vitest",
|
|
71
|
+
"check:models": "tsx scripts/check-models.ts",
|
|
70
72
|
"gen:schema": "pi-settings-schema -p src/config/types.ts -t NeuralwattConfig -o schema.json --version 0.12.0",
|
|
71
73
|
"check:schema": "pi-settings-schema -p src/config/types.ts -t NeuralwattConfig -o schema.json --version 0.12.0 --check",
|
|
72
74
|
"check:lockfile": "pnpm install --frozen-lockfile --ignore-scripts",
|