@aliou/pi-neuralwatt 0.15.1 → 0.15.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -144,5 +144,5 @@ This repository uses [Changesets](https://github.com/changesets/changesets) for
144
144
  ## Links
145
145
 
146
146
  - [Neuralwatt](https://portal.neuralwatt.com/auth/register?ref=NW-ALIOU-Q7MF)
147
- - [Neuralwatt API Docs](https://neuralwatt.com/docs)
147
+ - [Neuralwatt API Docs](https://docs.neuralwatt.com/quickstart.md)
148
148
  - [Pi Documentation](https://buildwithpi.ai/)
@@ -13,7 +13,7 @@ export type ThinkingLevelMap = NonNullable<
13
13
  * streams. A non-streaming request to a `-flex` model silently falls back to
14
14
  * the standard tier and the standard price.
15
15
  *
16
- * https://portal.neuralwatt.com/docs/guides/flex-tier
16
+ * https://docs.neuralwatt.com/guides/flex-tier.md
17
17
  */
18
18
  export const FLEX_COST_MULTIPLIER = 0.65;
19
19
 
@@ -107,12 +107,14 @@ export function buildThinkingLevelMap(
107
107
 
108
108
  /**
109
109
  * Neuralwatt reports `max_output_tokens: null` for models whose output is only
110
- * bounded by the context window. Mirror the API instead of inventing a cap.
110
+ * bounded by the context window. Some models incorrectly report 0; treat 0
111
+ * like null so we never emit maxTokens: 0.
111
112
  */
112
113
  export function resolveMaxTokens(
113
114
  maxOutputTokens: number | null | undefined,
114
115
  contextWindow: number,
115
116
  ): number {
117
+ if (maxOutputTokens === 0) return contextWindow;
116
118
  return maxOutputTokens ?? contextWindow;
117
119
  }
118
120
 
@@ -10,12 +10,6 @@ import { NEURALWATT_MODELS } from "./public-models";
10
10
 
11
11
  export type NeuralwattModel = ProviderModelConfig;
12
12
 
13
- const CONTEXT_WINDOW_OVERRIDES: ReadonlyMap<string, number> = new Map([
14
- ["kimi-k3", 327_680],
15
- ["kimi-k3-fast", 327_680],
16
- ["kimi-k3-flex", 327_680],
17
- ]);
18
-
19
13
  // Chat-template thinking: the API exposes a `reasoning` block, but the
20
14
  // underlying mechanism is chat_template_kwargs, so Pi needs the mapping.
21
15
  const COMPAT_OVERRIDES: Partial<
@@ -62,8 +56,7 @@ function apiModelToProviderModel(model: NeuralwattApiModel): NeuralwattModel {
62
56
  if (reasoning) compat.requiresReasoningContentOnAssistantMessages = true;
63
57
  Object.assign(compat, COMPAT_OVERRIDES[model.id]);
64
58
 
65
- const contextWindow =
66
- CONTEXT_WINDOW_OVERRIDES.get(model.id) ?? model.max_model_len;
59
+ const contextWindow = model.max_model_len;
67
60
 
68
61
  const result: NeuralwattModel = {
69
62
  id: model.id,
@@ -135,7 +128,14 @@ export function buildNeuralwattProviderModelsFromApi(
135
128
  const models = apiModels
136
129
  .filter(
137
130
  (m) =>
138
- m.metadata && !m.metadata.deprecated && !m.metadata.pricing.pricing_tbd,
131
+ m.metadata &&
132
+ !m.metadata.deprecated &&
133
+ !m.metadata.pricing.pricing_tbd &&
134
+ // Exclude non-chat models (e.g. embeddings) by task
135
+ !(
136
+ m.metadata.capabilities.task &&
137
+ !["chat", "completions"].includes(m.metadata.capabilities.task)
138
+ ),
139
139
  )
140
140
  .map(apiModelToProviderModel);
141
141
  return [...models, ...buildAliases(models, apiModels)];
@@ -31,7 +31,7 @@ const DEEPSEEK_V4_FLASH: NeuralwattModelFamily = {
31
31
  // `max` and `none`; every non-`none` request resolves to `max` upstream.
32
32
  // It does not reason by default (`default_enabled: false`), but the model
33
33
  // can produce reasoning traces when asked. See
34
- // https://portal.neuralwatt.com/docs/api/chat-completions#reasoning-effort
34
+ // https://docs.neuralwatt.com/api/chat-completions.md
35
35
  const GEMMA_4: NeuralwattModelFamily = {
36
36
  cost: { input: 0.144, output: 0.42, cacheRead: 0.0144 },
37
37
  vision: true,
@@ -53,6 +53,19 @@ const GLM_5_2: NeuralwattModelFamily = {
53
53
  },
54
54
  };
55
55
 
56
+ // ZhipuAI. GLM-5.3 ships as a GLM-5.2 weight swap in gated preview, with
57
+ // GLM-5.2 pricing parity (per the API metadata; review at launch). Unlike
58
+ // 5.2, reasoning is mandatory and `none` is not offered: efforts are
59
+ // max/high/low (default max).
60
+ const GLM_5_3: NeuralwattModelFamily = {
61
+ cost: { input: 1.45, output: 4.5, cacheRead: 0.145 },
62
+ vision: false,
63
+ reasoningMetadata: {
64
+ supported_efforts: ["max", "high", "low"],
65
+ mandatory: true,
66
+ },
67
+ };
68
+
56
69
  // MoonshotAI. K3 supports reasoning efforts low/high/max (default max) and
57
70
  // can be turned off (`mandatory: false`). The `-fast` endpoint is a shorthand
58
71
  // to set thinking to off.
@@ -183,37 +196,40 @@ const FAMILIES: [NeuralwattModelFamily, NeuralwattVariantSpec[]][] = [
183
196
  },
184
197
  ],
185
198
  ],
186
- // The kimi-k3 endpoint rejects anything above 327,680 total tokens with
187
- // `400: max_completion_tokens is too large … supports at most 327680
188
- // completion tokens` (verified at runtime), even though the API advertises
189
- // `max_model_len: 1048560` with a null output cap for the whole family.
190
- // The -fast/-flex endpoints don't enforce any cap server-side yet (they
191
- // accept max_completion_tokens beyond the advertised window), but they are
192
- // the same K3 deployment and are expected to share the 327,680 limit, so
193
- // all three variants are pinned to it. The drift check in models.test.ts
194
- // whitelists this divergence via CONTEXT_WINDOW_OVERRIDES.
199
+ [
200
+ GLM_5_3,
201
+ [
202
+ {
203
+ id: "glm-5.3",
204
+ name: "GLM-5.3",
205
+ contextWindow: 1048560,
206
+ maxOutputTokens: null,
207
+ reasoning: true,
208
+ },
209
+ ],
210
+ ],
195
211
  [
196
212
  KIMI_K3,
197
213
  [
198
214
  {
199
215
  id: "kimi-k3",
200
216
  name: "Kimi K3",
201
- contextWindow: 327680,
202
- maxOutputTokens: 327680,
217
+ contextWindow: 1048560,
218
+ maxOutputTokens: null,
203
219
  reasoning: true,
204
220
  },
205
221
  {
206
222
  id: "kimi-k3-fast",
207
223
  name: "Kimi K3 Fast",
208
- contextWindow: 327680,
209
- maxOutputTokens: 327680,
224
+ contextWindow: 1048560,
225
+ maxOutputTokens: null,
210
226
  reasoning: false,
211
227
  },
212
228
  {
213
229
  id: "kimi-k3-flex",
214
230
  name: "Kimi K3 (flex)",
215
- contextWindow: 327680,
216
- maxOutputTokens: 327680,
231
+ contextWindow: 1048560,
232
+ maxOutputTokens: null,
217
233
  reasoning: true,
218
234
  costMultiplier: FLEX_COST_MULTIPLIER,
219
235
  },
@@ -274,7 +290,7 @@ const FAMILIES: [NeuralwattModelFamily, NeuralwattVariantSpec[]][] = [
274
290
  // The API now advertises flex variants but lists them at standard pricing;
275
291
  // the 35% Flex discount is a billing-time concept applied here via
276
292
  // `costMultiplier` rather than reflected in the catalog metadata.
277
- // https://portal.neuralwatt.com/docs/guides/flex-tier
293
+ // https://docs.neuralwatt.com/guides/flex-tier.md
278
294
 
279
295
  export const NEURALWATT_MODELS: ProviderModelConfig[] = FAMILIES.flatMap(
280
296
  ([family, variants]) => buildNeuralwattFamily(family, variants),
@@ -12,6 +12,38 @@ import type {
12
12
 
13
13
  export const MODEL_STORE_TTL_MS = 4 * 60 * 60 * 1000;
14
14
 
15
+ /**
16
+ * Scope a store entry applies to. The public catalog is a subset of any
17
+ * key-scoped catalog (preview, grant-gated, private models), so an entry
18
+ * stamped "public" must not shadow a keyed refresh — and a key-scoped entry
19
+ * must not be replayed for an anonymous user. Matches the anonymous-key
20
+ * convention in src/lib/neuralwatt-api.ts (authHeaders).
21
+ */
22
+ const CATALOG_SCOPE_VERSION = "v1";
23
+ type CatalogScope = "public" | "key";
24
+
25
+ type ScopedModelsStoreEntry = ModelsStoreEntry & { catalogKey?: string };
26
+
27
+ function catalogScope(apiKey: string | undefined): CatalogScope {
28
+ return apiKey !== undefined && apiKey !== "" && apiKey !== "-"
29
+ ? "key"
30
+ : "public";
31
+ }
32
+
33
+ function storedCatalogKey(entry: ScopedModelsStoreEntry): string | undefined {
34
+ return entry.catalogKey;
35
+ }
36
+
37
+ function catalogKeyMatches(
38
+ entry: ScopedModelsStoreEntry | undefined,
39
+ scope: CatalogScope,
40
+ ): boolean {
41
+ return (
42
+ entry !== undefined &&
43
+ storedCatalogKey(entry) === `${scope} ${CATALOG_SCOPE_VERSION}`
44
+ );
45
+ }
46
+
15
47
  export type FetchNeuralwattApiModels = (
16
48
  apiKey: string | undefined,
17
49
  signal?: AbortSignal,
@@ -25,6 +57,13 @@ function isFreshStoreEntry(
25
57
  return Date.now() - checkedAt < MODEL_STORE_TTL_MS;
26
58
  }
27
59
 
60
+ function isUsableStoreEntry(
61
+ entry: Readonly<ModelsStoreEntry> | undefined,
62
+ scope: CatalogScope,
63
+ ): entry is ModelsStoreEntry {
64
+ return isFreshStoreEntry(entry) && catalogKeyMatches(entry, scope);
65
+ }
66
+
28
67
  export function createNeuralwattRefreshModels(
29
68
  staticModels: ReturnType<typeof buildNeuralwattProviderModels>,
30
69
  fetchApiModels: FetchNeuralwattApiModels,
@@ -35,29 +74,29 @@ export function createNeuralwattRefreshModels(
35
74
  context.signal.throwIfAborted();
36
75
  const fallback = buildFromStore(staticModels);
37
76
  try {
38
- if (!context.allowNetwork) {
39
- return context.stored
40
- ? buildFromStore(context.stored.models)
41
- : fallback;
42
- }
43
- if (!context.force && isFreshStoreEntry(context.stored)) {
44
- return buildFromStore(context.stored.models);
45
- }
46
77
  const apiKey =
47
78
  context.credential?.type === "api_key"
48
79
  ? context.credential.key
49
80
  : undefined;
81
+ const scope = catalogScope(apiKey);
82
+ const stored = context.stored as ScopedModelsStoreEntry | undefined;
83
+ if (!context.allowNetwork) {
84
+ return stored !== undefined && catalogKeyMatches(stored, scope)
85
+ ? buildFromStore(stored.models)
86
+ : fallback;
87
+ }
88
+ if (!context.force && isUsableStoreEntry(stored, scope)) {
89
+ return buildFromStore(stored.models);
90
+ }
50
91
  const apiModels = await fetchApiModels(apiKey, context.signal);
51
92
  context.signal.throwIfAborted();
52
93
  const models = buildFromApi(apiModels);
53
- await context
54
- .publish({
55
- persist: {
56
- models: models as unknown as ModelsStoreEntry["models"],
57
- checkedAt: Date.now(),
58
- },
59
- })
60
- .catch(() => undefined);
94
+ const entry: ScopedModelsStoreEntry = {
95
+ models: models as unknown as ModelsStoreEntry["models"],
96
+ checkedAt: Date.now(),
97
+ catalogKey: `${scope} ${CATALOG_SCOPE_VERSION}`,
98
+ };
99
+ await context.publish({ persist: entry }).catch(() => undefined);
61
100
  context.signal.throwIfAborted();
62
101
  return models;
63
102
  } catch (error) {
@@ -12,7 +12,7 @@ interface AssistantErrorLike {
12
12
  * layer. Each sets unique headers so the client can tell which layer
13
13
  * triggered the rejection.
14
14
  *
15
- * @see https://portal.neuralwatt.com/docs/guides/rate-limits
15
+ * @see https://docs.neuralwatt.com/guides/rate-limits.md
16
16
  */
17
17
  export interface NeuralwattRateLimitInfo {
18
18
  /** Which rate-limit layer triggered the 429 */
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@aliou/pi-neuralwatt",
3
- "version": "0.15.1",
3
+ "version": "0.15.3",
4
4
  "license": "MIT",
5
5
  "type": "module",
6
6
  "private": false,
@@ -52,6 +52,7 @@
52
52
  "@types/node": "^25.0.10",
53
53
  "husky": "^9.1.7",
54
54
  "ts-json-schema-generator": "^2.4.0",
55
+ "tsx": "4.23.12",
55
56
  "typescript": "^5.9.3",
56
57
  "vitest": "^4.0.18"
57
58
  },
@@ -67,6 +68,7 @@
67
68
  "format": "biome check --write",
68
69
  "test": "vitest run",
69
70
  "test:watch": "vitest",
71
+ "check:models": "tsx scripts/check-models.ts",
70
72
  "gen:schema": "pi-settings-schema -p src/config/types.ts -t NeuralwattConfig -o schema.json --version 0.12.0",
71
73
  "check:schema": "pi-settings-schema -p src/config/types.ts -t NeuralwattConfig -o schema.json --version 0.12.0 --check",
72
74
  "check:lockfile": "pnpm install --frozen-lockfile --ignore-scripts",
@@ -16,6 +16,8 @@ export interface NeuralwattApiModelCapabilities {
16
16
  streaming: boolean;
17
17
  system_role: boolean;
18
18
  developer_role: boolean;
19
+ task?: string;
20
+ embedding_dimensions?: number;
19
21
  }
20
22
 
21
23
  /**