@aliou/pi-neuralwatt 0.15.2 → 0.15.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -144,5 +144,5 @@ This repository uses [Changesets](https://github.com/changesets/changesets) for
144
144
  ## Links
145
145
 
146
146
  - [Neuralwatt](https://portal.neuralwatt.com/auth/register?ref=NW-ALIOU-Q7MF)
147
- - [Neuralwatt API Docs](https://neuralwatt.com/docs)
147
+ - [Neuralwatt API Docs](https://docs.neuralwatt.com/quickstart.md)
148
148
  - [Pi Documentation](https://buildwithpi.ai/)
@@ -13,7 +13,7 @@ export type ThinkingLevelMap = NonNullable<
13
13
  * streams. A non-streaming request to a `-flex` model silently falls back to
14
14
  * the standard tier and the standard price.
15
15
  *
16
- * https://portal.neuralwatt.com/docs/guides/flex-tier
16
+ * https://docs.neuralwatt.com/guides/flex-tier.md
17
17
  */
18
18
  export const FLEX_COST_MULTIPLIER = 0.65;
19
19
 
@@ -107,12 +107,14 @@ export function buildThinkingLevelMap(
107
107
 
108
108
  /**
109
109
  * Neuralwatt reports `max_output_tokens: null` for models whose output is only
110
- * bounded by the context window. Mirror the API instead of inventing a cap.
110
+ * bounded by the context window. Some models incorrectly report 0; treat 0
111
+ * like null so we never emit maxTokens: 0.
111
112
  */
112
113
  export function resolveMaxTokens(
113
114
  maxOutputTokens: number | null | undefined,
114
115
  contextWindow: number,
115
116
  ): number {
117
+ if (maxOutputTokens === 0) return contextWindow;
116
118
  return maxOutputTokens ?? contextWindow;
117
119
  }
118
120
 
@@ -10,12 +10,6 @@ import { NEURALWATT_MODELS } from "./public-models";
10
10
 
11
11
  export type NeuralwattModel = ProviderModelConfig;
12
12
 
13
- const CONTEXT_WINDOW_OVERRIDES: ReadonlyMap<string, number> = new Map([
14
- ["kimi-k3", 327_680],
15
- ["kimi-k3-fast", 327_680],
16
- ["kimi-k3-flex", 327_680],
17
- ]);
18
-
19
13
  // Chat-template thinking: the API exposes a `reasoning` block, but the
20
14
  // underlying mechanism is chat_template_kwargs, so Pi needs the mapping.
21
15
  const COMPAT_OVERRIDES: Partial<
@@ -62,8 +56,7 @@ function apiModelToProviderModel(model: NeuralwattApiModel): NeuralwattModel {
62
56
  if (reasoning) compat.requiresReasoningContentOnAssistantMessages = true;
63
57
  Object.assign(compat, COMPAT_OVERRIDES[model.id]);
64
58
 
65
- const contextWindow =
66
- CONTEXT_WINDOW_OVERRIDES.get(model.id) ?? model.max_model_len;
59
+ const contextWindow = model.max_model_len;
67
60
 
68
61
  const result: NeuralwattModel = {
69
62
  id: model.id,
@@ -135,7 +128,14 @@ export function buildNeuralwattProviderModelsFromApi(
135
128
  const models = apiModels
136
129
  .filter(
137
130
  (m) =>
138
- m.metadata && !m.metadata.deprecated && !m.metadata.pricing.pricing_tbd,
131
+ m.metadata &&
132
+ !m.metadata.deprecated &&
133
+ !m.metadata.pricing.pricing_tbd &&
134
+ // Exclude non-chat models (e.g. embeddings) by task
135
+ !(
136
+ m.metadata.capabilities.task &&
137
+ !["chat", "completions"].includes(m.metadata.capabilities.task)
138
+ ),
139
139
  )
140
140
  .map(apiModelToProviderModel);
141
141
  return [...models, ...buildAliases(models, apiModels)];
@@ -31,7 +31,7 @@ const DEEPSEEK_V4_FLASH: NeuralwattModelFamily = {
31
31
  // `max` and `none`; every non-`none` request resolves to `max` upstream.
32
32
  // It does not reason by default (`default_enabled: false`), but the model
33
33
  // can produce reasoning traces when asked. See
34
- // https://portal.neuralwatt.com/docs/api/chat-completions#reasoning-effort
34
+ // https://docs.neuralwatt.com/api/chat-completions.md
35
35
  const GEMMA_4: NeuralwattModelFamily = {
36
36
  cost: { input: 0.144, output: 0.42, cacheRead: 0.0144 },
37
37
  vision: true,
@@ -208,37 +208,28 @@ const FAMILIES: [NeuralwattModelFamily, NeuralwattVariantSpec[]][] = [
208
208
  },
209
209
  ],
210
210
  ],
211
- // The kimi-k3 endpoint rejects anything above 327,680 total tokens with
212
- // `400: max_completion_tokens is too large … supports at most 327680
213
- // completion tokens` (verified at runtime), even though the API advertises
214
- // `max_model_len: 1048560` with a null output cap for the whole family.
215
- // The -fast/-flex endpoints don't enforce any cap server-side yet (they
216
- // accept max_completion_tokens beyond the advertised window), but they are
217
- // the same K3 deployment and are expected to share the 327,680 limit, so
218
- // all three variants are pinned to it. The drift check in models.test.ts
219
- // whitelists this divergence via CONTEXT_WINDOW_OVERRIDES.
220
211
  [
221
212
  KIMI_K3,
222
213
  [
223
214
  {
224
215
  id: "kimi-k3",
225
216
  name: "Kimi K3",
226
- contextWindow: 327680,
227
- maxOutputTokens: 327680,
217
+ contextWindow: 1048560,
218
+ maxOutputTokens: null,
228
219
  reasoning: true,
229
220
  },
230
221
  {
231
222
  id: "kimi-k3-fast",
232
223
  name: "Kimi K3 Fast",
233
- contextWindow: 327680,
234
- maxOutputTokens: 327680,
224
+ contextWindow: 1048560,
225
+ maxOutputTokens: null,
235
226
  reasoning: false,
236
227
  },
237
228
  {
238
229
  id: "kimi-k3-flex",
239
230
  name: "Kimi K3 (flex)",
240
- contextWindow: 327680,
241
- maxOutputTokens: 327680,
231
+ contextWindow: 1048560,
232
+ maxOutputTokens: null,
242
233
  reasoning: true,
243
234
  costMultiplier: FLEX_COST_MULTIPLIER,
244
235
  },
@@ -299,7 +290,7 @@ const FAMILIES: [NeuralwattModelFamily, NeuralwattVariantSpec[]][] = [
299
290
  // The API now advertises flex variants but lists them at standard pricing;
300
291
  // the 35% Flex discount is a billing-time concept applied here via
301
292
  // `costMultiplier` rather than reflected in the catalog metadata.
302
- // https://portal.neuralwatt.com/docs/guides/flex-tier
293
+ // https://docs.neuralwatt.com/guides/flex-tier.md
303
294
 
304
295
  export const NEURALWATT_MODELS: ProviderModelConfig[] = FAMILIES.flatMap(
305
296
  ([family, variants]) => buildNeuralwattFamily(family, variants),
@@ -12,6 +12,38 @@ import type {
12
12
 
13
13
  export const MODEL_STORE_TTL_MS = 4 * 60 * 60 * 1000;
14
14
 
15
+ /**
16
+ * Scope a store entry applies to. The public catalog is a subset of any
17
+ * key-scoped catalog (preview, grant-gated, private models), so an entry
18
+ * stamped "public" must not shadow a keyed refresh — and a key-scoped entry
19
+ * must not be replayed for an anonymous user. Matches the anonymous-key
20
+ * convention in src/lib/neuralwatt-api.ts (authHeaders).
21
+ */
22
+ const CATALOG_SCOPE_VERSION = "v1";
23
+ type CatalogScope = "public" | "key";
24
+
25
+ type ScopedModelsStoreEntry = ModelsStoreEntry & { catalogKey?: string };
26
+
27
+ function catalogScope(apiKey: string | undefined): CatalogScope {
28
+ return apiKey !== undefined && apiKey !== "" && apiKey !== "-"
29
+ ? "key"
30
+ : "public";
31
+ }
32
+
33
+ function storedCatalogKey(entry: ScopedModelsStoreEntry): string | undefined {
34
+ return entry.catalogKey;
35
+ }
36
+
37
+ function catalogKeyMatches(
38
+ entry: ScopedModelsStoreEntry | undefined,
39
+ scope: CatalogScope,
40
+ ): boolean {
41
+ return (
42
+ entry !== undefined &&
43
+ storedCatalogKey(entry) === `${scope} ${CATALOG_SCOPE_VERSION}`
44
+ );
45
+ }
46
+
15
47
  export type FetchNeuralwattApiModels = (
16
48
  apiKey: string | undefined,
17
49
  signal?: AbortSignal,
@@ -25,6 +57,13 @@ function isFreshStoreEntry(
25
57
  return Date.now() - checkedAt < MODEL_STORE_TTL_MS;
26
58
  }
27
59
 
60
+ function isUsableStoreEntry(
61
+ entry: Readonly<ModelsStoreEntry> | undefined,
62
+ scope: CatalogScope,
63
+ ): entry is ModelsStoreEntry {
64
+ return isFreshStoreEntry(entry) && catalogKeyMatches(entry, scope);
65
+ }
66
+
28
67
  export function createNeuralwattRefreshModels(
29
68
  staticModels: ReturnType<typeof buildNeuralwattProviderModels>,
30
69
  fetchApiModels: FetchNeuralwattApiModels,
@@ -35,29 +74,29 @@ export function createNeuralwattRefreshModels(
35
74
  context.signal.throwIfAborted();
36
75
  const fallback = buildFromStore(staticModels);
37
76
  try {
38
- if (!context.allowNetwork) {
39
- return context.stored
40
- ? buildFromStore(context.stored.models)
41
- : fallback;
42
- }
43
- if (!context.force && isFreshStoreEntry(context.stored)) {
44
- return buildFromStore(context.stored.models);
45
- }
46
77
  const apiKey =
47
78
  context.credential?.type === "api_key"
48
79
  ? context.credential.key
49
80
  : undefined;
81
+ const scope = catalogScope(apiKey);
82
+ const stored = context.stored as ScopedModelsStoreEntry | undefined;
83
+ if (!context.allowNetwork) {
84
+ return stored !== undefined && catalogKeyMatches(stored, scope)
85
+ ? buildFromStore(stored.models)
86
+ : fallback;
87
+ }
88
+ if (!context.force && isUsableStoreEntry(stored, scope)) {
89
+ return buildFromStore(stored.models);
90
+ }
50
91
  const apiModels = await fetchApiModels(apiKey, context.signal);
51
92
  context.signal.throwIfAborted();
52
93
  const models = buildFromApi(apiModels);
53
- await context
54
- .publish({
55
- persist: {
56
- models: models as unknown as ModelsStoreEntry["models"],
57
- checkedAt: Date.now(),
58
- },
59
- })
60
- .catch(() => undefined);
94
+ const entry: ScopedModelsStoreEntry = {
95
+ models: models as unknown as ModelsStoreEntry["models"],
96
+ checkedAt: Date.now(),
97
+ catalogKey: `${scope} ${CATALOG_SCOPE_VERSION}`,
98
+ };
99
+ await context.publish({ persist: entry }).catch(() => undefined);
61
100
  context.signal.throwIfAborted();
62
101
  return models;
63
102
  } catch (error) {
@@ -12,7 +12,7 @@ interface AssistantErrorLike {
12
12
  * layer. Each sets unique headers so the client can tell which layer
13
13
  * triggered the rejection.
14
14
  *
15
- * @see https://portal.neuralwatt.com/docs/guides/rate-limits
15
+ * @see https://docs.neuralwatt.com/guides/rate-limits.md
16
16
  */
17
17
  export interface NeuralwattRateLimitInfo {
18
18
  /** Which rate-limit layer triggered the 429 */
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@aliou/pi-neuralwatt",
3
- "version": "0.15.2",
3
+ "version": "0.15.3",
4
4
  "license": "MIT",
5
5
  "type": "module",
6
6
  "private": false,
@@ -16,6 +16,8 @@ export interface NeuralwattApiModelCapabilities {
16
16
  streaming: boolean;
17
17
  system_role: boolean;
18
18
  developer_role: boolean;
19
+ task?: string;
20
+ embedding_dimensions?: number;
19
21
  }
20
22
 
21
23
  /**