@aliou/pi-neuralwatt 0.16.0 → 0.16.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,3 +1,4 @@
1
+ import type { StreamOptions } from "@earendil-works/pi-ai";
1
2
  import { stream, streamSimple } from "@earendil-works/pi-ai/compat";
2
3
  import {
3
4
  NEURALWATT_BASE_URL,
@@ -8,9 +9,56 @@ import type { NeuralwattModel } from "../models/catalog";
8
9
  import type { AnyStreamSimple } from "../stream-simple";
9
10
  import type { NeuralwattApiHandler } from "./types";
10
11
 
12
+ type OpenAiCompletionsBody = {
13
+ messages?: Array<Record<string, unknown>>;
14
+ [key: string]: unknown;
15
+ };
16
+
17
+ /**
18
+ * Neuralwatt streams chain-of-thought in the `reasoning` field, and pi-ai
19
+ * replays prior thinking under the field name it recorded from the stream —
20
+ * also `reasoning`. The served chat templates only render `reasoning_content`
21
+ * (verified against the Kimi K3 template), so replayed thinking never reaches
22
+ * the model. Rename the replayed field on the wire. This is a move, not a
23
+ * copy: sending an empty `reasoning_content` next to a populated `reasoning`
24
+ * makes the gateway prefer the empty field and silently drops the replay.
25
+ */
26
+ function makeReasoningReplayInjector(
27
+ upstream?: StreamOptions["onPayload"],
28
+ ): NonNullable<StreamOptions["onPayload"]> {
29
+ return async (payload, model) => {
30
+ const next = await upstream?.(payload, model);
31
+ const body = (next !== undefined ? next : payload) as OpenAiCompletionsBody;
32
+ const messages = body.messages;
33
+ if (!Array.isArray(messages)) return body;
34
+
35
+ return {
36
+ ...body,
37
+ messages: messages.map((message) => {
38
+ if (message?.role !== "assistant" || !("reasoning" in message)) {
39
+ return message;
40
+ }
41
+ const { reasoning, ...rest } = message;
42
+ // A pre-set non-empty reasoning_content wins; drop the duplicate.
43
+ return typeof rest.reasoning_content === "string" &&
44
+ rest.reasoning_content.length > 0
45
+ ? rest
46
+ : { ...rest, reasoning_content: reasoning };
47
+ }),
48
+ };
49
+ };
50
+ }
51
+
11
52
  export function createOpenAiCompletionsApi(options?: {
12
53
  streamSimple?: AnyStreamSimple;
13
54
  }): NeuralwattApiHandler {
55
+ const withReasoningReplay = (options?: {
56
+ onPayload?: StreamOptions["onPayload"];
57
+ }) => ({
58
+ ...options,
59
+ onPayload: makeReasoningReplayInjector(options?.onPayload),
60
+ });
61
+
14
62
  return {
15
63
  stampModels: (models: NeuralwattModel[]) =>
16
64
  models.map((model) => {
@@ -24,7 +72,12 @@ export function createOpenAiCompletionsApi(options?: {
24
72
  };
25
73
  }),
26
74
  stream: (model, context, streamOptions) =>
27
- stream(model, context, streamOptions as never),
28
- streamSimple: options?.streamSimple ?? streamSimple,
75
+ stream(model, context, withReasoningReplay(streamOptions) as never),
76
+ streamSimple: (model, context, simpleOptions) =>
77
+ (options?.streamSimple ?? streamSimple)(
78
+ model,
79
+ context,
80
+ withReasoningReplay(simpleOptions) as never,
81
+ ),
29
82
  };
30
83
  }
@@ -161,10 +161,6 @@ export function buildNeuralwattModel(
161
161
  supportsDeveloperRole: false,
162
162
  maxTokensField: "max_tokens",
163
163
  };
164
- if (variant.reasoning) {
165
- compat.requiresReasoningContentOnAssistantMessages = true;
166
- }
167
-
168
164
  const multiplier = variant.costMultiplier ?? 1;
169
165
  const scale = (value: number): number =>
170
166
  multiplier === 1 ? value : Number((value * multiplier).toFixed(6));
@@ -35,9 +35,8 @@ function apiModelToProviderModel(model: NeuralwattApiModel): NeuralwattModel {
35
35
  const compat: NonNullable<ProviderModelConfig["compat"]> = {
36
36
  supportsDeveloperRole: meta.capabilities.developer_role,
37
37
  maxTokensField: "max_tokens",
38
+ ...COMPAT_OVERRIDES[model.id],
38
39
  };
39
- if (reasoning) compat.requiresReasoningContentOnAssistantMessages = true;
40
- Object.assign(compat, COMPAT_OVERRIDES[model.id]);
41
40
 
42
41
  const contextWindow = model.max_model_len;
43
42
 
@@ -9,10 +9,12 @@ import {
9
9
  // Pricing, capabilities, and limits are sourced from the API metadata fields;
10
10
  // `maxTokens` is `metadata.limits.max_output_tokens ?? max_model_len`.
11
11
  //
12
- // Each reasoning family snapshots its `reasoning.supported_efforts` +
13
- // `reasoning.mandatory` from the API; `buildThinkingLevelMap` turns that into
14
- // the Pi thinking level map by identity (no aliasing). See `models.test.ts`
15
- // for the drift check against the live catalog.
12
+ // Each reasoning family snapshots its `reasoning.supported_efforts`,
13
+ // `reasoning.mandatory`, and `reasoning.effort_aliases` from the API;
14
+ // `buildThinkingLevelMap` turns that into the Pi thinking level map by
15
+ // identity (no aliasing), while the anthropic-messages surface map resolves
16
+ // unsupported levels through the aliases. See `models.test.ts` for the drift
17
+ // check against the live catalog.
16
18
 
17
19
  // DeepSeek V4 Flash: efforts max/high/none, not mandatory.
18
20
  // https://api-docs.deepseek.com/guides/thinking_mode/
@@ -22,6 +24,24 @@ const DEEPSEEK_V4_FLASH: NeuralwattModelFamily = {
22
24
  reasoningMetadata: {
23
25
  supported_efforts: ["max", "high", "none"],
24
26
  mandatory: false,
27
+ effort_aliases: {
28
+ xhigh: "max",
29
+ medium: "high",
30
+ low: "high",
31
+ minimal: "high",
32
+ },
33
+ },
34
+ };
35
+
36
+ // DeepSeek. V4.1 Flash is vision-capable, unlike V4 Flash, and admits
37
+ // reasoning levels up to `xhigh`. Not mandatory (off defaults).
38
+ const DEEPSEEK_V4_1_FLASH: NeuralwattModelFamily = {
39
+ cost: { input: 0.15, output: 0.6, cacheRead: 0.015 },
40
+ vision: true,
41
+ reasoningMetadata: {
42
+ supported_efforts: ["max", "xhigh", "high", "low", "none"],
43
+ mandatory: false,
44
+ effort_aliases: { medium: "high", minimal: "low" },
25
45
  },
26
46
  };
27
47
 
@@ -37,6 +57,13 @@ const GEMMA_4: NeuralwattModelFamily = {
37
57
  reasoningMetadata: {
38
58
  supported_efforts: ["max", "none"],
39
59
  mandatory: false,
60
+ effort_aliases: {
61
+ xhigh: "max",
62
+ high: "max",
63
+ medium: "max",
64
+ low: "max",
65
+ minimal: "max",
66
+ },
40
67
  },
41
68
  };
42
69
 
@@ -48,6 +75,7 @@ const GLM_5_3: NeuralwattModelFamily = {
48
75
  reasoningMetadata: {
49
76
  supported_efforts: ["max", "high", "low"],
50
77
  mandatory: true,
78
+ effort_aliases: { xhigh: "max", medium: "high", minimal: "low" },
51
79
  },
52
80
  };
53
81
 
@@ -60,6 +88,7 @@ const GLM_5_3_FLASH: NeuralwattModelFamily = {
60
88
  reasoningMetadata: {
61
89
  supported_efforts: ["max", "high", "low"],
62
90
  mandatory: true,
91
+ effort_aliases: { xhigh: "max", medium: "high", minimal: "low" },
63
92
  },
64
93
  };
65
94
 
@@ -72,6 +101,7 @@ const KIMI_K3: NeuralwattModelFamily = {
72
101
  reasoningMetadata: {
73
102
  supported_efforts: ["max", "high", "low", "none"],
74
103
  mandatory: false,
104
+ effort_aliases: { xhigh: "max", medium: "high", minimal: "low" },
75
105
  },
76
106
  };
77
107
 
@@ -93,6 +123,13 @@ const QWEN_3_6_35B: NeuralwattModelFamily = {
93
123
  reasoningMetadata: {
94
124
  supported_efforts: ["high", "none"],
95
125
  mandatory: false,
126
+ effort_aliases: {
127
+ max: "high",
128
+ xhigh: "high",
129
+ medium: "high",
130
+ low: "high",
131
+ minimal: "high",
132
+ },
96
133
  },
97
134
  };
98
135
 
@@ -105,6 +142,7 @@ const QWEN_3_8_27B: NeuralwattModelFamily = {
105
142
  reasoningMetadata: {
106
143
  supported_efforts: ["xhigh", "medium", "low", "none"],
107
144
  mandatory: false,
145
+ effort_aliases: { max: "xhigh", high: "xhigh", minimal: "low" },
108
146
  },
109
147
  };
110
148
 
@@ -116,14 +154,14 @@ const FAMILIES: [NeuralwattModelFamily, NeuralwattVariantSpec[]][] = [
116
154
  id: "deepseek-v4-flash",
117
155
  name: "DeepSeek V4 Flash",
118
156
  contextWindow: 1048560,
119
- maxOutputTokens: 65536,
157
+ maxOutputTokens: 393216,
120
158
  reasoning: true,
121
159
  },
122
160
  {
123
161
  id: "deepseek-v4-flash-flex",
124
162
  name: "DeepSeek V4 Flash (flex)",
125
163
  contextWindow: 1048560,
126
- maxOutputTokens: 65536,
164
+ maxOutputTokens: 393216,
127
165
  reasoning: true,
128
166
  costMultiplier: 0.65,
129
167
  },
@@ -131,9 +169,29 @@ const FAMILIES: [NeuralwattModelFamily, NeuralwattVariantSpec[]][] = [
131
169
  id: "deepseek-v4-flash-speed",
132
170
  name: "DeepSeek V4 Flash (Speed)",
133
171
  contextWindow: 1048560,
134
- maxOutputTokens: 65536,
172
+ maxOutputTokens: 393216,
173
+ reasoning: true,
174
+ },
175
+ ],
176
+ ],
177
+ [
178
+ DEEPSEEK_V4_1_FLASH,
179
+ [
180
+ {
181
+ id: "deepseek-v4.1-flash",
182
+ name: "DeepSeek V4.1 Flash",
183
+ contextWindow: 1048560,
184
+ maxOutputTokens: 393216,
135
185
  reasoning: true,
136
186
  },
187
+ {
188
+ id: "deepseek-v4.1-flash-flex",
189
+ name: "DeepSeek V4.1 Flash (flex)",
190
+ contextWindow: 1048560,
191
+ maxOutputTokens: 393216,
192
+ reasoning: true,
193
+ costMultiplier: 0.65,
194
+ },
137
195
  ],
138
196
  ],
139
197
  [
@@ -250,21 +308,21 @@ const FAMILIES: [NeuralwattModelFamily, NeuralwattVariantSpec[]][] = [
250
308
  {
251
309
  id: "qwen3.6-35b",
252
310
  name: "Qwen3.6 35B",
253
- contextWindow: 131056,
311
+ contextWindow: 262128,
254
312
  maxOutputTokens: null,
255
313
  reasoning: true,
256
314
  },
257
315
  {
258
316
  id: "qwen3.6-35b-fast",
259
317
  name: "Qwen3.6 35B Fast",
260
- contextWindow: 131056,
318
+ contextWindow: 262128,
261
319
  maxOutputTokens: null,
262
320
  reasoning: false,
263
321
  },
264
322
  {
265
323
  id: "qwen3.6-35b-flex",
266
324
  name: "Qwen3.6 35B (flex)",
267
- contextWindow: 131056,
325
+ contextWindow: 262128,
268
326
  maxOutputTokens: null,
269
327
  reasoning: true,
270
328
  costMultiplier: 0.65,
@@ -19,7 +19,7 @@ export const MODEL_STORE_TTL_MS = 4 * 60 * 60 * 1000;
19
19
  * must not be replayed for an anonymous user. Matches the anonymous-key
20
20
  * convention in src/lib/neuralwatt-api.ts (authHeaders).
21
21
  */
22
- const CATALOG_SCOPE_VERSION = "v1";
22
+ const CATALOG_SCOPE_VERSION = "v2";
23
23
  type CatalogScope = "public" | "key";
24
24
 
25
25
  type ScopedModelsStoreEntry = ModelsStoreEntry & { catalogKey?: string };
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@aliou/pi-neuralwatt",
3
- "version": "0.16.0",
3
+ "version": "0.16.2",
4
4
  "license": "MIT",
5
5
  "type": "module",
6
6
  "private": false,