@aliou/pi-neuralwatt 0.16.1 → 0.16.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
import type { StreamOptions } from "@earendil-works/pi-ai";
|
|
1
2
|
import { stream, streamSimple } from "@earendil-works/pi-ai/compat";
|
|
2
3
|
import {
|
|
3
4
|
NEURALWATT_BASE_URL,
|
|
@@ -8,9 +9,56 @@ import type { NeuralwattModel } from "../models/catalog";
|
|
|
8
9
|
import type { AnyStreamSimple } from "../stream-simple";
|
|
9
10
|
import type { NeuralwattApiHandler } from "./types";
|
|
10
11
|
|
|
12
|
+
type OpenAiCompletionsBody = {
|
|
13
|
+
messages?: Array<Record<string, unknown>>;
|
|
14
|
+
[key: string]: unknown;
|
|
15
|
+
};
|
|
16
|
+
|
|
17
|
+
/**
|
|
18
|
+
* Neuralwatt streams chain-of-thought in the `reasoning` field, and pi-ai
|
|
19
|
+
* replays prior thinking under the field name it recorded from the stream —
|
|
20
|
+
* also `reasoning`. The served chat templates only render `reasoning_content`
|
|
21
|
+
* (verified against the Kimi K3 template), so replayed thinking never reaches
|
|
22
|
+
* the model. Rename the replayed field on the wire. This is a move, not a
|
|
23
|
+
* copy: sending an empty `reasoning_content` next to a populated `reasoning`
|
|
24
|
+
* makes the gateway prefer the empty field and silently drops the replay.
|
|
25
|
+
*/
|
|
26
|
+
function makeReasoningReplayInjector(
|
|
27
|
+
upstream?: StreamOptions["onPayload"],
|
|
28
|
+
): NonNullable<StreamOptions["onPayload"]> {
|
|
29
|
+
return async (payload, model) => {
|
|
30
|
+
const next = await upstream?.(payload, model);
|
|
31
|
+
const body = (next !== undefined ? next : payload) as OpenAiCompletionsBody;
|
|
32
|
+
const messages = body.messages;
|
|
33
|
+
if (!Array.isArray(messages)) return body;
|
|
34
|
+
|
|
35
|
+
return {
|
|
36
|
+
...body,
|
|
37
|
+
messages: messages.map((message) => {
|
|
38
|
+
if (message?.role !== "assistant" || !("reasoning" in message)) {
|
|
39
|
+
return message;
|
|
40
|
+
}
|
|
41
|
+
const { reasoning, ...rest } = message;
|
|
42
|
+
// A pre-set non-empty reasoning_content wins; drop the duplicate.
|
|
43
|
+
return typeof rest.reasoning_content === "string" &&
|
|
44
|
+
rest.reasoning_content.length > 0
|
|
45
|
+
? rest
|
|
46
|
+
: { ...rest, reasoning_content: reasoning };
|
|
47
|
+
}),
|
|
48
|
+
};
|
|
49
|
+
};
|
|
50
|
+
}
|
|
51
|
+
|
|
11
52
|
export function createOpenAiCompletionsApi(options?: {
|
|
12
53
|
streamSimple?: AnyStreamSimple;
|
|
13
54
|
}): NeuralwattApiHandler {
|
|
55
|
+
const withReasoningReplay = (options?: {
|
|
56
|
+
onPayload?: StreamOptions["onPayload"];
|
|
57
|
+
}) => ({
|
|
58
|
+
...options,
|
|
59
|
+
onPayload: makeReasoningReplayInjector(options?.onPayload),
|
|
60
|
+
});
|
|
61
|
+
|
|
14
62
|
return {
|
|
15
63
|
stampModels: (models: NeuralwattModel[]) =>
|
|
16
64
|
models.map((model) => {
|
|
@@ -24,7 +72,12 @@ export function createOpenAiCompletionsApi(options?: {
|
|
|
24
72
|
};
|
|
25
73
|
}),
|
|
26
74
|
stream: (model, context, streamOptions) =>
|
|
27
|
-
stream(model, context, streamOptions as never),
|
|
28
|
-
streamSimple:
|
|
75
|
+
stream(model, context, withReasoningReplay(streamOptions) as never),
|
|
76
|
+
streamSimple: (model, context, simpleOptions) =>
|
|
77
|
+
(options?.streamSimple ?? streamSimple)(
|
|
78
|
+
model,
|
|
79
|
+
context,
|
|
80
|
+
withReasoningReplay(simpleOptions) as never,
|
|
81
|
+
),
|
|
29
82
|
};
|
|
30
83
|
}
|
|
@@ -308,21 +308,21 @@ const FAMILIES: [NeuralwattModelFamily, NeuralwattVariantSpec[]][] = [
|
|
|
308
308
|
{
|
|
309
309
|
id: "qwen3.6-35b",
|
|
310
310
|
name: "Qwen3.6 35B",
|
|
311
|
-
contextWindow:
|
|
311
|
+
contextWindow: 262128,
|
|
312
312
|
maxOutputTokens: null,
|
|
313
313
|
reasoning: true,
|
|
314
314
|
},
|
|
315
315
|
{
|
|
316
316
|
id: "qwen3.6-35b-fast",
|
|
317
317
|
name: "Qwen3.6 35B Fast",
|
|
318
|
-
contextWindow:
|
|
318
|
+
contextWindow: 262128,
|
|
319
319
|
maxOutputTokens: null,
|
|
320
320
|
reasoning: false,
|
|
321
321
|
},
|
|
322
322
|
{
|
|
323
323
|
id: "qwen3.6-35b-flex",
|
|
324
324
|
name: "Qwen3.6 35B (flex)",
|
|
325
|
-
contextWindow:
|
|
325
|
+
contextWindow: 262128,
|
|
326
326
|
maxOutputTokens: null,
|
|
327
327
|
reasoning: true,
|
|
328
328
|
costMultiplier: 0.65,
|