@aliou/pi-neuralwatt 0.15.4 → 0.16.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +12 -5
- package/extensions/provider/api/anthropic-messages.ts +114 -0
- package/extensions/provider/api/openai-completions.ts +30 -0
- package/extensions/provider/api/types.ts +24 -0
- package/extensions/provider/commands/settings/index.ts +35 -1
- package/extensions/provider/constants.ts +7 -0
- package/extensions/provider/index.ts +14 -1
- package/extensions/provider/models/build.ts +52 -10
- package/extensions/provider/models/catalog.ts +6 -45
- package/extensions/provider/models/public-models.ts +7 -0
- package/extensions/provider/provider.ts +42 -31
- package/extensions/provider/sse-quotas.ts +17 -4
- package/extensions/provider/stream-simple.ts +4 -3
- package/package.json +2 -1
- package/schema.json +22 -0
- package/src/config/defaults.ts +3 -0
- package/src/config/index.ts +6 -2
- package/src/config/loader.ts +15 -1
- package/src/config/migration/index.ts +0 -3
- package/src/config/types.ts +18 -0
- package/src/types/models-api.ts +9 -9
- package/src/config/migration/04-enable-aliases-for-legacy-users.ts +0 -41
package/README.md
CHANGED
|
@@ -47,6 +47,15 @@ Once installed, select `neuralwatt` as your provider and choose from available m
|
|
|
47
47
|
/model neuralwatt meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8
|
|
48
48
|
```
|
|
49
49
|
|
|
50
|
+
### API surface
|
|
51
|
+
|
|
52
|
+
Neuralwatt serves every model on two APIs. Pick one via `/neuralwatt:settings` → **API** (or set `provider.api` in the extension config):
|
|
53
|
+
|
|
54
|
+
- `openai-completions` (default) — the OpenAI-compatible `chat/completions` endpoint;
|
|
55
|
+
- `anthropic-messages` — the Anthropic-compatible `POST /v1/messages` endpoint (vLLM-backed), which streams native tool use and thinking blocks.
|
|
56
|
+
|
|
57
|
+
The setting swaps the whole provider (same model ids on both sides) and applies on `/reload`. Usage/cost accounting and quota tracking work on both surfaces: per-request quota headers only exist on chat-completions responses, while `/v1/messages` streams carry the same data as `: energy` / `: cost` SSE comments.
|
|
58
|
+
|
|
50
59
|
### Quota Command
|
|
51
60
|
|
|
52
61
|
Check your API usage at a glance:
|
|
@@ -74,12 +83,10 @@ When a Neuralwatt model is active, the footer status bar shows live quota usage
|
|
|
74
83
|
|
|
75
84
|
Configure features with `/neuralwatt:settings`:
|
|
76
85
|
|
|
86
|
+
- **API** — Choose between `openai-completions` (default) and `anthropic-messages`; applies on `/reload`
|
|
77
87
|
- **Quota command** — Show/hide `/neuralwatt:quota`
|
|
78
88
|
- **Quota warnings** — Enable/disable low quota notifications
|
|
79
89
|
- **Sub-bar integration** — Show/hide usage in status bar
|
|
80
|
-
- **Legacy model IDs** — Include deprecated model aliases
|
|
81
|
-
- **Alias model IDs** — Include active creator-scoped model aliases
|
|
82
|
-
- **Early access models** — Include pre-release models available only to the configured API key
|
|
83
90
|
|
|
84
91
|
The provider itself cannot be disabled — it is always loaded.
|
|
85
92
|
|
|
@@ -87,9 +94,9 @@ Configuration uses nested per-feature sections. Existing flat config files are m
|
|
|
87
94
|
|
|
88
95
|
### Model Refresh
|
|
89
96
|
|
|
90
|
-
Neuralwatt registers its public models without network access.
|
|
97
|
+
Neuralwatt registers its public models without network access. Opening `/model` refreshes the catalog from the API in the background (authenticated when an API key is configured). `pi update --models` forces an immediate refresh.
|
|
91
98
|
|
|
92
|
-
Pi stores the complete effective Neuralwatt catalog in `~/.pi/agent/models-store.json` for offline startup. Current hardcoded public
|
|
99
|
+
Pi stores the complete effective Neuralwatt catalog in `~/.pi/agent/models-store.json` for offline startup. Current hardcoded public definitions remain authoritative when cached models are restored.
|
|
93
100
|
|
|
94
101
|
## Adding or Updating Models
|
|
95
102
|
|
|
@@ -0,0 +1,114 @@
|
|
|
1
|
+
import type { StreamOptions } from "@earendil-works/pi-ai";
|
|
2
|
+
import { stream, streamSimple } from "@earendil-works/pi-ai/compat";
|
|
3
|
+
import {
|
|
4
|
+
NEURALWATT_BASE_URL,
|
|
5
|
+
NEURALWATT_PROVIDER_ID,
|
|
6
|
+
NEURALWATT_REQUEST_HEADERS,
|
|
7
|
+
} from "../constants";
|
|
8
|
+
import { buildAnthropicThinkingLevelMap } from "../models/build";
|
|
9
|
+
import type { NeuralwattModel } from "../models/catalog";
|
|
10
|
+
import type { AnyStreamSimple } from "../stream-simple";
|
|
11
|
+
import type { NeuralwattApiHandler } from "./types";
|
|
12
|
+
|
|
13
|
+
type AnthropicMessagesBody = {
|
|
14
|
+
thinking?: { type?: string };
|
|
15
|
+
output_config?: { effort?: string };
|
|
16
|
+
chat_template_kwargs?: Record<string, unknown>;
|
|
17
|
+
[key: string]: unknown;
|
|
18
|
+
};
|
|
19
|
+
|
|
20
|
+
// Outside vLLM's effort enum (HTTP 400); both mean "reasoning off".
|
|
21
|
+
const EFFORT_OFF_VALUES = new Set(["none", "minimal"]);
|
|
22
|
+
|
|
23
|
+
function applyReasoningOff(body: AnthropicMessagesBody): AnthropicMessagesBody {
|
|
24
|
+
delete body.thinking;
|
|
25
|
+
delete body.output_config;
|
|
26
|
+
body.chat_template_kwargs = {
|
|
27
|
+
...(body.chat_template_kwargs ?? {}),
|
|
28
|
+
enable_thinking: false,
|
|
29
|
+
};
|
|
30
|
+
return body;
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
/**
|
|
34
|
+
* vLLM's reasoning controls differ from first-party Anthropic: positive levels
|
|
35
|
+
* go through `output_config.effort` (adaptive path, forced via compat) while
|
|
36
|
+
* `thinking:{type:"disabled"}` is accepted but ignored, so off is expressed
|
|
37
|
+
* through the chat-template kwarg instead.
|
|
38
|
+
*/
|
|
39
|
+
function makeReasoningInjector(
|
|
40
|
+
upstream?: StreamOptions["onPayload"],
|
|
41
|
+
): NonNullable<StreamOptions["onPayload"]> {
|
|
42
|
+
return async (payload, model) => {
|
|
43
|
+
const next = await upstream?.(payload, model);
|
|
44
|
+
const body = (next !== undefined ? next : payload) as AnthropicMessagesBody;
|
|
45
|
+
|
|
46
|
+
if (body.thinking?.type === "disabled") {
|
|
47
|
+
return applyReasoningOff(body);
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
const effort = body.output_config?.effort;
|
|
51
|
+
if (effort && EFFORT_OFF_VALUES.has(effort)) {
|
|
52
|
+
return applyReasoningOff(body);
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
return body;
|
|
56
|
+
};
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
// The Anthropic SDK appends `/v1/messages` to the client base URL.
|
|
60
|
+
function toMessagesBaseUrl(baseUrl: string): string {
|
|
61
|
+
return baseUrl.replace(/\/v1\/?$/, "");
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
function stampAnthropicModels(models: NeuralwattModel[]) {
|
|
65
|
+
return models.map((model) => {
|
|
66
|
+
const { reasoningContract, ...compiled } = model;
|
|
67
|
+
// No retained contract: the identity map is the alias-free special case.
|
|
68
|
+
const thinkingLevelMap = model.reasoning
|
|
69
|
+
? reasoningContract
|
|
70
|
+
? buildAnthropicThinkingLevelMap(reasoningContract)
|
|
71
|
+
: model.thinkingLevelMap
|
|
72
|
+
? { ...model.thinkingLevelMap }
|
|
73
|
+
: undefined
|
|
74
|
+
: undefined;
|
|
75
|
+
|
|
76
|
+
return {
|
|
77
|
+
...compiled,
|
|
78
|
+
api: "anthropic-messages" as const,
|
|
79
|
+
provider: NEURALWATT_PROVIDER_ID,
|
|
80
|
+
baseUrl: toMessagesBaseUrl(model.baseUrl ?? NEURALWATT_BASE_URL),
|
|
81
|
+
headers: NEURALWATT_REQUEST_HEADERS,
|
|
82
|
+
compat: {
|
|
83
|
+
forceAdaptiveThinking: true,
|
|
84
|
+
supportsTemperature: true,
|
|
85
|
+
supportsStrictTools: false,
|
|
86
|
+
supportsCacheControlOnTools: false,
|
|
87
|
+
},
|
|
88
|
+
...(thinkingLevelMap ? { thinkingLevelMap } : {}),
|
|
89
|
+
};
|
|
90
|
+
});
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
export function createAnthropicMessagesApi(options?: {
|
|
94
|
+
streamSimple?: AnyStreamSimple;
|
|
95
|
+
}): NeuralwattApiHandler {
|
|
96
|
+
const withReasoning = (options?: {
|
|
97
|
+
onPayload?: StreamOptions["onPayload"];
|
|
98
|
+
}) => ({
|
|
99
|
+
...options,
|
|
100
|
+
onPayload: makeReasoningInjector(options?.onPayload),
|
|
101
|
+
});
|
|
102
|
+
|
|
103
|
+
return {
|
|
104
|
+
stampModels: stampAnthropicModels,
|
|
105
|
+
stream: (model, context, streamOptions) =>
|
|
106
|
+
stream(model, context, withReasoning(streamOptions) as never),
|
|
107
|
+
streamSimple: (model, context, simpleOptions) =>
|
|
108
|
+
(options?.streamSimple ?? streamSimple)(
|
|
109
|
+
model,
|
|
110
|
+
context,
|
|
111
|
+
withReasoning(simpleOptions) as never,
|
|
112
|
+
),
|
|
113
|
+
};
|
|
114
|
+
}
|
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
import { stream, streamSimple } from "@earendil-works/pi-ai/compat";
|
|
2
|
+
import {
|
|
3
|
+
NEURALWATT_BASE_URL,
|
|
4
|
+
NEURALWATT_PROVIDER_ID,
|
|
5
|
+
NEURALWATT_REQUEST_HEADERS,
|
|
6
|
+
} from "../constants";
|
|
7
|
+
import type { NeuralwattModel } from "../models/catalog";
|
|
8
|
+
import type { AnyStreamSimple } from "../stream-simple";
|
|
9
|
+
import type { NeuralwattApiHandler } from "./types";
|
|
10
|
+
|
|
11
|
+
export function createOpenAiCompletionsApi(options?: {
|
|
12
|
+
streamSimple?: AnyStreamSimple;
|
|
13
|
+
}): NeuralwattApiHandler {
|
|
14
|
+
return {
|
|
15
|
+
stampModels: (models: NeuralwattModel[]) =>
|
|
16
|
+
models.map((model) => {
|
|
17
|
+
const { reasoningContract: _reasoningContract, ...compiled } = model;
|
|
18
|
+
return {
|
|
19
|
+
...compiled,
|
|
20
|
+
api: "openai-completions" as const,
|
|
21
|
+
provider: NEURALWATT_PROVIDER_ID,
|
|
22
|
+
baseUrl: model.baseUrl ?? NEURALWATT_BASE_URL,
|
|
23
|
+
headers: NEURALWATT_REQUEST_HEADERS,
|
|
24
|
+
};
|
|
25
|
+
}),
|
|
26
|
+
stream: (model, context, streamOptions) =>
|
|
27
|
+
stream(model, context, streamOptions as never),
|
|
28
|
+
streamSimple: options?.streamSimple ?? streamSimple,
|
|
29
|
+
};
|
|
30
|
+
}
|
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
import type {
|
|
2
|
+
Api,
|
|
3
|
+
AssistantMessageEventStream,
|
|
4
|
+
Context,
|
|
5
|
+
Model,
|
|
6
|
+
SimpleStreamOptions,
|
|
7
|
+
StreamOptions,
|
|
8
|
+
} from "@earendil-works/pi-ai";
|
|
9
|
+
import type { NeuralwattModel } from "../models/catalog";
|
|
10
|
+
|
|
11
|
+
/** One Neuralwatt API surface: model stamping plus submission plumbing. */
|
|
12
|
+
export interface NeuralwattApiHandler {
|
|
13
|
+
stampModels(models: NeuralwattModel[]): Model<Api>[];
|
|
14
|
+
stream(
|
|
15
|
+
model: Model<Api>,
|
|
16
|
+
context: Context,
|
|
17
|
+
options?: StreamOptions,
|
|
18
|
+
): AssistantMessageEventStream;
|
|
19
|
+
streamSimple(
|
|
20
|
+
model: Model<Api>,
|
|
21
|
+
context: Context,
|
|
22
|
+
options?: SimpleStreamOptions,
|
|
23
|
+
): AssistantMessageEventStream;
|
|
24
|
+
}
|
|
@@ -6,6 +6,7 @@ import type { ExtensionAPI } from "@earendil-works/pi-coding-agent";
|
|
|
6
6
|
import type { SettingItem } from "@earendil-works/pi-tui";
|
|
7
7
|
import {
|
|
8
8
|
configLoader,
|
|
9
|
+
type NeuralwattApi,
|
|
9
10
|
type NeuralwattConfig,
|
|
10
11
|
type ResolvedNeuralwattConfig,
|
|
11
12
|
} from "../../../../src/config";
|
|
@@ -61,6 +62,9 @@ export function registerNeuralwattSettings(
|
|
|
61
62
|
options: RegisterNeuralwattSettingsOptions,
|
|
62
63
|
): void {
|
|
63
64
|
const { getLoadedFeatures } = options;
|
|
65
|
+
// The provider stamps `provider.api` at extension load; a saved change only
|
|
66
|
+
// reaches it after `/reload`.
|
|
67
|
+
let pendingApi: NeuralwattApi | undefined;
|
|
64
68
|
|
|
65
69
|
registerSettingsCommand<NeuralwattConfig, ResolvedNeuralwattConfig>(pi, {
|
|
66
70
|
commandName: "neuralwatt:settings",
|
|
@@ -69,6 +73,19 @@ export function registerNeuralwattSettings(
|
|
|
69
73
|
buildSections: (tabConfig, resolved): SettingsSection[] => {
|
|
70
74
|
const loaded = getLoadedFeatures();
|
|
71
75
|
return [
|
|
76
|
+
{
|
|
77
|
+
label: "Provider",
|
|
78
|
+
items: [
|
|
79
|
+
{
|
|
80
|
+
id: "api",
|
|
81
|
+
label: "API",
|
|
82
|
+
description:
|
|
83
|
+
"Serve models via the OpenAI-compatible chat/completions endpoint or the Anthropic-compatible /v1/messages endpoint",
|
|
84
|
+
currentValue: tabConfig?.provider?.api ?? resolved.provider.api,
|
|
85
|
+
values: ["openai-completions", "anthropic-messages"],
|
|
86
|
+
},
|
|
87
|
+
],
|
|
88
|
+
},
|
|
72
89
|
{
|
|
73
90
|
label: "Features",
|
|
74
91
|
items: [
|
|
@@ -107,6 +124,20 @@ export function registerNeuralwattSettings(
|
|
|
107
124
|
];
|
|
108
125
|
},
|
|
109
126
|
onSettingChange: (id, newValue, config) => {
|
|
127
|
+
if (id === "api") {
|
|
128
|
+
if (
|
|
129
|
+
newValue !== "openai-completions" &&
|
|
130
|
+
newValue !== "anthropic-messages"
|
|
131
|
+
) {
|
|
132
|
+
return null;
|
|
133
|
+
}
|
|
134
|
+
pendingApi = newValue;
|
|
135
|
+
return {
|
|
136
|
+
...config,
|
|
137
|
+
provider: { ...config.provider, api: newValue },
|
|
138
|
+
};
|
|
139
|
+
}
|
|
140
|
+
|
|
110
141
|
if (!getLoadedFeatures().has(id as NeuralwattFeatureId)) {
|
|
111
142
|
return null;
|
|
112
143
|
}
|
|
@@ -132,8 +163,11 @@ export function registerNeuralwattSettings(
|
|
|
132
163
|
return null;
|
|
133
164
|
}
|
|
134
165
|
},
|
|
135
|
-
onSave: async () => {
|
|
166
|
+
onSave: async (ctx) => {
|
|
136
167
|
emitConfigUpdated(pi);
|
|
168
|
+
if (pendingApi === undefined) return;
|
|
169
|
+
pendingApi = undefined;
|
|
170
|
+
ctx.ui.notify("Run /reload to apply the new API", "info");
|
|
137
171
|
},
|
|
138
172
|
});
|
|
139
173
|
}
|
|
@@ -0,0 +1,7 @@
|
|
|
1
|
+
export const NEURALWATT_PROVIDER_ID = "neuralwatt";
|
|
2
|
+
export const NEURALWATT_BASE_URL = "https://api.neuralwatt.com/v1";
|
|
3
|
+
export const NEURALWATT_API_KEY_ENV = "NEURALWATT_API_KEY";
|
|
4
|
+
export const NEURALWATT_REQUEST_HEADERS = {
|
|
5
|
+
Referer: "https://pi.dev",
|
|
6
|
+
"X-Title": "npm:@aliou/pi-neuralwatt",
|
|
7
|
+
};
|
|
@@ -55,6 +55,15 @@ function registerNeuralwattProvider(
|
|
|
55
55
|
) as never)
|
|
56
56
|
: undefined;
|
|
57
57
|
|
|
58
|
+
const messagesApiProvider = getApiProvider("anthropic-messages");
|
|
59
|
+
const messagesBaseStreamSimple = messagesApiProvider?.streamSimple;
|
|
60
|
+
const messagesStreamSimple = messagesBaseStreamSimple
|
|
61
|
+
? (wrapNeuralwattStreamSimple(
|
|
62
|
+
messagesBaseStreamSimple as never,
|
|
63
|
+
onSseQuota,
|
|
64
|
+
) as never)
|
|
65
|
+
: undefined;
|
|
66
|
+
|
|
58
67
|
pi.registerProvider(
|
|
59
68
|
createNeuralwattProvider(
|
|
60
69
|
staticModels,
|
|
@@ -65,7 +74,11 @@ function registerNeuralwattProvider(
|
|
|
65
74
|
}
|
|
66
75
|
return result.data;
|
|
67
76
|
},
|
|
68
|
-
|
|
77
|
+
{
|
|
78
|
+
api: configLoader.getConfig().provider.api,
|
|
79
|
+
openAiStreamSimple: streamSimple,
|
|
80
|
+
messagesStreamSimple,
|
|
81
|
+
},
|
|
69
82
|
),
|
|
70
83
|
);
|
|
71
84
|
}
|
|
@@ -8,6 +8,15 @@ export type ThinkingLevelMap = NonNullable<
|
|
|
8
8
|
ProviderModelConfig["thinkingLevelMap"]
|
|
9
9
|
>;
|
|
10
10
|
|
|
11
|
+
/**
|
|
12
|
+
* A compiled provider model plus the reasoning contract it was compiled from,
|
|
13
|
+
* retained for anthropic-messages map derivation. Rides the models store
|
|
14
|
+
* (JSON passthrough); stripped from stamped runtime models.
|
|
15
|
+
*/
|
|
16
|
+
export type NeuralwattCompiledModel = ProviderModelConfig & {
|
|
17
|
+
reasoningContract?: NeuralwattReasoningMapSource;
|
|
18
|
+
};
|
|
19
|
+
|
|
11
20
|
export interface NeuralwattCost {
|
|
12
21
|
input: number;
|
|
13
22
|
output: number;
|
|
@@ -56,7 +65,7 @@ export interface NeuralwattVariantSpec {
|
|
|
56
65
|
*/
|
|
57
66
|
export type NeuralwattReasoningMapSource = Pick<
|
|
58
67
|
NeuralwattApiModelReasoning,
|
|
59
|
-
"supported_efforts" | "mandatory"
|
|
68
|
+
"supported_efforts" | "mandatory" | "effort_aliases"
|
|
60
69
|
>;
|
|
61
70
|
|
|
62
71
|
/**
|
|
@@ -71,9 +80,9 @@ export type NeuralwattReasoningMapSource = Pick<
|
|
|
71
80
|
* exposes none), falls back to a conservative `high`-only map with `off: null`,
|
|
72
81
|
* matching the upstream binary thinking toggle.
|
|
73
82
|
*
|
|
74
|
-
* `
|
|
75
|
-
*
|
|
76
|
-
*
|
|
83
|
+
* `effort_aliases` is deliberately ignored here (the openai-completions
|
|
84
|
+
* gateway aliases unsupported efforts server-side); it is consumed by the
|
|
85
|
+
* anthropic-messages map below.
|
|
77
86
|
*/
|
|
78
87
|
export function buildThinkingLevelMap(
|
|
79
88
|
reasoning: NeuralwattReasoningMapSource | undefined,
|
|
@@ -96,6 +105,39 @@ export function buildThinkingLevelMap(
|
|
|
96
105
|
};
|
|
97
106
|
}
|
|
98
107
|
|
|
108
|
+
/**
|
|
109
|
+
* Thinking level map for the anthropic-messages surface. vLLM's
|
|
110
|
+
* `output_config.effort` accepts only the model's native efforts, so
|
|
111
|
+
* unsupported Pi levels resolve through `effort_aliases` (or `null`). A level
|
|
112
|
+
* may resolve to `"none"` — off on this surface, handled by the payload
|
|
113
|
+
* injector in `api/anthropic-messages.ts`.
|
|
114
|
+
*/
|
|
115
|
+
export function buildAnthropicThinkingLevelMap(
|
|
116
|
+
reasoning: NeuralwattReasoningMapSource | undefined,
|
|
117
|
+
): ThinkingLevelMap {
|
|
118
|
+
const supported = new Set<string>(reasoning?.supported_efforts ?? ["high"]);
|
|
119
|
+
const mandatory = reasoning?.mandatory ?? true;
|
|
120
|
+
const aliases = reasoning?.effort_aliases ?? {};
|
|
121
|
+
|
|
122
|
+
const resolve = (level: string): string | null => {
|
|
123
|
+
if (supported.has(level)) return level;
|
|
124
|
+
const alias = aliases[level as keyof typeof aliases];
|
|
125
|
+
return alias && supported.has(alias) ? alias : null;
|
|
126
|
+
};
|
|
127
|
+
|
|
128
|
+
return {
|
|
129
|
+
// "none" is a marker so pi-ai enables the off path (off !== null);
|
|
130
|
+
// vLLM rejects it on the wire, so the injector never sends it verbatim.
|
|
131
|
+
off: !mandatory && supported.has("none") ? "none" : null,
|
|
132
|
+
minimal: resolve("minimal"),
|
|
133
|
+
low: resolve("low"),
|
|
134
|
+
medium: resolve("medium"),
|
|
135
|
+
high: resolve("high"),
|
|
136
|
+
xhigh: resolve("xhigh"),
|
|
137
|
+
max: resolve("max"),
|
|
138
|
+
};
|
|
139
|
+
}
|
|
140
|
+
|
|
99
141
|
/**
|
|
100
142
|
* Neuralwatt reports `max_output_tokens: null` for models whose output is only
|
|
101
143
|
* bounded by the context window. Some models incorrectly report 0; treat 0
|
|
@@ -112,7 +154,7 @@ export function resolveMaxTokens(
|
|
|
112
154
|
export function buildNeuralwattModel(
|
|
113
155
|
family: NeuralwattModelFamily,
|
|
114
156
|
variant: NeuralwattVariantSpec,
|
|
115
|
-
):
|
|
157
|
+
): NeuralwattCompiledModel {
|
|
116
158
|
const vision = variant.vision ?? family.vision;
|
|
117
159
|
|
|
118
160
|
const compat: NonNullable<ProviderModelConfig["compat"]> = {
|
|
@@ -127,7 +169,7 @@ export function buildNeuralwattModel(
|
|
|
127
169
|
const scale = (value: number): number =>
|
|
128
170
|
multiplier === 1 ? value : Number((value * multiplier).toFixed(6));
|
|
129
171
|
|
|
130
|
-
const model:
|
|
172
|
+
const model: NeuralwattCompiledModel = {
|
|
131
173
|
id: variant.id,
|
|
132
174
|
name: variant.name,
|
|
133
175
|
reasoning: variant.reasoning,
|
|
@@ -144,14 +186,14 @@ export function buildNeuralwattModel(
|
|
|
144
186
|
};
|
|
145
187
|
|
|
146
188
|
if (variant.reasoning) {
|
|
189
|
+
const contract = variant.reasoningMetadata ?? family.reasoningMetadata;
|
|
147
190
|
// Clone so variants never share a family map instance. The map is derived
|
|
148
191
|
// from the API reasoning contract; missing metadata falls back to a
|
|
149
192
|
// high-only map rather than throwing.
|
|
150
193
|
model.thinkingLevelMap = {
|
|
151
|
-
...buildThinkingLevelMap(
|
|
152
|
-
variant.reasoningMetadata ?? family.reasoningMetadata,
|
|
153
|
-
),
|
|
194
|
+
...buildThinkingLevelMap(contract),
|
|
154
195
|
};
|
|
196
|
+
model.reasoningContract = contract;
|
|
155
197
|
}
|
|
156
198
|
|
|
157
199
|
return model;
|
|
@@ -160,6 +202,6 @@ export function buildNeuralwattModel(
|
|
|
160
202
|
export function buildNeuralwattFamily(
|
|
161
203
|
family: NeuralwattModelFamily,
|
|
162
204
|
variants: NeuralwattVariantSpec[],
|
|
163
|
-
):
|
|
205
|
+
): NeuralwattCompiledModel[] {
|
|
164
206
|
return variants.map((variant) => buildNeuralwattModel(family, variant));
|
|
165
207
|
}
|
|
@@ -2,12 +2,13 @@ import type { ProviderModelConfig } from "@earendil-works/pi-coding-agent";
|
|
|
2
2
|
import type { NeuralwattApiModel } from "../../../src/types/models-api";
|
|
3
3
|
import {
|
|
4
4
|
buildThinkingLevelMap,
|
|
5
|
+
type NeuralwattCompiledModel,
|
|
5
6
|
resolveMaxTokens,
|
|
6
7
|
type ThinkingLevelMap,
|
|
7
8
|
} from "./build";
|
|
8
9
|
import { NEURALWATT_MODELS } from "./public-models";
|
|
9
10
|
|
|
10
|
-
export type NeuralwattModel =
|
|
11
|
+
export type NeuralwattModel = NeuralwattCompiledModel;
|
|
11
12
|
|
|
12
13
|
// Chat-template thinking: the API exposes a `reasoning` block, but the
|
|
13
14
|
// underlying mechanism is chat_template_kwargs, so Pi needs the mapping.
|
|
@@ -22,16 +23,6 @@ const COMPAT_OVERRIDES: Partial<
|
|
|
22
23
|
},
|
|
23
24
|
};
|
|
24
25
|
|
|
25
|
-
const HARDCODED_ALIASES: Record<string, string> = {
|
|
26
|
-
"moonshotai/Kimi-K2.7-Code": "kimi-k2.7-code",
|
|
27
|
-
"Qwen/Qwen3.6-35B-A3B": "qwen3.6-35b",
|
|
28
|
-
"deepseek-ai/DeepSeek-V4-Flash": "deepseek-v4-flash",
|
|
29
|
-
};
|
|
30
|
-
|
|
31
|
-
function isVariantId(id: string): boolean {
|
|
32
|
-
return id.includes("-fast") || id.includes("-flex") || id.includes("-short");
|
|
33
|
-
}
|
|
34
|
-
|
|
35
26
|
function apiModelToProviderModel(model: NeuralwattApiModel): NeuralwattModel {
|
|
36
27
|
const meta = model.metadata;
|
|
37
28
|
if (!meta)
|
|
@@ -72,44 +63,14 @@ function apiModelToProviderModel(model: NeuralwattApiModel): NeuralwattModel {
|
|
|
72
63
|
result.thinkingLevelMap = buildThinkingLevelMap(
|
|
73
64
|
meta.reasoning,
|
|
74
65
|
) as ThinkingLevelMap;
|
|
66
|
+
// Kept for anthropic-messages stamping, which resolves levels through
|
|
67
|
+
// `effort_aliases`.
|
|
68
|
+
result.reasoningContract = meta.reasoning;
|
|
75
69
|
}
|
|
76
70
|
|
|
77
71
|
return result;
|
|
78
72
|
}
|
|
79
73
|
|
|
80
|
-
function buildAliases(
|
|
81
|
-
models: NeuralwattModel[],
|
|
82
|
-
apiModels: readonly NeuralwattApiModel[],
|
|
83
|
-
): NeuralwattModel[] {
|
|
84
|
-
const existingIds = new Set(models.map((m) => m.id));
|
|
85
|
-
const aliases: NeuralwattModel[] = [];
|
|
86
|
-
const seen = new Set<string>();
|
|
87
|
-
|
|
88
|
-
const addAlias = (aliasId: string, canonicalId: string): void => {
|
|
89
|
-
if (seen.has(aliasId) || existingIds.has(aliasId)) return;
|
|
90
|
-
const canonical = models.find((m) => m.id === canonicalId);
|
|
91
|
-
if (!canonical) return;
|
|
92
|
-
seen.add(aliasId);
|
|
93
|
-
aliases.push({
|
|
94
|
-
...canonical,
|
|
95
|
-
id: aliasId,
|
|
96
|
-
name: `${canonical.name} (alias ID)`,
|
|
97
|
-
});
|
|
98
|
-
};
|
|
99
|
-
|
|
100
|
-
for (const [aliasId, canonicalId] of Object.entries(HARDCODED_ALIASES)) {
|
|
101
|
-
addAlias(aliasId, canonicalId);
|
|
102
|
-
}
|
|
103
|
-
|
|
104
|
-
for (const apiModel of apiModels) {
|
|
105
|
-
const hfId = apiModel.metadata?.huggingface_id;
|
|
106
|
-
if (!hfId || hfId === apiModel.id || isVariantId(apiModel.id)) continue;
|
|
107
|
-
addAlias(hfId, apiModel.id);
|
|
108
|
-
}
|
|
109
|
-
|
|
110
|
-
return aliases;
|
|
111
|
-
}
|
|
112
|
-
|
|
113
74
|
export function buildNeuralwattProviderModels(): NeuralwattModel[] {
|
|
114
75
|
return NEURALWATT_MODELS.map((model) => ({ ...model }));
|
|
115
76
|
}
|
|
@@ -130,7 +91,7 @@ export function buildNeuralwattProviderModelsFromApi(
|
|
|
130
91
|
),
|
|
131
92
|
)
|
|
132
93
|
.map(apiModelToProviderModel);
|
|
133
|
-
return
|
|
94
|
+
return models;
|
|
134
95
|
}
|
|
135
96
|
|
|
136
97
|
export function buildNeuralwattProviderModelsFromStore(
|
|
@@ -127,6 +127,13 @@ const FAMILIES: [NeuralwattModelFamily, NeuralwattVariantSpec[]][] = [
|
|
|
127
127
|
reasoning: true,
|
|
128
128
|
costMultiplier: 0.65,
|
|
129
129
|
},
|
|
130
|
+
{
|
|
131
|
+
id: "deepseek-v4-flash-speed",
|
|
132
|
+
name: "DeepSeek V4 Flash (Speed)",
|
|
133
|
+
contextWindow: 1048560,
|
|
134
|
+
maxOutputTokens: 65536,
|
|
135
|
+
reasoning: true,
|
|
136
|
+
},
|
|
130
137
|
],
|
|
131
138
|
],
|
|
132
139
|
[
|
|
@@ -1,10 +1,14 @@
|
|
|
1
|
-
import type {
|
|
2
|
-
|
|
3
|
-
|
|
4
|
-
|
|
5
|
-
|
|
6
|
-
|
|
7
|
-
|
|
1
|
+
import type { Provider } from "@earendil-works/pi-ai";
|
|
2
|
+
import type { NeuralwattApi } from "../../src/config";
|
|
3
|
+
import { createAnthropicMessagesApi } from "./api/anthropic-messages";
|
|
4
|
+
import { createOpenAiCompletionsApi } from "./api/openai-completions";
|
|
5
|
+
import type { NeuralwattApiHandler } from "./api/types";
|
|
6
|
+
import {
|
|
7
|
+
NEURALWATT_API_KEY_ENV,
|
|
8
|
+
NEURALWATT_BASE_URL,
|
|
9
|
+
NEURALWATT_PROVIDER_ID,
|
|
10
|
+
NEURALWATT_REQUEST_HEADERS,
|
|
11
|
+
} from "./constants";
|
|
8
12
|
import type { NeuralwattModel } from "./models/catalog";
|
|
9
13
|
import {
|
|
10
14
|
buildNeuralwattProviderModelsFromApi,
|
|
@@ -16,33 +20,39 @@ import {
|
|
|
16
20
|
} from "./models/refresh";
|
|
17
21
|
import type { AnyStreamSimple } from "./stream-simple";
|
|
18
22
|
|
|
19
|
-
export
|
|
20
|
-
export const NEURALWATT_BASE_URL = "https://api.neuralwatt.com/v1";
|
|
21
|
-
export const NEURALWATT_API_KEY_ENV = "NEURALWATT_API_KEY";
|
|
22
|
-
|
|
23
|
-
const NEURALWATT_REQUEST_HEADERS = {
|
|
24
|
-
Referer: "https://pi.dev",
|
|
25
|
-
"X-Title": "npm:@aliou/pi-neuralwatt",
|
|
26
|
-
};
|
|
23
|
+
export { NEURALWATT_API_KEY_ENV, NEURALWATT_BASE_URL, NEURALWATT_PROVIDER_ID };
|
|
27
24
|
|
|
28
|
-
|
|
25
|
+
export interface NeuralwattProviderOptions {
|
|
26
|
+
/** Active API surface; resolved once. Changes need a `/reload`. */
|
|
27
|
+
api?: NeuralwattApi;
|
|
28
|
+
openAiStreamSimple?: AnyStreamSimple;
|
|
29
|
+
messagesStreamSimple?: AnyStreamSimple;
|
|
30
|
+
}
|
|
29
31
|
|
|
30
|
-
function
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
|
|
32
|
+
function createApiHandler(
|
|
33
|
+
api: NeuralwattApi,
|
|
34
|
+
options?: NeuralwattProviderOptions,
|
|
35
|
+
): NeuralwattApiHandler {
|
|
36
|
+
if (api === "anthropic-messages") {
|
|
37
|
+
return createAnthropicMessagesApi({
|
|
38
|
+
streamSimple: options?.messagesStreamSimple,
|
|
39
|
+
});
|
|
40
|
+
}
|
|
41
|
+
return createOpenAiCompletionsApi({
|
|
42
|
+
streamSimple: options?.openAiStreamSimple,
|
|
43
|
+
});
|
|
38
44
|
}
|
|
39
45
|
|
|
40
46
|
export function createNeuralwattProvider(
|
|
41
47
|
staticModels: NeuralwattModel[],
|
|
42
48
|
fetchApiModels: FetchNeuralwattApiModels,
|
|
43
|
-
|
|
49
|
+
options?: NeuralwattProviderOptions,
|
|
44
50
|
): Provider {
|
|
45
|
-
|
|
51
|
+
const handler = createApiHandler(
|
|
52
|
+
options?.api ?? "openai-completions",
|
|
53
|
+
options,
|
|
54
|
+
);
|
|
55
|
+
let canonicalModels = staticModels;
|
|
46
56
|
const refreshCatalog = createNeuralwattRefreshModels(
|
|
47
57
|
staticModels,
|
|
48
58
|
fetchApiModels,
|
|
@@ -92,17 +102,18 @@ export function createNeuralwattProvider(
|
|
|
92
102
|
},
|
|
93
103
|
},
|
|
94
104
|
},
|
|
95
|
-
getModels: () =>
|
|
105
|
+
getModels: () => handler.stampModels(canonicalModels),
|
|
96
106
|
refreshModels: async (context) => {
|
|
97
107
|
const models = await refreshCatalog(context);
|
|
98
108
|
await context.publish({
|
|
99
109
|
update: () => {
|
|
100
|
-
|
|
110
|
+
canonicalModels = models;
|
|
101
111
|
},
|
|
102
112
|
});
|
|
103
113
|
},
|
|
104
|
-
stream: (model, context,
|
|
105
|
-
stream(model, context,
|
|
106
|
-
streamSimple:
|
|
114
|
+
stream: (model, context, streamOptions) =>
|
|
115
|
+
handler.stream(model, context, streamOptions as never),
|
|
116
|
+
streamSimple: (model, context, simpleOptions) =>
|
|
117
|
+
handler.streamSimple(model, context, simpleOptions),
|
|
107
118
|
};
|
|
108
119
|
}
|
|
@@ -31,13 +31,26 @@ export function updateQuotasFromSseComment(
|
|
|
31
31
|
if (trimmed.startsWith(": cost ")) {
|
|
32
32
|
const cost = JSON.parse(trimmed.slice(7)) as {
|
|
33
33
|
request_cost_usd?: number;
|
|
34
|
+
allowance_remaining_usd?: number;
|
|
34
35
|
};
|
|
35
36
|
const requestCostUsd = cost.request_cost_usd ?? 0;
|
|
36
37
|
if (requestCostUsd <= 0) return quotas;
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
38
|
+
// The absolute allowance in the comment is fresher than the locally
|
|
39
|
+
// tracked total.
|
|
40
|
+
if (
|
|
41
|
+
typeof cost.allowance_remaining_usd === "number" &&
|
|
42
|
+
Number.isFinite(cost.allowance_remaining_usd)
|
|
43
|
+
) {
|
|
44
|
+
next.balance.credits_remaining_usd = Math.max(
|
|
45
|
+
0,
|
|
46
|
+
cost.allowance_remaining_usd,
|
|
47
|
+
);
|
|
48
|
+
} else {
|
|
49
|
+
next.balance.credits_remaining_usd = Math.max(
|
|
50
|
+
0,
|
|
51
|
+
next.balance.credits_remaining_usd - requestCostUsd,
|
|
52
|
+
);
|
|
53
|
+
}
|
|
41
54
|
next.balance.credits_used_usd += requestCostUsd;
|
|
42
55
|
next.usage.current_month.cost_usd += requestCostUsd;
|
|
43
56
|
next.usage.lifetime.cost_usd += requestCostUsd;
|
|
@@ -35,7 +35,7 @@ function headersToRecord(headers: Headers): Record<string, string> {
|
|
|
35
35
|
return record;
|
|
36
36
|
}
|
|
37
37
|
|
|
38
|
-
function
|
|
38
|
+
function isProviderStreamUrl(
|
|
39
39
|
input: RequestInfo | URL,
|
|
40
40
|
providerOrigin: string,
|
|
41
41
|
): boolean {
|
|
@@ -50,7 +50,8 @@ function isProviderChatCompletionsUrl(
|
|
|
50
50
|
const url = new URL(rawUrl);
|
|
51
51
|
return (
|
|
52
52
|
url.origin === providerOrigin &&
|
|
53
|
-
url.pathname.endsWith("/chat/completions")
|
|
53
|
+
(url.pathname.endsWith("/chat/completions") ||
|
|
54
|
+
url.pathname.endsWith("/messages"))
|
|
54
55
|
);
|
|
55
56
|
} catch {
|
|
56
57
|
return false;
|
|
@@ -96,7 +97,7 @@ export function wrapNeuralwattStreamSimple(
|
|
|
96
97
|
const wrappedFetch: typeof fetch = async (input, init) => {
|
|
97
98
|
const response = await originalFetch(input, init);
|
|
98
99
|
|
|
99
|
-
if (!
|
|
100
|
+
if (!isProviderStreamUrl(input, providerOrigin)) return response;
|
|
100
101
|
|
|
101
102
|
const headers = headersToRecord(response.headers);
|
|
102
103
|
if (response.status === 429) {
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@aliou/pi-neuralwatt",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.16.0",
|
|
4
4
|
"license": "MIT",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"private": false,
|
|
@@ -69,6 +69,7 @@
|
|
|
69
69
|
"test": "vitest run",
|
|
70
70
|
"test:watch": "vitest",
|
|
71
71
|
"check:models": "tsx scripts/check-models.ts",
|
|
72
|
+
"check:changesets": "tsx scripts/check-changesets.ts",
|
|
72
73
|
"gen:schema": "pi-settings-schema -p src/config/types.ts -t NeuralwattConfig -o schema.json --version 0.12.0",
|
|
73
74
|
"check:schema": "pi-settings-schema -p src/config/types.ts -t NeuralwattConfig -o schema.json --version 0.12.0 --check",
|
|
74
75
|
"check:lockfile": "pnpm install --frozen-lockfile --ignore-scripts",
|
package/schema.json
CHANGED
|
@@ -20,6 +20,10 @@
|
|
|
20
20
|
"$ref": "#/definitions/NeuralwattSubBarIntegrationConfig",
|
|
21
21
|
"description": "Sub-bar/status-bar integration feature."
|
|
22
22
|
},
|
|
23
|
+
"provider": {
|
|
24
|
+
"$ref": "#/definitions/NeuralwattProviderConfig",
|
|
25
|
+
"description": "Provider behavior (API surface)."
|
|
26
|
+
},
|
|
23
27
|
"version": {
|
|
24
28
|
"anyOf": [
|
|
25
29
|
{
|
|
@@ -65,6 +69,24 @@
|
|
|
65
69
|
}
|
|
66
70
|
},
|
|
67
71
|
"additionalProperties": false
|
|
72
|
+
},
|
|
73
|
+
"NeuralwattProviderConfig": {
|
|
74
|
+
"type": "object",
|
|
75
|
+
"properties": {
|
|
76
|
+
"api": {
|
|
77
|
+
"$ref": "#/definitions/NeuralwattApi",
|
|
78
|
+
"description": "Which API serves model requests."
|
|
79
|
+
}
|
|
80
|
+
},
|
|
81
|
+
"additionalProperties": false
|
|
82
|
+
},
|
|
83
|
+
"NeuralwattApi": {
|
|
84
|
+
"type": "string",
|
|
85
|
+
"enum": [
|
|
86
|
+
"openai-completions",
|
|
87
|
+
"anthropic-messages"
|
|
88
|
+
],
|
|
89
|
+
"description": "Neuralwatt serves every chat model twice: on an OpenAI-compatible `chat/completions` endpoint and on a vLLM-backed Anthropic-compatible `POST /v1/messages` endpoint. Exactly one serves the provider at a time."
|
|
68
90
|
}
|
|
69
91
|
}
|
|
70
92
|
}
|
package/src/config/defaults.ts
CHANGED
package/src/config/index.ts
CHANGED
|
@@ -1,4 +1,8 @@
|
|
|
1
1
|
export { DEFAULT_CONFIG } from "./defaults";
|
|
2
|
-
export { configLoader } from "./loader";
|
|
2
|
+
export { configLoader, resolveApi } from "./loader";
|
|
3
3
|
export { migrations } from "./migration";
|
|
4
|
-
export type {
|
|
4
|
+
export type {
|
|
5
|
+
NeuralwattApi,
|
|
6
|
+
NeuralwattConfig,
|
|
7
|
+
ResolvedNeuralwattConfig,
|
|
8
|
+
} from "./types";
|
package/src/config/loader.ts
CHANGED
|
@@ -2,7 +2,11 @@ import { buildSchemaUrl, ConfigLoader } from "@aliou/pi-utils-settings";
|
|
|
2
2
|
import packageJson from "../../package.json";
|
|
3
3
|
import { DEFAULT_CONFIG } from "./defaults";
|
|
4
4
|
import { migrations } from "./migration";
|
|
5
|
-
import type {
|
|
5
|
+
import type {
|
|
6
|
+
NeuralwattApi,
|
|
7
|
+
NeuralwattConfig,
|
|
8
|
+
ResolvedNeuralwattConfig,
|
|
9
|
+
} from "./types";
|
|
6
10
|
|
|
7
11
|
/**
|
|
8
12
|
* Fill in every field the rest of the code reads. Migrations already normalized
|
|
@@ -27,9 +31,19 @@ function normalizeResolvedConfig(
|
|
|
27
31
|
config.subBarIntegration?.enabled ??
|
|
28
32
|
DEFAULT_CONFIG.subBarIntegration.enabled,
|
|
29
33
|
},
|
|
34
|
+
provider: {
|
|
35
|
+
api: resolveApi(config.provider?.api),
|
|
36
|
+
},
|
|
30
37
|
};
|
|
31
38
|
}
|
|
32
39
|
|
|
40
|
+
export function resolveApi(value: string | undefined): NeuralwattApi {
|
|
41
|
+
if (value === "anthropic-messages" || value === "openai-completions") {
|
|
42
|
+
return value;
|
|
43
|
+
}
|
|
44
|
+
return DEFAULT_CONFIG.provider.api;
|
|
45
|
+
}
|
|
46
|
+
|
|
33
47
|
export const configLoader = new ConfigLoader<
|
|
34
48
|
NeuralwattConfig,
|
|
35
49
|
ResolvedNeuralwattConfig
|
|
@@ -6,11 +6,9 @@ export {
|
|
|
6
6
|
flatToNestedConfigMigration,
|
|
7
7
|
} from "./02-flat-to-nested-config";
|
|
8
8
|
export { renameHiddenToEarlyAccessMigration } from "./03-rename-hidden-to-early-access";
|
|
9
|
-
export { enableAliasesForLegacyUsersMigration } from "./04-enable-aliases-for-legacy-users";
|
|
10
9
|
|
|
11
10
|
import { flatToNestedConfigMigration } from "./02-flat-to-nested-config";
|
|
12
11
|
import { renameHiddenToEarlyAccessMigration } from "./03-rename-hidden-to-early-access";
|
|
13
|
-
import { enableAliasesForLegacyUsersMigration } from "./04-enable-aliases-for-legacy-users";
|
|
14
12
|
|
|
15
13
|
// Each migration is typed against its own historical input shape. The loader
|
|
16
14
|
// applies them in sequence on the raw config record, so they are cast to the
|
|
@@ -18,5 +16,4 @@ import { enableAliasesForLegacyUsersMigration } from "./04-enable-aliases-for-le
|
|
|
18
16
|
export const migrations = [
|
|
19
17
|
flatToNestedConfigMigration,
|
|
20
18
|
renameHiddenToEarlyAccessMigration,
|
|
21
|
-
enableAliasesForLegacyUsersMigration,
|
|
22
19
|
] as unknown as Migration<NeuralwattConfig>[];
|
package/src/config/types.ts
CHANGED
|
@@ -13,6 +13,18 @@ export interface NeuralwattSubBarIntegrationConfig {
|
|
|
13
13
|
enabled?: boolean;
|
|
14
14
|
}
|
|
15
15
|
|
|
16
|
+
/**
|
|
17
|
+
* Neuralwatt serves every chat model twice: on an OpenAI-compatible
|
|
18
|
+
* `chat/completions` endpoint and on a vLLM-backed Anthropic-compatible
|
|
19
|
+
* `POST /v1/messages` endpoint. Exactly one serves the provider at a time.
|
|
20
|
+
*/
|
|
21
|
+
export type NeuralwattApi = "openai-completions" | "anthropic-messages";
|
|
22
|
+
|
|
23
|
+
export interface NeuralwattProviderConfig {
|
|
24
|
+
/** Which API serves model requests. */
|
|
25
|
+
api?: NeuralwattApi;
|
|
26
|
+
}
|
|
27
|
+
|
|
16
28
|
export interface NeuralwattConfig {
|
|
17
29
|
/** $schema URL for editor autocomplete. */
|
|
18
30
|
$schema?: string;
|
|
@@ -25,6 +37,9 @@ export interface NeuralwattConfig {
|
|
|
25
37
|
|
|
26
38
|
/** Sub-bar/status-bar integration feature. */
|
|
27
39
|
subBarIntegration?: NeuralwattSubBarIntegrationConfig;
|
|
40
|
+
|
|
41
|
+
/** Provider behavior (API surface). */
|
|
42
|
+
provider?: NeuralwattProviderConfig;
|
|
28
43
|
}
|
|
29
44
|
|
|
30
45
|
export interface ResolvedNeuralwattConfig {
|
|
@@ -37,4 +52,7 @@ export interface ResolvedNeuralwattConfig {
|
|
|
37
52
|
subBarIntegration: {
|
|
38
53
|
enabled: boolean;
|
|
39
54
|
};
|
|
55
|
+
provider: {
|
|
56
|
+
api: NeuralwattApi;
|
|
57
|
+
};
|
|
40
58
|
}
|
package/src/types/models-api.ts
CHANGED
|
@@ -38,14 +38,11 @@ export type NeuralwattReasoningEffort =
|
|
|
38
38
|
| "max";
|
|
39
39
|
|
|
40
40
|
/**
|
|
41
|
-
* Per-model reasoning contract from `/v1/models`.
|
|
42
|
-
*
|
|
43
|
-
*
|
|
44
|
-
*
|
|
45
|
-
*
|
|
46
|
-
* `default_effort` and `effort_aliases` are typed for fidelity but are not
|
|
47
|
-
* consumed — Pi has no default-reasoning field and we expose native efforts
|
|
48
|
-
* rather than aliasing unsupported ones.
|
|
41
|
+
* Per-model reasoning contract from `/v1/models`. The openai-completions
|
|
42
|
+
* thinking map uses `supported_efforts` by identity; the anthropic-messages
|
|
43
|
+
* map additionally resolves through `effort_aliases` (vLLM's effort enum
|
|
44
|
+
* accepts only native values). `default_enabled`/`default_effort` are typed
|
|
45
|
+
* for fidelity but not consumed.
|
|
49
46
|
*/
|
|
50
47
|
export interface NeuralwattApiModelReasoning {
|
|
51
48
|
/** Whether the model reasons by default. */
|
|
@@ -58,7 +55,10 @@ export interface NeuralwattApiModelReasoning {
|
|
|
58
55
|
accepted_efforts?: NeuralwattReasoningEffort[];
|
|
59
56
|
/** Server-side default. Not consumed; Pi has no default-reasoning field. */
|
|
60
57
|
default_effort: NeuralwattReasoningEffort;
|
|
61
|
-
/**
|
|
58
|
+
/**
|
|
59
|
+
* Wire-level aliases from accepted to supported efforts. Consumed by the
|
|
60
|
+
* anthropic-messages thinking level map; ignored by openai-completions.
|
|
61
|
+
*/
|
|
62
62
|
effort_aliases?: Partial<
|
|
63
63
|
Record<NeuralwattReasoningEffort, NeuralwattReasoningEffort>
|
|
64
64
|
>;
|
|
@@ -1,41 +0,0 @@
|
|
|
1
|
-
import type { Migration } from "@aliou/pi-utils-settings";
|
|
2
|
-
|
|
3
|
-
/** Nested config shape before aliases were split out (pre-0.11.0). */
|
|
4
|
-
interface PreAliasNeuralwattConfig {
|
|
5
|
-
$schema?: string;
|
|
6
|
-
provider?: {
|
|
7
|
-
includeLegacyModelIds?: boolean;
|
|
8
|
-
includeAliasedModelIds?: boolean;
|
|
9
|
-
includeEarlyAccessModels?: boolean;
|
|
10
|
-
};
|
|
11
|
-
quotaCommand?: { enabled?: boolean };
|
|
12
|
-
quotaWarnings?: { enabled?: boolean };
|
|
13
|
-
subBarIntegration?: { enabled?: boolean };
|
|
14
|
-
}
|
|
15
|
-
|
|
16
|
-
/**
|
|
17
|
-
* Creator-scoped active model IDs were split out of the legacy model ID setting.
|
|
18
|
-
* Preserve behavior for users who had explicitly enabled legacy model IDs.
|
|
19
|
-
*/
|
|
20
|
-
export const enableAliasesForLegacyUsersMigration: Migration<PreAliasNeuralwattConfig> =
|
|
21
|
-
{
|
|
22
|
-
name: "enable-alias-model-ids-for-legacy-users",
|
|
23
|
-
version: "0.11.0",
|
|
24
|
-
shouldRun: (config) =>
|
|
25
|
-
config.provider?.includeLegacyModelIds === true &&
|
|
26
|
-
config.provider?.includeAliasedModelIds === undefined,
|
|
27
|
-
message:
|
|
28
|
-
"[neuralwatt] active model aliases now use `provider.includeAliasedModelIds`; it was enabled because legacy model IDs were enabled.",
|
|
29
|
-
run: (config) => {
|
|
30
|
-
const provider = config.provider;
|
|
31
|
-
if (!provider) return config;
|
|
32
|
-
|
|
33
|
-
return {
|
|
34
|
-
...config,
|
|
35
|
-
provider: {
|
|
36
|
-
...provider,
|
|
37
|
-
includeAliasedModelIds: true,
|
|
38
|
-
},
|
|
39
|
-
};
|
|
40
|
-
},
|
|
41
|
-
};
|