@aliou/pi-neuralwatt 0.15.4 → 0.16.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -47,6 +47,15 @@ Once installed, select `neuralwatt` as your provider and choose from available m
47
47
  /model neuralwatt meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8
48
48
  ```
49
49
 
50
+ ### API surface
51
+
52
+ Neuralwatt serves every model on two APIs. Pick one via `/neuralwatt:settings` → **API** (or set `provider.api` in the extension config):
53
+
54
+ - `openai-completions` (default) — the OpenAI-compatible `chat/completions` endpoint;
55
+ - `anthropic-messages` — the Anthropic-compatible `POST /v1/messages` endpoint (vLLM-backed), which streams native tool use and thinking blocks.
56
+
57
+ The setting swaps the whole provider (same model ids on both sides) and applies on `/reload`. Usage/cost accounting and quota tracking work on both surfaces: per-request quota headers only exist on chat-completions responses, while `/v1/messages` streams carry the same data as `: energy` / `: cost` SSE comments.
58
+
50
59
  ### Quota Command
51
60
 
52
61
  Check your API usage at a glance:
@@ -74,12 +83,10 @@ When a Neuralwatt model is active, the footer status bar shows live quota usage
74
83
 
75
84
  Configure features with `/neuralwatt:settings`:
76
85
 
86
+ - **API** — Choose between `openai-completions` (default) and `anthropic-messages`; applies on `/reload`
77
87
  - **Quota command** — Show/hide `/neuralwatt:quota`
78
88
  - **Quota warnings** — Enable/disable low quota notifications
79
89
  - **Sub-bar integration** — Show/hide usage in status bar
80
- - **Legacy model IDs** — Include deprecated model aliases
81
- - **Alias model IDs** — Include active creator-scoped model aliases
82
- - **Early access models** — Include pre-release models available only to the configured API key
83
90
 
84
91
  The provider itself cannot be disabled — it is always loaded.
85
92
 
@@ -87,9 +94,9 @@ Configuration uses nested per-feature sections. Existing flat config files are m
87
94
 
88
95
  ### Model Refresh
89
96
 
90
- Neuralwatt registers its public models without network access. When early-access models are enabled, opening `/model` refreshes the authenticated catalog in the background. `pi update --models` forces an immediate refresh.
97
+ Neuralwatt registers its public models without network access. Opening `/model` refreshes the catalog from the API in the background (authenticated when an API key is configured). `pi update --models` forces an immediate refresh.
91
98
 
92
- Pi stores the complete effective Neuralwatt catalog in `~/.pi/agent/models-store.json` for offline startup. Current hardcoded public and legacy definitions remain authoritative when cached models are restored.
99
+ Pi stores the complete effective Neuralwatt catalog in `~/.pi/agent/models-store.json` for offline startup. Current hardcoded public definitions remain authoritative when cached models are restored.
93
100
 
94
101
  ## Adding or Updating Models
95
102
 
@@ -0,0 +1,114 @@
1
+ import type { StreamOptions } from "@earendil-works/pi-ai";
2
+ import { stream, streamSimple } from "@earendil-works/pi-ai/compat";
3
+ import {
4
+ NEURALWATT_BASE_URL,
5
+ NEURALWATT_PROVIDER_ID,
6
+ NEURALWATT_REQUEST_HEADERS,
7
+ } from "../constants";
8
+ import { buildAnthropicThinkingLevelMap } from "../models/build";
9
+ import type { NeuralwattModel } from "../models/catalog";
10
+ import type { AnyStreamSimple } from "../stream-simple";
11
+ import type { NeuralwattApiHandler } from "./types";
12
+
13
+ type AnthropicMessagesBody = {
14
+ thinking?: { type?: string };
15
+ output_config?: { effort?: string };
16
+ chat_template_kwargs?: Record<string, unknown>;
17
+ [key: string]: unknown;
18
+ };
19
+
20
+ // Outside vLLM's effort enum (HTTP 400); both mean "reasoning off".
21
+ const EFFORT_OFF_VALUES = new Set(["none", "minimal"]);
22
+
23
+ function applyReasoningOff(body: AnthropicMessagesBody): AnthropicMessagesBody {
24
+ delete body.thinking;
25
+ delete body.output_config;
26
+ body.chat_template_kwargs = {
27
+ ...(body.chat_template_kwargs ?? {}),
28
+ enable_thinking: false,
29
+ };
30
+ return body;
31
+ }
32
+
33
+ /**
34
+ * vLLM's reasoning controls differ from first-party Anthropic: positive levels
35
+ * go through `output_config.effort` (adaptive path, forced via compat) while
36
+ * `thinking:{type:"disabled"}` is accepted but ignored, so off is expressed
37
+ * through the chat-template kwarg instead.
38
+ */
39
+ function makeReasoningInjector(
40
+ upstream?: StreamOptions["onPayload"],
41
+ ): NonNullable<StreamOptions["onPayload"]> {
42
+ return async (payload, model) => {
43
+ const next = await upstream?.(payload, model);
44
+ const body = (next !== undefined ? next : payload) as AnthropicMessagesBody;
45
+
46
+ if (body.thinking?.type === "disabled") {
47
+ return applyReasoningOff(body);
48
+ }
49
+
50
+ const effort = body.output_config?.effort;
51
+ if (effort && EFFORT_OFF_VALUES.has(effort)) {
52
+ return applyReasoningOff(body);
53
+ }
54
+
55
+ return body;
56
+ };
57
+ }
58
+
59
+ // The Anthropic SDK appends `/v1/messages` to the client base URL.
60
+ function toMessagesBaseUrl(baseUrl: string): string {
61
+ return baseUrl.replace(/\/v1\/?$/, "");
62
+ }
63
+
64
+ function stampAnthropicModels(models: NeuralwattModel[]) {
65
+ return models.map((model) => {
66
+ const { reasoningContract, ...compiled } = model;
67
+ // No retained contract: the identity map is the alias-free special case.
68
+ const thinkingLevelMap = model.reasoning
69
+ ? reasoningContract
70
+ ? buildAnthropicThinkingLevelMap(reasoningContract)
71
+ : model.thinkingLevelMap
72
+ ? { ...model.thinkingLevelMap }
73
+ : undefined
74
+ : undefined;
75
+
76
+ return {
77
+ ...compiled,
78
+ api: "anthropic-messages" as const,
79
+ provider: NEURALWATT_PROVIDER_ID,
80
+ baseUrl: toMessagesBaseUrl(model.baseUrl ?? NEURALWATT_BASE_URL),
81
+ headers: NEURALWATT_REQUEST_HEADERS,
82
+ compat: {
83
+ forceAdaptiveThinking: true,
84
+ supportsTemperature: true,
85
+ supportsStrictTools: false,
86
+ supportsCacheControlOnTools: false,
87
+ },
88
+ ...(thinkingLevelMap ? { thinkingLevelMap } : {}),
89
+ };
90
+ });
91
+ }
92
+
93
+ export function createAnthropicMessagesApi(options?: {
94
+ streamSimple?: AnyStreamSimple;
95
+ }): NeuralwattApiHandler {
96
+ const withReasoning = (options?: {
97
+ onPayload?: StreamOptions["onPayload"];
98
+ }) => ({
99
+ ...options,
100
+ onPayload: makeReasoningInjector(options?.onPayload),
101
+ });
102
+
103
+ return {
104
+ stampModels: stampAnthropicModels,
105
+ stream: (model, context, streamOptions) =>
106
+ stream(model, context, withReasoning(streamOptions) as never),
107
+ streamSimple: (model, context, simpleOptions) =>
108
+ (options?.streamSimple ?? streamSimple)(
109
+ model,
110
+ context,
111
+ withReasoning(simpleOptions) as never,
112
+ ),
113
+ };
114
+ }
@@ -0,0 +1,30 @@
1
+ import { stream, streamSimple } from "@earendil-works/pi-ai/compat";
2
+ import {
3
+ NEURALWATT_BASE_URL,
4
+ NEURALWATT_PROVIDER_ID,
5
+ NEURALWATT_REQUEST_HEADERS,
6
+ } from "../constants";
7
+ import type { NeuralwattModel } from "../models/catalog";
8
+ import type { AnyStreamSimple } from "../stream-simple";
9
+ import type { NeuralwattApiHandler } from "./types";
10
+
11
+ export function createOpenAiCompletionsApi(options?: {
12
+ streamSimple?: AnyStreamSimple;
13
+ }): NeuralwattApiHandler {
14
+ return {
15
+ stampModels: (models: NeuralwattModel[]) =>
16
+ models.map((model) => {
17
+ const { reasoningContract: _reasoningContract, ...compiled } = model;
18
+ return {
19
+ ...compiled,
20
+ api: "openai-completions" as const,
21
+ provider: NEURALWATT_PROVIDER_ID,
22
+ baseUrl: model.baseUrl ?? NEURALWATT_BASE_URL,
23
+ headers: NEURALWATT_REQUEST_HEADERS,
24
+ };
25
+ }),
26
+ stream: (model, context, streamOptions) =>
27
+ stream(model, context, streamOptions as never),
28
+ streamSimple: options?.streamSimple ?? streamSimple,
29
+ };
30
+ }
@@ -0,0 +1,24 @@
1
+ import type {
2
+ Api,
3
+ AssistantMessageEventStream,
4
+ Context,
5
+ Model,
6
+ SimpleStreamOptions,
7
+ StreamOptions,
8
+ } from "@earendil-works/pi-ai";
9
+ import type { NeuralwattModel } from "../models/catalog";
10
+
11
+ /** One Neuralwatt API surface: model stamping plus submission plumbing. */
12
+ export interface NeuralwattApiHandler {
13
+ stampModels(models: NeuralwattModel[]): Model<Api>[];
14
+ stream(
15
+ model: Model<Api>,
16
+ context: Context,
17
+ options?: StreamOptions,
18
+ ): AssistantMessageEventStream;
19
+ streamSimple(
20
+ model: Model<Api>,
21
+ context: Context,
22
+ options?: SimpleStreamOptions,
23
+ ): AssistantMessageEventStream;
24
+ }
@@ -6,6 +6,7 @@ import type { ExtensionAPI } from "@earendil-works/pi-coding-agent";
6
6
  import type { SettingItem } from "@earendil-works/pi-tui";
7
7
  import {
8
8
  configLoader,
9
+ type NeuralwattApi,
9
10
  type NeuralwattConfig,
10
11
  type ResolvedNeuralwattConfig,
11
12
  } from "../../../../src/config";
@@ -61,6 +62,9 @@ export function registerNeuralwattSettings(
61
62
  options: RegisterNeuralwattSettingsOptions,
62
63
  ): void {
63
64
  const { getLoadedFeatures } = options;
65
+ // The provider stamps `provider.api` at extension load; a saved change only
66
+ // reaches it after `/reload`.
67
+ let pendingApi: NeuralwattApi | undefined;
64
68
 
65
69
  registerSettingsCommand<NeuralwattConfig, ResolvedNeuralwattConfig>(pi, {
66
70
  commandName: "neuralwatt:settings",
@@ -69,6 +73,19 @@ export function registerNeuralwattSettings(
69
73
  buildSections: (tabConfig, resolved): SettingsSection[] => {
70
74
  const loaded = getLoadedFeatures();
71
75
  return [
76
+ {
77
+ label: "Provider",
78
+ items: [
79
+ {
80
+ id: "api",
81
+ label: "API",
82
+ description:
83
+ "Serve models via the OpenAI-compatible chat/completions endpoint or the Anthropic-compatible /v1/messages endpoint",
84
+ currentValue: tabConfig?.provider?.api ?? resolved.provider.api,
85
+ values: ["openai-completions", "anthropic-messages"],
86
+ },
87
+ ],
88
+ },
72
89
  {
73
90
  label: "Features",
74
91
  items: [
@@ -107,6 +124,20 @@ export function registerNeuralwattSettings(
107
124
  ];
108
125
  },
109
126
  onSettingChange: (id, newValue, config) => {
127
+ if (id === "api") {
128
+ if (
129
+ newValue !== "openai-completions" &&
130
+ newValue !== "anthropic-messages"
131
+ ) {
132
+ return null;
133
+ }
134
+ pendingApi = newValue;
135
+ return {
136
+ ...config,
137
+ provider: { ...config.provider, api: newValue },
138
+ };
139
+ }
140
+
110
141
  if (!getLoadedFeatures().has(id as NeuralwattFeatureId)) {
111
142
  return null;
112
143
  }
@@ -132,8 +163,11 @@ export function registerNeuralwattSettings(
132
163
  return null;
133
164
  }
134
165
  },
135
- onSave: async () => {
166
+ onSave: async (ctx) => {
136
167
  emitConfigUpdated(pi);
168
+ if (pendingApi === undefined) return;
169
+ pendingApi = undefined;
170
+ ctx.ui.notify("Run /reload to apply the new API", "info");
137
171
  },
138
172
  });
139
173
  }
@@ -0,0 +1,7 @@
1
+ export const NEURALWATT_PROVIDER_ID = "neuralwatt";
2
+ export const NEURALWATT_BASE_URL = "https://api.neuralwatt.com/v1";
3
+ export const NEURALWATT_API_KEY_ENV = "NEURALWATT_API_KEY";
4
+ export const NEURALWATT_REQUEST_HEADERS = {
5
+ Referer: "https://pi.dev",
6
+ "X-Title": "npm:@aliou/pi-neuralwatt",
7
+ };
@@ -55,6 +55,15 @@ function registerNeuralwattProvider(
55
55
  ) as never)
56
56
  : undefined;
57
57
 
58
+ const messagesApiProvider = getApiProvider("anthropic-messages");
59
+ const messagesBaseStreamSimple = messagesApiProvider?.streamSimple;
60
+ const messagesStreamSimple = messagesBaseStreamSimple
61
+ ? (wrapNeuralwattStreamSimple(
62
+ messagesBaseStreamSimple as never,
63
+ onSseQuota,
64
+ ) as never)
65
+ : undefined;
66
+
58
67
  pi.registerProvider(
59
68
  createNeuralwattProvider(
60
69
  staticModels,
@@ -65,7 +74,11 @@ function registerNeuralwattProvider(
65
74
  }
66
75
  return result.data;
67
76
  },
68
- streamSimple,
77
+ {
78
+ api: configLoader.getConfig().provider.api,
79
+ openAiStreamSimple: streamSimple,
80
+ messagesStreamSimple,
81
+ },
69
82
  ),
70
83
  );
71
84
  }
@@ -8,6 +8,15 @@ export type ThinkingLevelMap = NonNullable<
8
8
  ProviderModelConfig["thinkingLevelMap"]
9
9
  >;
10
10
 
11
+ /**
12
+ * A compiled provider model plus the reasoning contract it was compiled from,
13
+ * retained for anthropic-messages map derivation. Rides the models store
14
+ * (JSON passthrough); stripped from stamped runtime models.
15
+ */
16
+ export type NeuralwattCompiledModel = ProviderModelConfig & {
17
+ reasoningContract?: NeuralwattReasoningMapSource;
18
+ };
19
+
11
20
  export interface NeuralwattCost {
12
21
  input: number;
13
22
  output: number;
@@ -56,7 +65,7 @@ export interface NeuralwattVariantSpec {
56
65
  */
57
66
  export type NeuralwattReasoningMapSource = Pick<
58
67
  NeuralwattApiModelReasoning,
59
- "supported_efforts" | "mandatory"
68
+ "supported_efforts" | "mandatory" | "effort_aliases"
60
69
  >;
61
70
 
62
71
  /**
@@ -71,9 +80,9 @@ export type NeuralwattReasoningMapSource = Pick<
71
80
  * exposes none), falls back to a conservative `high`-only map with `off: null`,
72
81
  * matching the upstream binary thinking toggle.
73
82
  *
74
- * `default_effort` and `effort_aliases` are deliberately ignored: Pi has no
75
- * default-reasoning field, and we expose native supported efforts rather than
76
- * aliasing unsupported ones.
83
+ * `effort_aliases` is deliberately ignored here (the openai-completions
84
+ * gateway aliases unsupported efforts server-side); it is consumed by the
85
+ * anthropic-messages map below.
77
86
  */
78
87
  export function buildThinkingLevelMap(
79
88
  reasoning: NeuralwattReasoningMapSource | undefined,
@@ -96,6 +105,39 @@ export function buildThinkingLevelMap(
96
105
  };
97
106
  }
98
107
 
108
+ /**
109
+ * Thinking level map for the anthropic-messages surface. vLLM's
110
+ * `output_config.effort` accepts only the model's native efforts, so
111
+ * unsupported Pi levels resolve through `effort_aliases` (or `null`). A level
112
+ * may resolve to `"none"` — off on this surface, handled by the payload
113
+ * injector in `api/anthropic-messages.ts`.
114
+ */
115
+ export function buildAnthropicThinkingLevelMap(
116
+ reasoning: NeuralwattReasoningMapSource | undefined,
117
+ ): ThinkingLevelMap {
118
+ const supported = new Set<string>(reasoning?.supported_efforts ?? ["high"]);
119
+ const mandatory = reasoning?.mandatory ?? true;
120
+ const aliases = reasoning?.effort_aliases ?? {};
121
+
122
+ const resolve = (level: string): string | null => {
123
+ if (supported.has(level)) return level;
124
+ const alias = aliases[level as keyof typeof aliases];
125
+ return alias && supported.has(alias) ? alias : null;
126
+ };
127
+
128
+ return {
129
+ // "none" is a marker so pi-ai enables the off path (off !== null);
130
+ // vLLM rejects it on the wire, so the injector never sends it verbatim.
131
+ off: !mandatory && supported.has("none") ? "none" : null,
132
+ minimal: resolve("minimal"),
133
+ low: resolve("low"),
134
+ medium: resolve("medium"),
135
+ high: resolve("high"),
136
+ xhigh: resolve("xhigh"),
137
+ max: resolve("max"),
138
+ };
139
+ }
140
+
99
141
  /**
100
142
  * Neuralwatt reports `max_output_tokens: null` for models whose output is only
101
143
  * bounded by the context window. Some models incorrectly report 0; treat 0
@@ -112,7 +154,7 @@ export function resolveMaxTokens(
112
154
  export function buildNeuralwattModel(
113
155
  family: NeuralwattModelFamily,
114
156
  variant: NeuralwattVariantSpec,
115
- ): ProviderModelConfig {
157
+ ): NeuralwattCompiledModel {
116
158
  const vision = variant.vision ?? family.vision;
117
159
 
118
160
  const compat: NonNullable<ProviderModelConfig["compat"]> = {
@@ -127,7 +169,7 @@ export function buildNeuralwattModel(
127
169
  const scale = (value: number): number =>
128
170
  multiplier === 1 ? value : Number((value * multiplier).toFixed(6));
129
171
 
130
- const model: ProviderModelConfig = {
172
+ const model: NeuralwattCompiledModel = {
131
173
  id: variant.id,
132
174
  name: variant.name,
133
175
  reasoning: variant.reasoning,
@@ -144,14 +186,14 @@ export function buildNeuralwattModel(
144
186
  };
145
187
 
146
188
  if (variant.reasoning) {
189
+ const contract = variant.reasoningMetadata ?? family.reasoningMetadata;
147
190
  // Clone so variants never share a family map instance. The map is derived
148
191
  // from the API reasoning contract; missing metadata falls back to a
149
192
  // high-only map rather than throwing.
150
193
  model.thinkingLevelMap = {
151
- ...buildThinkingLevelMap(
152
- variant.reasoningMetadata ?? family.reasoningMetadata,
153
- ),
194
+ ...buildThinkingLevelMap(contract),
154
195
  };
196
+ model.reasoningContract = contract;
155
197
  }
156
198
 
157
199
  return model;
@@ -160,6 +202,6 @@ export function buildNeuralwattModel(
160
202
  export function buildNeuralwattFamily(
161
203
  family: NeuralwattModelFamily,
162
204
  variants: NeuralwattVariantSpec[],
163
- ): ProviderModelConfig[] {
205
+ ): NeuralwattCompiledModel[] {
164
206
  return variants.map((variant) => buildNeuralwattModel(family, variant));
165
207
  }
@@ -2,12 +2,13 @@ import type { ProviderModelConfig } from "@earendil-works/pi-coding-agent";
2
2
  import type { NeuralwattApiModel } from "../../../src/types/models-api";
3
3
  import {
4
4
  buildThinkingLevelMap,
5
+ type NeuralwattCompiledModel,
5
6
  resolveMaxTokens,
6
7
  type ThinkingLevelMap,
7
8
  } from "./build";
8
9
  import { NEURALWATT_MODELS } from "./public-models";
9
10
 
10
- export type NeuralwattModel = ProviderModelConfig;
11
+ export type NeuralwattModel = NeuralwattCompiledModel;
11
12
 
12
13
  // Chat-template thinking: the API exposes a `reasoning` block, but the
13
14
  // underlying mechanism is chat_template_kwargs, so Pi needs the mapping.
@@ -22,16 +23,6 @@ const COMPAT_OVERRIDES: Partial<
22
23
  },
23
24
  };
24
25
 
25
- const HARDCODED_ALIASES: Record<string, string> = {
26
- "moonshotai/Kimi-K2.7-Code": "kimi-k2.7-code",
27
- "Qwen/Qwen3.6-35B-A3B": "qwen3.6-35b",
28
- "deepseek-ai/DeepSeek-V4-Flash": "deepseek-v4-flash",
29
- };
30
-
31
- function isVariantId(id: string): boolean {
32
- return id.includes("-fast") || id.includes("-flex") || id.includes("-short");
33
- }
34
-
35
26
  function apiModelToProviderModel(model: NeuralwattApiModel): NeuralwattModel {
36
27
  const meta = model.metadata;
37
28
  if (!meta)
@@ -72,44 +63,14 @@ function apiModelToProviderModel(model: NeuralwattApiModel): NeuralwattModel {
72
63
  result.thinkingLevelMap = buildThinkingLevelMap(
73
64
  meta.reasoning,
74
65
  ) as ThinkingLevelMap;
66
+ // Kept for anthropic-messages stamping, which resolves levels through
67
+ // `effort_aliases`.
68
+ result.reasoningContract = meta.reasoning;
75
69
  }
76
70
 
77
71
  return result;
78
72
  }
79
73
 
80
- function buildAliases(
81
- models: NeuralwattModel[],
82
- apiModels: readonly NeuralwattApiModel[],
83
- ): NeuralwattModel[] {
84
- const existingIds = new Set(models.map((m) => m.id));
85
- const aliases: NeuralwattModel[] = [];
86
- const seen = new Set<string>();
87
-
88
- const addAlias = (aliasId: string, canonicalId: string): void => {
89
- if (seen.has(aliasId) || existingIds.has(aliasId)) return;
90
- const canonical = models.find((m) => m.id === canonicalId);
91
- if (!canonical) return;
92
- seen.add(aliasId);
93
- aliases.push({
94
- ...canonical,
95
- id: aliasId,
96
- name: `${canonical.name} (alias ID)`,
97
- });
98
- };
99
-
100
- for (const [aliasId, canonicalId] of Object.entries(HARDCODED_ALIASES)) {
101
- addAlias(aliasId, canonicalId);
102
- }
103
-
104
- for (const apiModel of apiModels) {
105
- const hfId = apiModel.metadata?.huggingface_id;
106
- if (!hfId || hfId === apiModel.id || isVariantId(apiModel.id)) continue;
107
- addAlias(hfId, apiModel.id);
108
- }
109
-
110
- return aliases;
111
- }
112
-
113
74
  export function buildNeuralwattProviderModels(): NeuralwattModel[] {
114
75
  return NEURALWATT_MODELS.map((model) => ({ ...model }));
115
76
  }
@@ -130,7 +91,7 @@ export function buildNeuralwattProviderModelsFromApi(
130
91
  ),
131
92
  )
132
93
  .map(apiModelToProviderModel);
133
- return [...models, ...buildAliases(models, apiModels)];
94
+ return models;
134
95
  }
135
96
 
136
97
  export function buildNeuralwattProviderModelsFromStore(
@@ -127,6 +127,13 @@ const FAMILIES: [NeuralwattModelFamily, NeuralwattVariantSpec[]][] = [
127
127
  reasoning: true,
128
128
  costMultiplier: 0.65,
129
129
  },
130
+ {
131
+ id: "deepseek-v4-flash-speed",
132
+ name: "DeepSeek V4 Flash (Speed)",
133
+ contextWindow: 1048560,
134
+ maxOutputTokens: 65536,
135
+ reasoning: true,
136
+ },
130
137
  ],
131
138
  ],
132
139
  [
@@ -1,10 +1,14 @@
1
- import type {
2
- Api,
3
- Model,
4
- Provider,
5
- ProviderStreamOptions,
6
- } from "@earendil-works/pi-ai";
7
- import { stream, streamSimple } from "@earendil-works/pi-ai/compat";
1
+ import type { Provider } from "@earendil-works/pi-ai";
2
+ import type { NeuralwattApi } from "../../src/config";
3
+ import { createAnthropicMessagesApi } from "./api/anthropic-messages";
4
+ import { createOpenAiCompletionsApi } from "./api/openai-completions";
5
+ import type { NeuralwattApiHandler } from "./api/types";
6
+ import {
7
+ NEURALWATT_API_KEY_ENV,
8
+ NEURALWATT_BASE_URL,
9
+ NEURALWATT_PROVIDER_ID,
10
+ NEURALWATT_REQUEST_HEADERS,
11
+ } from "./constants";
8
12
  import type { NeuralwattModel } from "./models/catalog";
9
13
  import {
10
14
  buildNeuralwattProviderModelsFromApi,
@@ -16,33 +20,39 @@ import {
16
20
  } from "./models/refresh";
17
21
  import type { AnyStreamSimple } from "./stream-simple";
18
22
 
19
- export const NEURALWATT_PROVIDER_ID = "neuralwatt";
20
- export const NEURALWATT_BASE_URL = "https://api.neuralwatt.com/v1";
21
- export const NEURALWATT_API_KEY_ENV = "NEURALWATT_API_KEY";
22
-
23
- const NEURALWATT_REQUEST_HEADERS = {
24
- Referer: "https://pi.dev",
25
- "X-Title": "npm:@aliou/pi-neuralwatt",
26
- };
23
+ export { NEURALWATT_API_KEY_ENV, NEURALWATT_BASE_URL, NEURALWATT_PROVIDER_ID };
27
24
 
28
- const API = "openai-completions" as const;
25
+ export interface NeuralwattProviderOptions {
26
+ /** Active API surface; resolved once. Changes need a `/reload`. */
27
+ api?: NeuralwattApi;
28
+ openAiStreamSimple?: AnyStreamSimple;
29
+ messagesStreamSimple?: AnyStreamSimple;
30
+ }
29
31
 
30
- function toProviderModels(models: NeuralwattModel[]): Model<Api>[] {
31
- return models.map((model) => ({
32
- ...model,
33
- api: model.api ?? API,
34
- provider: NEURALWATT_PROVIDER_ID,
35
- baseUrl: model.baseUrl ?? NEURALWATT_BASE_URL,
36
- headers: NEURALWATT_REQUEST_HEADERS,
37
- }));
32
+ function createApiHandler(
33
+ api: NeuralwattApi,
34
+ options?: NeuralwattProviderOptions,
35
+ ): NeuralwattApiHandler {
36
+ if (api === "anthropic-messages") {
37
+ return createAnthropicMessagesApi({
38
+ streamSimple: options?.messagesStreamSimple,
39
+ });
40
+ }
41
+ return createOpenAiCompletionsApi({
42
+ streamSimple: options?.openAiStreamSimple,
43
+ });
38
44
  }
39
45
 
40
46
  export function createNeuralwattProvider(
41
47
  staticModels: NeuralwattModel[],
42
48
  fetchApiModels: FetchNeuralwattApiModels,
43
- streamSimpleOverride?: AnyStreamSimple,
49
+ options?: NeuralwattProviderOptions,
44
50
  ): Provider {
45
- let liveModels = toProviderModels(staticModels);
51
+ const handler = createApiHandler(
52
+ options?.api ?? "openai-completions",
53
+ options,
54
+ );
55
+ let canonicalModels = staticModels;
46
56
  const refreshCatalog = createNeuralwattRefreshModels(
47
57
  staticModels,
48
58
  fetchApiModels,
@@ -92,17 +102,18 @@ export function createNeuralwattProvider(
92
102
  },
93
103
  },
94
104
  },
95
- getModels: () => liveModels,
105
+ getModels: () => handler.stampModels(canonicalModels),
96
106
  refreshModels: async (context) => {
97
107
  const models = await refreshCatalog(context);
98
108
  await context.publish({
99
109
  update: () => {
100
- liveModels = toProviderModels(models);
110
+ canonicalModels = models;
101
111
  },
102
112
  });
103
113
  },
104
- stream: (model, context, options) =>
105
- stream(model, context, options as ProviderStreamOptions | undefined),
106
- streamSimple: streamSimpleOverride ?? streamSimple,
114
+ stream: (model, context, streamOptions) =>
115
+ handler.stream(model, context, streamOptions as never),
116
+ streamSimple: (model, context, simpleOptions) =>
117
+ handler.streamSimple(model, context, simpleOptions),
107
118
  };
108
119
  }
@@ -31,13 +31,26 @@ export function updateQuotasFromSseComment(
31
31
  if (trimmed.startsWith(": cost ")) {
32
32
  const cost = JSON.parse(trimmed.slice(7)) as {
33
33
  request_cost_usd?: number;
34
+ allowance_remaining_usd?: number;
34
35
  };
35
36
  const requestCostUsd = cost.request_cost_usd ?? 0;
36
37
  if (requestCostUsd <= 0) return quotas;
37
- next.balance.credits_remaining_usd = Math.max(
38
- 0,
39
- next.balance.credits_remaining_usd - requestCostUsd,
40
- );
38
+ // The absolute allowance in the comment is fresher than the locally
39
+ // tracked total.
40
+ if (
41
+ typeof cost.allowance_remaining_usd === "number" &&
42
+ Number.isFinite(cost.allowance_remaining_usd)
43
+ ) {
44
+ next.balance.credits_remaining_usd = Math.max(
45
+ 0,
46
+ cost.allowance_remaining_usd,
47
+ );
48
+ } else {
49
+ next.balance.credits_remaining_usd = Math.max(
50
+ 0,
51
+ next.balance.credits_remaining_usd - requestCostUsd,
52
+ );
53
+ }
41
54
  next.balance.credits_used_usd += requestCostUsd;
42
55
  next.usage.current_month.cost_usd += requestCostUsd;
43
56
  next.usage.lifetime.cost_usd += requestCostUsd;
@@ -35,7 +35,7 @@ function headersToRecord(headers: Headers): Record<string, string> {
35
35
  return record;
36
36
  }
37
37
 
38
- function isProviderChatCompletionsUrl(
38
+ function isProviderStreamUrl(
39
39
  input: RequestInfo | URL,
40
40
  providerOrigin: string,
41
41
  ): boolean {
@@ -50,7 +50,8 @@ function isProviderChatCompletionsUrl(
50
50
  const url = new URL(rawUrl);
51
51
  return (
52
52
  url.origin === providerOrigin &&
53
- url.pathname.endsWith("/chat/completions")
53
+ (url.pathname.endsWith("/chat/completions") ||
54
+ url.pathname.endsWith("/messages"))
54
55
  );
55
56
  } catch {
56
57
  return false;
@@ -96,7 +97,7 @@ export function wrapNeuralwattStreamSimple(
96
97
  const wrappedFetch: typeof fetch = async (input, init) => {
97
98
  const response = await originalFetch(input, init);
98
99
 
99
- if (!isProviderChatCompletionsUrl(input, providerOrigin)) return response;
100
+ if (!isProviderStreamUrl(input, providerOrigin)) return response;
100
101
 
101
102
  const headers = headersToRecord(response.headers);
102
103
  if (response.status === 429) {
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@aliou/pi-neuralwatt",
3
- "version": "0.15.4",
3
+ "version": "0.16.0",
4
4
  "license": "MIT",
5
5
  "type": "module",
6
6
  "private": false,
@@ -69,6 +69,7 @@
69
69
  "test": "vitest run",
70
70
  "test:watch": "vitest",
71
71
  "check:models": "tsx scripts/check-models.ts",
72
+ "check:changesets": "tsx scripts/check-changesets.ts",
72
73
  "gen:schema": "pi-settings-schema -p src/config/types.ts -t NeuralwattConfig -o schema.json --version 0.12.0",
73
74
  "check:schema": "pi-settings-schema -p src/config/types.ts -t NeuralwattConfig -o schema.json --version 0.12.0 --check",
74
75
  "check:lockfile": "pnpm install --frozen-lockfile --ignore-scripts",
package/schema.json CHANGED
@@ -20,6 +20,10 @@
20
20
  "$ref": "#/definitions/NeuralwattSubBarIntegrationConfig",
21
21
  "description": "Sub-bar/status-bar integration feature."
22
22
  },
23
+ "provider": {
24
+ "$ref": "#/definitions/NeuralwattProviderConfig",
25
+ "description": "Provider behavior (API surface)."
26
+ },
23
27
  "version": {
24
28
  "anyOf": [
25
29
  {
@@ -65,6 +69,24 @@
65
69
  }
66
70
  },
67
71
  "additionalProperties": false
72
+ },
73
+ "NeuralwattProviderConfig": {
74
+ "type": "object",
75
+ "properties": {
76
+ "api": {
77
+ "$ref": "#/definitions/NeuralwattApi",
78
+ "description": "Which API serves model requests."
79
+ }
80
+ },
81
+ "additionalProperties": false
82
+ },
83
+ "NeuralwattApi": {
84
+ "type": "string",
85
+ "enum": [
86
+ "openai-completions",
87
+ "anthropic-messages"
88
+ ],
89
+ "description": "Neuralwatt serves every chat model twice: on an OpenAI-compatible `chat/completions` endpoint and on a vLLM-backed Anthropic-compatible `POST /v1/messages` endpoint. Exactly one serves the provider at a time."
68
90
  }
69
91
  }
70
92
  }
@@ -10,4 +10,7 @@ export const DEFAULT_CONFIG: ResolvedNeuralwattConfig = {
10
10
  subBarIntegration: {
11
11
  enabled: true,
12
12
  },
13
+ provider: {
14
+ api: "openai-completions",
15
+ },
13
16
  };
@@ -1,4 +1,8 @@
1
1
  export { DEFAULT_CONFIG } from "./defaults";
2
- export { configLoader } from "./loader";
2
+ export { configLoader, resolveApi } from "./loader";
3
3
  export { migrations } from "./migration";
4
- export type { NeuralwattConfig, ResolvedNeuralwattConfig } from "./types";
4
+ export type {
5
+ NeuralwattApi,
6
+ NeuralwattConfig,
7
+ ResolvedNeuralwattConfig,
8
+ } from "./types";
@@ -2,7 +2,11 @@ import { buildSchemaUrl, ConfigLoader } from "@aliou/pi-utils-settings";
2
2
  import packageJson from "../../package.json";
3
3
  import { DEFAULT_CONFIG } from "./defaults";
4
4
  import { migrations } from "./migration";
5
- import type { NeuralwattConfig, ResolvedNeuralwattConfig } from "./types";
5
+ import type {
6
+ NeuralwattApi,
7
+ NeuralwattConfig,
8
+ ResolvedNeuralwattConfig,
9
+ } from "./types";
6
10
 
7
11
  /**
8
12
  * Fill in every field the rest of the code reads. Migrations already normalized
@@ -27,9 +31,19 @@ function normalizeResolvedConfig(
27
31
  config.subBarIntegration?.enabled ??
28
32
  DEFAULT_CONFIG.subBarIntegration.enabled,
29
33
  },
34
+ provider: {
35
+ api: resolveApi(config.provider?.api),
36
+ },
30
37
  };
31
38
  }
32
39
 
40
+ export function resolveApi(value: string | undefined): NeuralwattApi {
41
+ if (value === "anthropic-messages" || value === "openai-completions") {
42
+ return value;
43
+ }
44
+ return DEFAULT_CONFIG.provider.api;
45
+ }
46
+
33
47
  export const configLoader = new ConfigLoader<
34
48
  NeuralwattConfig,
35
49
  ResolvedNeuralwattConfig
@@ -6,11 +6,9 @@ export {
6
6
  flatToNestedConfigMigration,
7
7
  } from "./02-flat-to-nested-config";
8
8
  export { renameHiddenToEarlyAccessMigration } from "./03-rename-hidden-to-early-access";
9
- export { enableAliasesForLegacyUsersMigration } from "./04-enable-aliases-for-legacy-users";
10
9
 
11
10
  import { flatToNestedConfigMigration } from "./02-flat-to-nested-config";
12
11
  import { renameHiddenToEarlyAccessMigration } from "./03-rename-hidden-to-early-access";
13
- import { enableAliasesForLegacyUsersMigration } from "./04-enable-aliases-for-legacy-users";
14
12
 
15
13
  // Each migration is typed against its own historical input shape. The loader
16
14
  // applies them in sequence on the raw config record, so they are cast to the
@@ -18,5 +16,4 @@ import { enableAliasesForLegacyUsersMigration } from "./04-enable-aliases-for-le
18
16
  export const migrations = [
19
17
  flatToNestedConfigMigration,
20
18
  renameHiddenToEarlyAccessMigration,
21
- enableAliasesForLegacyUsersMigration,
22
19
  ] as unknown as Migration<NeuralwattConfig>[];
@@ -13,6 +13,18 @@ export interface NeuralwattSubBarIntegrationConfig {
13
13
  enabled?: boolean;
14
14
  }
15
15
 
16
+ /**
17
+ * Neuralwatt serves every chat model twice: on an OpenAI-compatible
18
+ * `chat/completions` endpoint and on a vLLM-backed Anthropic-compatible
19
+ * `POST /v1/messages` endpoint. Exactly one serves the provider at a time.
20
+ */
21
+ export type NeuralwattApi = "openai-completions" | "anthropic-messages";
22
+
23
+ export interface NeuralwattProviderConfig {
24
+ /** Which API serves model requests. */
25
+ api?: NeuralwattApi;
26
+ }
27
+
16
28
  export interface NeuralwattConfig {
17
29
  /** $schema URL for editor autocomplete. */
18
30
  $schema?: string;
@@ -25,6 +37,9 @@ export interface NeuralwattConfig {
25
37
 
26
38
  /** Sub-bar/status-bar integration feature. */
27
39
  subBarIntegration?: NeuralwattSubBarIntegrationConfig;
40
+
41
+ /** Provider behavior (API surface). */
42
+ provider?: NeuralwattProviderConfig;
28
43
  }
29
44
 
30
45
  export interface ResolvedNeuralwattConfig {
@@ -37,4 +52,7 @@ export interface ResolvedNeuralwattConfig {
37
52
  subBarIntegration: {
38
53
  enabled: boolean;
39
54
  };
55
+ provider: {
56
+ api: NeuralwattApi;
57
+ };
40
58
  }
@@ -38,14 +38,11 @@ export type NeuralwattReasoningEffort =
38
38
  | "max";
39
39
 
40
40
  /**
41
- * Per-model reasoning contract from `/v1/models`.
42
- *
43
- * `supported_efforts` is authoritative for which Pi thinking levels to expose:
44
- * the Pi map is built by identity (a level is enabled iff it appears here),
45
- * see `buildThinkingLevelMap` in `extensions/provider/models/build.ts`.
46
- * `default_effort` and `effort_aliases` are typed for fidelity but are not
47
- * consumed — Pi has no default-reasoning field and we expose native efforts
48
- * rather than aliasing unsupported ones.
41
+ * Per-model reasoning contract from `/v1/models`. The openai-completions
42
+ * thinking map uses `supported_efforts` by identity; the anthropic-messages
43
+ * map additionally resolves through `effort_aliases` (vLLM's effort enum
44
+ * accepts only native values). `default_enabled`/`default_effort` are typed
45
+ * for fidelity but not consumed.
49
46
  */
50
47
  export interface NeuralwattApiModelReasoning {
51
48
  /** Whether the model reasons by default. */
@@ -58,7 +55,10 @@ export interface NeuralwattApiModelReasoning {
58
55
  accepted_efforts?: NeuralwattReasoningEffort[];
59
56
  /** Server-side default. Not consumed; Pi has no default-reasoning field. */
60
57
  default_effort: NeuralwattReasoningEffort;
61
- /** Wire-level aliases from accepted to supported efforts. Not consumed. */
58
+ /**
59
+ * Wire-level aliases from accepted to supported efforts. Consumed by the
60
+ * anthropic-messages thinking level map; ignored by openai-completions.
61
+ */
62
62
  effort_aliases?: Partial<
63
63
  Record<NeuralwattReasoningEffort, NeuralwattReasoningEffort>
64
64
  >;
@@ -1,41 +0,0 @@
1
- import type { Migration } from "@aliou/pi-utils-settings";
2
-
3
- /** Nested config shape before aliases were split out (pre-0.11.0). */
4
- interface PreAliasNeuralwattConfig {
5
- $schema?: string;
6
- provider?: {
7
- includeLegacyModelIds?: boolean;
8
- includeAliasedModelIds?: boolean;
9
- includeEarlyAccessModels?: boolean;
10
- };
11
- quotaCommand?: { enabled?: boolean };
12
- quotaWarnings?: { enabled?: boolean };
13
- subBarIntegration?: { enabled?: boolean };
14
- }
15
-
16
- /**
17
- * Creator-scoped active model IDs were split out of the legacy model ID setting.
18
- * Preserve behavior for users who had explicitly enabled legacy model IDs.
19
- */
20
- export const enableAliasesForLegacyUsersMigration: Migration<PreAliasNeuralwattConfig> =
21
- {
22
- name: "enable-alias-model-ids-for-legacy-users",
23
- version: "0.11.0",
24
- shouldRun: (config) =>
25
- config.provider?.includeLegacyModelIds === true &&
26
- config.provider?.includeAliasedModelIds === undefined,
27
- message:
28
- "[neuralwatt] active model aliases now use `provider.includeAliasedModelIds`; it was enabled because legacy model IDs were enabled.",
29
- run: (config) => {
30
- const provider = config.provider;
31
- if (!provider) return config;
32
-
33
- return {
34
- ...config,
35
- provider: {
36
- ...provider,
37
- includeAliasedModelIds: true,
38
- },
39
- };
40
- },
41
- };