@aliou/pi-neuralwatt 0.15.3 → 0.16.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -47,6 +47,15 @@ Once installed, select `neuralwatt` as your provider and choose from available m
47
47
  /model neuralwatt meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8
48
48
  ```
49
49
 
50
+ ### API surface
51
+
52
+ Neuralwatt serves every model on two APIs. Pick one via `/neuralwatt:settings` → **API** (or set `provider.api` in the extension config):
53
+
54
+ - `openai-completions` (default) — the OpenAI-compatible `chat/completions` endpoint;
55
+ - `anthropic-messages` — the Anthropic-compatible `POST /v1/messages` endpoint (vLLM-backed), which streams native tool use and thinking blocks.
56
+
57
+ The setting swaps the whole provider (same model ids on both sides) and applies on `/reload`. Usage/cost accounting and quota tracking work on both surfaces: per-request quota headers only exist on chat-completions responses, while `/v1/messages` streams carry the same data as `: energy` / `: cost` SSE comments.
58
+
50
59
  ### Quota Command
51
60
 
52
61
  Check your API usage at a glance:
@@ -74,12 +83,10 @@ When a Neuralwatt model is active, the footer status bar shows live quota usage
74
83
 
75
84
  Configure features with `/neuralwatt:settings`:
76
85
 
86
+ - **API** — Choose between `openai-completions` (default) and `anthropic-messages`; applies on `/reload`
77
87
  - **Quota command** — Show/hide `/neuralwatt:quota`
78
88
  - **Quota warnings** — Enable/disable low quota notifications
79
89
  - **Sub-bar integration** — Show/hide usage in status bar
80
- - **Legacy model IDs** — Include deprecated model aliases
81
- - **Alias model IDs** — Include active creator-scoped model aliases
82
- - **Early access models** — Include pre-release models available only to the configured API key
83
90
 
84
91
  The provider itself cannot be disabled — it is always loaded.
85
92
 
@@ -87,9 +94,9 @@ Configuration uses nested per-feature sections. Existing flat config files are m
87
94
 
88
95
  ### Model Refresh
89
96
 
90
- Neuralwatt registers its public models without network access. When early-access models are enabled, opening `/model` refreshes the authenticated catalog in the background. `pi update --models` forces an immediate refresh.
97
+ Neuralwatt registers its public models without network access. Opening `/model` refreshes the catalog from the API in the background (authenticated when an API key is configured). `pi update --models` forces an immediate refresh.
91
98
 
92
- Pi stores the complete effective Neuralwatt catalog in `~/.pi/agent/models-store.json` for offline startup. Current hardcoded public and legacy definitions remain authoritative when cached models are restored.
99
+ Pi stores the complete effective Neuralwatt catalog in `~/.pi/agent/models-store.json` for offline startup. Current hardcoded public definitions remain authoritative when cached models are restored.
93
100
 
94
101
  ## Adding or Updating Models
95
102
 
@@ -0,0 +1,114 @@
1
+ import type { StreamOptions } from "@earendil-works/pi-ai";
2
+ import { stream, streamSimple } from "@earendil-works/pi-ai/compat";
3
+ import {
4
+ NEURALWATT_BASE_URL,
5
+ NEURALWATT_PROVIDER_ID,
6
+ NEURALWATT_REQUEST_HEADERS,
7
+ } from "../constants";
8
+ import { buildAnthropicThinkingLevelMap } from "../models/build";
9
+ import type { NeuralwattModel } from "../models/catalog";
10
+ import type { AnyStreamSimple } from "../stream-simple";
11
+ import type { NeuralwattApiHandler } from "./types";
12
+
13
+ type AnthropicMessagesBody = {
14
+ thinking?: { type?: string };
15
+ output_config?: { effort?: string };
16
+ chat_template_kwargs?: Record<string, unknown>;
17
+ [key: string]: unknown;
18
+ };
19
+
20
+ // Outside vLLM's effort enum (HTTP 400); both mean "reasoning off".
21
+ const EFFORT_OFF_VALUES = new Set(["none", "minimal"]);
22
+
23
+ function applyReasoningOff(body: AnthropicMessagesBody): AnthropicMessagesBody {
24
+ delete body.thinking;
25
+ delete body.output_config;
26
+ body.chat_template_kwargs = {
27
+ ...(body.chat_template_kwargs ?? {}),
28
+ enable_thinking: false,
29
+ };
30
+ return body;
31
+ }
32
+
33
+ /**
34
+ * vLLM's reasoning controls differ from first-party Anthropic: positive levels
35
+ * go through `output_config.effort` (adaptive path, forced via compat) while
36
+ * `thinking:{type:"disabled"}` is accepted but ignored, so off is expressed
37
+ * through the chat-template kwarg instead.
38
+ */
39
+ function makeReasoningInjector(
40
+ upstream?: StreamOptions["onPayload"],
41
+ ): NonNullable<StreamOptions["onPayload"]> {
42
+ return async (payload, model) => {
43
+ const next = await upstream?.(payload, model);
44
+ const body = (next !== undefined ? next : payload) as AnthropicMessagesBody;
45
+
46
+ if (body.thinking?.type === "disabled") {
47
+ return applyReasoningOff(body);
48
+ }
49
+
50
+ const effort = body.output_config?.effort;
51
+ if (effort && EFFORT_OFF_VALUES.has(effort)) {
52
+ return applyReasoningOff(body);
53
+ }
54
+
55
+ return body;
56
+ };
57
+ }
58
+
59
+ // The Anthropic SDK appends `/v1/messages` to the client base URL.
60
+ function toMessagesBaseUrl(baseUrl: string): string {
61
+ return baseUrl.replace(/\/v1\/?$/, "");
62
+ }
63
+
64
+ function stampAnthropicModels(models: NeuralwattModel[]) {
65
+ return models.map((model) => {
66
+ const { reasoningContract, ...compiled } = model;
67
+ // No retained contract: the identity map is the alias-free special case.
68
+ const thinkingLevelMap = model.reasoning
69
+ ? reasoningContract
70
+ ? buildAnthropicThinkingLevelMap(reasoningContract)
71
+ : model.thinkingLevelMap
72
+ ? { ...model.thinkingLevelMap }
73
+ : undefined
74
+ : undefined;
75
+
76
+ return {
77
+ ...compiled,
78
+ api: "anthropic-messages" as const,
79
+ provider: NEURALWATT_PROVIDER_ID,
80
+ baseUrl: toMessagesBaseUrl(model.baseUrl ?? NEURALWATT_BASE_URL),
81
+ headers: NEURALWATT_REQUEST_HEADERS,
82
+ compat: {
83
+ forceAdaptiveThinking: true,
84
+ supportsTemperature: true,
85
+ supportsStrictTools: false,
86
+ supportsCacheControlOnTools: false,
87
+ },
88
+ ...(thinkingLevelMap ? { thinkingLevelMap } : {}),
89
+ };
90
+ });
91
+ }
92
+
93
+ export function createAnthropicMessagesApi(options?: {
94
+ streamSimple?: AnyStreamSimple;
95
+ }): NeuralwattApiHandler {
96
+ const withReasoning = (options?: {
97
+ onPayload?: StreamOptions["onPayload"];
98
+ }) => ({
99
+ ...options,
100
+ onPayload: makeReasoningInjector(options?.onPayload),
101
+ });
102
+
103
+ return {
104
+ stampModels: stampAnthropicModels,
105
+ stream: (model, context, streamOptions) =>
106
+ stream(model, context, withReasoning(streamOptions) as never),
107
+ streamSimple: (model, context, simpleOptions) =>
108
+ (options?.streamSimple ?? streamSimple)(
109
+ model,
110
+ context,
111
+ withReasoning(simpleOptions) as never,
112
+ ),
113
+ };
114
+ }
@@ -0,0 +1,30 @@
1
+ import { stream, streamSimple } from "@earendil-works/pi-ai/compat";
2
+ import {
3
+ NEURALWATT_BASE_URL,
4
+ NEURALWATT_PROVIDER_ID,
5
+ NEURALWATT_REQUEST_HEADERS,
6
+ } from "../constants";
7
+ import type { NeuralwattModel } from "../models/catalog";
8
+ import type { AnyStreamSimple } from "../stream-simple";
9
+ import type { NeuralwattApiHandler } from "./types";
10
+
11
+ export function createOpenAiCompletionsApi(options?: {
12
+ streamSimple?: AnyStreamSimple;
13
+ }): NeuralwattApiHandler {
14
+ return {
15
+ stampModels: (models: NeuralwattModel[]) =>
16
+ models.map((model) => {
17
+ const { reasoningContract: _reasoningContract, ...compiled } = model;
18
+ return {
19
+ ...compiled,
20
+ api: "openai-completions" as const,
21
+ provider: NEURALWATT_PROVIDER_ID,
22
+ baseUrl: model.baseUrl ?? NEURALWATT_BASE_URL,
23
+ headers: NEURALWATT_REQUEST_HEADERS,
24
+ };
25
+ }),
26
+ stream: (model, context, streamOptions) =>
27
+ stream(model, context, streamOptions as never),
28
+ streamSimple: options?.streamSimple ?? streamSimple,
29
+ };
30
+ }
@@ -0,0 +1,24 @@
1
+ import type {
2
+ Api,
3
+ AssistantMessageEventStream,
4
+ Context,
5
+ Model,
6
+ SimpleStreamOptions,
7
+ StreamOptions,
8
+ } from "@earendil-works/pi-ai";
9
+ import type { NeuralwattModel } from "../models/catalog";
10
+
11
+ /** One Neuralwatt API surface: model stamping plus submission plumbing. */
12
+ export interface NeuralwattApiHandler {
13
+ stampModels(models: NeuralwattModel[]): Model<Api>[];
14
+ stream(
15
+ model: Model<Api>,
16
+ context: Context,
17
+ options?: StreamOptions,
18
+ ): AssistantMessageEventStream;
19
+ streamSimple(
20
+ model: Model<Api>,
21
+ context: Context,
22
+ options?: SimpleStreamOptions,
23
+ ): AssistantMessageEventStream;
24
+ }
@@ -6,6 +6,7 @@ import type { ExtensionAPI } from "@earendil-works/pi-coding-agent";
6
6
  import type { SettingItem } from "@earendil-works/pi-tui";
7
7
  import {
8
8
  configLoader,
9
+ type NeuralwattApi,
9
10
  type NeuralwattConfig,
10
11
  type ResolvedNeuralwattConfig,
11
12
  } from "../../../../src/config";
@@ -61,6 +62,9 @@ export function registerNeuralwattSettings(
61
62
  options: RegisterNeuralwattSettingsOptions,
62
63
  ): void {
63
64
  const { getLoadedFeatures } = options;
65
+ // The provider stamps `provider.api` at extension load; a saved change only
66
+ // reaches it after `/reload`.
67
+ let pendingApi: NeuralwattApi | undefined;
64
68
 
65
69
  registerSettingsCommand<NeuralwattConfig, ResolvedNeuralwattConfig>(pi, {
66
70
  commandName: "neuralwatt:settings",
@@ -69,6 +73,19 @@ export function registerNeuralwattSettings(
69
73
  buildSections: (tabConfig, resolved): SettingsSection[] => {
70
74
  const loaded = getLoadedFeatures();
71
75
  return [
76
+ {
77
+ label: "Provider",
78
+ items: [
79
+ {
80
+ id: "api",
81
+ label: "API",
82
+ description:
83
+ "Serve models via the OpenAI-compatible chat/completions endpoint or the Anthropic-compatible /v1/messages endpoint",
84
+ currentValue: tabConfig?.provider?.api ?? resolved.provider.api,
85
+ values: ["openai-completions", "anthropic-messages"],
86
+ },
87
+ ],
88
+ },
72
89
  {
73
90
  label: "Features",
74
91
  items: [
@@ -107,6 +124,20 @@ export function registerNeuralwattSettings(
107
124
  ];
108
125
  },
109
126
  onSettingChange: (id, newValue, config) => {
127
+ if (id === "api") {
128
+ if (
129
+ newValue !== "openai-completions" &&
130
+ newValue !== "anthropic-messages"
131
+ ) {
132
+ return null;
133
+ }
134
+ pendingApi = newValue;
135
+ return {
136
+ ...config,
137
+ provider: { ...config.provider, api: newValue },
138
+ };
139
+ }
140
+
110
141
  if (!getLoadedFeatures().has(id as NeuralwattFeatureId)) {
111
142
  return null;
112
143
  }
@@ -132,8 +163,11 @@ export function registerNeuralwattSettings(
132
163
  return null;
133
164
  }
134
165
  },
135
- onSave: async () => {
166
+ onSave: async (ctx) => {
136
167
  emitConfigUpdated(pi);
168
+ if (pendingApi === undefined) return;
169
+ pendingApi = undefined;
170
+ ctx.ui.notify("Run /reload to apply the new API", "info");
137
171
  },
138
172
  });
139
173
  }
@@ -0,0 +1,7 @@
1
+ export const NEURALWATT_PROVIDER_ID = "neuralwatt";
2
+ export const NEURALWATT_BASE_URL = "https://api.neuralwatt.com/v1";
3
+ export const NEURALWATT_API_KEY_ENV = "NEURALWATT_API_KEY";
4
+ export const NEURALWATT_REQUEST_HEADERS = {
5
+ Referer: "https://pi.dev",
6
+ "X-Title": "npm:@aliou/pi-neuralwatt",
7
+ };
@@ -55,6 +55,15 @@ function registerNeuralwattProvider(
55
55
  ) as never)
56
56
  : undefined;
57
57
 
58
+ const messagesApiProvider = getApiProvider("anthropic-messages");
59
+ const messagesBaseStreamSimple = messagesApiProvider?.streamSimple;
60
+ const messagesStreamSimple = messagesBaseStreamSimple
61
+ ? (wrapNeuralwattStreamSimple(
62
+ messagesBaseStreamSimple as never,
63
+ onSseQuota,
64
+ ) as never)
65
+ : undefined;
66
+
58
67
  pi.registerProvider(
59
68
  createNeuralwattProvider(
60
69
  staticModels,
@@ -65,7 +74,11 @@ function registerNeuralwattProvider(
65
74
  }
66
75
  return result.data;
67
76
  },
68
- streamSimple,
77
+ {
78
+ api: configLoader.getConfig().provider.api,
79
+ openAiStreamSimple: streamSimple,
80
+ messagesStreamSimple,
81
+ },
69
82
  ),
70
83
  );
71
84
  }
@@ -9,13 +9,13 @@ export type ThinkingLevelMap = NonNullable<
9
9
  >;
10
10
 
11
11
  /**
12
- * Flex tier is billed at 65% of standard pricing (35% off) when the request
13
- * streams. A non-streaming request to a `-flex` model silently falls back to
14
- * the standard tier and the standard price.
15
- *
16
- * https://docs.neuralwatt.com/guides/flex-tier.md
12
+ * A compiled provider model plus the reasoning contract it was compiled from,
13
+ * retained for anthropic-messages map derivation. Rides the models store
14
+ * (JSON passthrough); stripped from stamped runtime models.
17
15
  */
18
- export const FLEX_COST_MULTIPLIER = 0.65;
16
+ export type NeuralwattCompiledModel = ProviderModelConfig & {
17
+ reasoningContract?: NeuralwattReasoningMapSource;
18
+ };
19
19
 
20
20
  export interface NeuralwattCost {
21
21
  input: number;
@@ -65,7 +65,7 @@ export interface NeuralwattVariantSpec {
65
65
  */
66
66
  export type NeuralwattReasoningMapSource = Pick<
67
67
  NeuralwattApiModelReasoning,
68
- "supported_efforts" | "mandatory"
68
+ "supported_efforts" | "mandatory" | "effort_aliases"
69
69
  >;
70
70
 
71
71
  /**
@@ -80,9 +80,9 @@ export type NeuralwattReasoningMapSource = Pick<
80
80
  * exposes none), falls back to a conservative `high`-only map with `off: null`,
81
81
  * matching the upstream binary thinking toggle.
82
82
  *
83
- * `default_effort` and `effort_aliases` are deliberately ignored: Pi has no
84
- * default-reasoning field, and we expose native supported efforts rather than
85
- * aliasing unsupported ones.
83
+ * `effort_aliases` is deliberately ignored here (the openai-completions
84
+ * gateway aliases unsupported efforts server-side); it is consumed by the
85
+ * anthropic-messages map below.
86
86
  */
87
87
  export function buildThinkingLevelMap(
88
88
  reasoning: NeuralwattReasoningMapSource | undefined,
@@ -105,6 +105,39 @@ export function buildThinkingLevelMap(
105
105
  };
106
106
  }
107
107
 
108
+ /**
109
+ * Thinking level map for the anthropic-messages surface. vLLM's
110
+ * `output_config.effort` accepts only the model's native efforts, so
111
+ * unsupported Pi levels resolve through `effort_aliases` (or `null`). A level
112
+ * may resolve to `"none"` — off on this surface, handled by the payload
113
+ * injector in `api/anthropic-messages.ts`.
114
+ */
115
+ export function buildAnthropicThinkingLevelMap(
116
+ reasoning: NeuralwattReasoningMapSource | undefined,
117
+ ): ThinkingLevelMap {
118
+ const supported = new Set<string>(reasoning?.supported_efforts ?? ["high"]);
119
+ const mandatory = reasoning?.mandatory ?? true;
120
+ const aliases = reasoning?.effort_aliases ?? {};
121
+
122
+ const resolve = (level: string): string | null => {
123
+ if (supported.has(level)) return level;
124
+ const alias = aliases[level as keyof typeof aliases];
125
+ return alias && supported.has(alias) ? alias : null;
126
+ };
127
+
128
+ return {
129
+ // "none" is a marker so pi-ai enables the off path (off !== null);
130
+ // vLLM rejects it on the wire, so the injector never sends it verbatim.
131
+ off: !mandatory && supported.has("none") ? "none" : null,
132
+ minimal: resolve("minimal"),
133
+ low: resolve("low"),
134
+ medium: resolve("medium"),
135
+ high: resolve("high"),
136
+ xhigh: resolve("xhigh"),
137
+ max: resolve("max"),
138
+ };
139
+ }
140
+
108
141
  /**
109
142
  * Neuralwatt reports `max_output_tokens: null` for models whose output is only
110
143
  * bounded by the context window. Some models incorrectly report 0; treat 0
@@ -121,7 +154,7 @@ export function resolveMaxTokens(
121
154
  export function buildNeuralwattModel(
122
155
  family: NeuralwattModelFamily,
123
156
  variant: NeuralwattVariantSpec,
124
- ): ProviderModelConfig {
157
+ ): NeuralwattCompiledModel {
125
158
  const vision = variant.vision ?? family.vision;
126
159
 
127
160
  const compat: NonNullable<ProviderModelConfig["compat"]> = {
@@ -136,7 +169,7 @@ export function buildNeuralwattModel(
136
169
  const scale = (value: number): number =>
137
170
  multiplier === 1 ? value : Number((value * multiplier).toFixed(6));
138
171
 
139
- const model: ProviderModelConfig = {
172
+ const model: NeuralwattCompiledModel = {
140
173
  id: variant.id,
141
174
  name: variant.name,
142
175
  reasoning: variant.reasoning,
@@ -153,14 +186,14 @@ export function buildNeuralwattModel(
153
186
  };
154
187
 
155
188
  if (variant.reasoning) {
189
+ const contract = variant.reasoningMetadata ?? family.reasoningMetadata;
156
190
  // Clone so variants never share a family map instance. The map is derived
157
191
  // from the API reasoning contract; missing metadata falls back to a
158
192
  // high-only map rather than throwing.
159
193
  model.thinkingLevelMap = {
160
- ...buildThinkingLevelMap(
161
- variant.reasoningMetadata ?? family.reasoningMetadata,
162
- ),
194
+ ...buildThinkingLevelMap(contract),
163
195
  };
196
+ model.reasoningContract = contract;
164
197
  }
165
198
 
166
199
  return model;
@@ -169,6 +202,6 @@ export function buildNeuralwattModel(
169
202
  export function buildNeuralwattFamily(
170
203
  family: NeuralwattModelFamily,
171
204
  variants: NeuralwattVariantSpec[],
172
- ): ProviderModelConfig[] {
205
+ ): NeuralwattCompiledModel[] {
173
206
  return variants.map((variant) => buildNeuralwattModel(family, variant));
174
207
  }
@@ -2,13 +2,13 @@ import type { ProviderModelConfig } from "@earendil-works/pi-coding-agent";
2
2
  import type { NeuralwattApiModel } from "../../../src/types/models-api";
3
3
  import {
4
4
  buildThinkingLevelMap,
5
- FLEX_COST_MULTIPLIER,
5
+ type NeuralwattCompiledModel,
6
6
  resolveMaxTokens,
7
7
  type ThinkingLevelMap,
8
8
  } from "./build";
9
9
  import { NEURALWATT_MODELS } from "./public-models";
10
10
 
11
- export type NeuralwattModel = ProviderModelConfig;
11
+ export type NeuralwattModel = NeuralwattCompiledModel;
12
12
 
13
13
  // Chat-template thinking: the API exposes a `reasoning` block, but the
14
14
  // underlying mechanism is chat_template_kwargs, so Pi needs the mapping.
@@ -23,21 +23,6 @@ const COMPAT_OVERRIDES: Partial<
23
23
  },
24
24
  };
25
25
 
26
- const HARDCODED_ALIASES: Record<string, string> = {
27
- "zai-org/GLM-5.2-FP8": "glm-5.2",
28
- "moonshotai/Kimi-K2.7-Code": "kimi-k2.7-code",
29
- "Qwen/Qwen3.6-35B-A3B": "qwen3.6-35b",
30
- "deepseek-ai/DeepSeek-V4-Flash": "deepseek-v4-flash",
31
- };
32
-
33
- function isFlexModelId(id: string): boolean {
34
- return id.endsWith("-flex");
35
- }
36
-
37
- function isVariantId(id: string): boolean {
38
- return id.includes("-fast") || id.includes("-flex") || id.includes("-short");
39
- }
40
-
41
26
  function apiModelToProviderModel(model: NeuralwattApiModel): NeuralwattModel {
42
27
  const meta = model.metadata;
43
28
  if (!meta)
@@ -46,8 +31,6 @@ function apiModelToProviderModel(model: NeuralwattApiModel): NeuralwattModel {
46
31
  );
47
32
 
48
33
  const reasoning = meta.capabilities.reasoning;
49
- // Flex variants are billed at 0.65x when streaming (35% off).
50
- const multiplier = isFlexModelId(model.id) ? FLEX_COST_MULTIPLIER : 1;
51
34
 
52
35
  const compat: NonNullable<ProviderModelConfig["compat"]> = {
53
36
  supportsDeveloperRole: meta.capabilities.developer_role,
@@ -66,10 +49,10 @@ function apiModelToProviderModel(model: NeuralwattApiModel): NeuralwattModel {
66
49
  ? (["text", "image"] as const)
67
50
  : (["text"] as const),
68
51
  cost: {
69
- input: meta.pricing.input_per_million * multiplier,
70
- output: meta.pricing.output_per_million * multiplier,
71
- cacheRead: (meta.pricing.cached_input_per_million ?? 0) * multiplier,
72
- cacheWrite: (meta.pricing.cached_output_per_million ?? 0) * multiplier,
52
+ input: meta.pricing.input_per_million,
53
+ output: meta.pricing.output_per_million,
54
+ cacheRead: meta.pricing.cached_input_per_million ?? 0,
55
+ cacheWrite: meta.pricing.cached_output_per_million ?? 0,
73
56
  },
74
57
  contextWindow,
75
58
  maxTokens: resolveMaxTokens(meta.limits.max_output_tokens, contextWindow),
@@ -80,44 +63,14 @@ function apiModelToProviderModel(model: NeuralwattApiModel): NeuralwattModel {
80
63
  result.thinkingLevelMap = buildThinkingLevelMap(
81
64
  meta.reasoning,
82
65
  ) as ThinkingLevelMap;
66
+ // Kept for anthropic-messages stamping, which resolves levels through
67
+ // `effort_aliases`.
68
+ result.reasoningContract = meta.reasoning;
83
69
  }
84
70
 
85
71
  return result;
86
72
  }
87
73
 
88
- function buildAliases(
89
- models: NeuralwattModel[],
90
- apiModels: readonly NeuralwattApiModel[],
91
- ): NeuralwattModel[] {
92
- const existingIds = new Set(models.map((m) => m.id));
93
- const aliases: NeuralwattModel[] = [];
94
- const seen = new Set<string>();
95
-
96
- const addAlias = (aliasId: string, canonicalId: string): void => {
97
- if (seen.has(aliasId) || existingIds.has(aliasId)) return;
98
- const canonical = models.find((m) => m.id === canonicalId);
99
- if (!canonical) return;
100
- seen.add(aliasId);
101
- aliases.push({
102
- ...canonical,
103
- id: aliasId,
104
- name: `${canonical.name} (alias ID)`,
105
- });
106
- };
107
-
108
- for (const [aliasId, canonicalId] of Object.entries(HARDCODED_ALIASES)) {
109
- addAlias(aliasId, canonicalId);
110
- }
111
-
112
- for (const apiModel of apiModels) {
113
- const hfId = apiModel.metadata?.huggingface_id;
114
- if (!hfId || hfId === apiModel.id || isVariantId(apiModel.id)) continue;
115
- addAlias(hfId, apiModel.id);
116
- }
117
-
118
- return aliases;
119
- }
120
-
121
74
  export function buildNeuralwattProviderModels(): NeuralwattModel[] {
122
75
  return NEURALWATT_MODELS.map((model) => ({ ...model }));
123
76
  }
@@ -138,7 +91,7 @@ export function buildNeuralwattProviderModelsFromApi(
138
91
  ),
139
92
  )
140
93
  .map(apiModelToProviderModel);
141
- return [...models, ...buildAliases(models, apiModels)];
94
+ return models;
142
95
  }
143
96
 
144
97
  export function buildNeuralwattProviderModelsFromStore(
@@ -1,7 +1,6 @@
1
1
  import type { ProviderModelConfig } from "@earendil-works/pi-coding-agent";
2
2
  import {
3
3
  buildNeuralwattFamily,
4
- FLEX_COST_MULTIPLIER,
5
4
  type NeuralwattModelFamily,
6
5
  type NeuralwattVariantSpec,
7
6
  } from "./build";
@@ -41,25 +40,23 @@ const GEMMA_4: NeuralwattModelFamily = {
41
40
  },
42
41
  };
43
42
 
44
- // ZhipuAI. GLM-5.2 natively supports `high` and `max` reasoning efforts;
45
- // `xhigh` is an unsupported hole between them. Pi's `max` level (0.80.6) maps
46
- // to GLM's top tier.
47
- const GLM_5_2: NeuralwattModelFamily = {
43
+ // ZhipuAI. GLM-5.3 has mandatory reasoning and `none` is not offered:
44
+ // efforts are max/high/low (default max).
45
+ const GLM_5_3: NeuralwattModelFamily = {
48
46
  cost: { input: 1.45, output: 4.5, cacheRead: 0.145 },
49
47
  vision: false,
50
48
  reasoningMetadata: {
51
- supported_efforts: ["max", "high", "none"],
52
- mandatory: false,
49
+ supported_efforts: ["max", "high", "low"],
50
+ mandatory: true,
53
51
  },
54
52
  };
55
53
 
56
- // ZhipuAI. GLM-5.3 ships as a GLM-5.2 weight swap in gated preview, with
57
- // GLM-5.2 pricing parity (per the API metadata; review at launch). Unlike
58
- // 5.2, reasoning is mandatory and `none` is not offered: efforts are
59
- // max/high/low (default max).
60
- const GLM_5_3: NeuralwattModelFamily = {
61
- cost: { input: 1.45, output: 4.5, cacheRead: 0.145 },
62
- vision: false,
54
+ // ZhipuAI. GLM-5.3 Flash is the small GLM-5.3 tier: vision-capable, much
55
+ // cheaper than the flagship, with the same mandatory max/high/low reasoning
56
+ // contract as GLM-5.3.
57
+ const GLM_5_3_FLASH: NeuralwattModelFamily = {
58
+ cost: { input: 0.15, output: 0.5, cacheRead: 0.03 },
59
+ vision: true,
63
60
  reasoningMetadata: {
64
61
  supported_efforts: ["max", "high", "low"],
65
62
  mandatory: true,
@@ -99,6 +96,18 @@ const QWEN_3_6_35B: NeuralwattModelFamily = {
99
96
  },
100
97
  };
101
98
 
99
+ // Qwen. Qwen 3.8 27B tops out at `xhigh` (its default) and also supports
100
+ // `medium`, `low`, and `none`; there is no `max` effort. Reasoning is on by
101
+ // default but can be disabled.
102
+ const QWEN_3_8_27B: NeuralwattModelFamily = {
103
+ cost: { input: 0.45, output: 3.2, cacheRead: 0.25 },
104
+ vision: true,
105
+ reasoningMetadata: {
106
+ supported_efforts: ["xhigh", "medium", "low", "none"],
107
+ mandatory: false,
108
+ },
109
+ };
110
+
102
111
  const FAMILIES: [NeuralwattModelFamily, NeuralwattVariantSpec[]][] = [
103
112
  [
104
113
  DEEPSEEK_V4_FLASH,
@@ -116,7 +125,14 @@ const FAMILIES: [NeuralwattModelFamily, NeuralwattVariantSpec[]][] = [
116
125
  contextWindow: 1048560,
117
126
  maxOutputTokens: 65536,
118
127
  reasoning: true,
119
- costMultiplier: FLEX_COST_MULTIPLIER,
128
+ costMultiplier: 0.65,
129
+ },
130
+ {
131
+ id: "deepseek-v4-flash-speed",
132
+ name: "DeepSeek V4 Flash (Speed)",
133
+ contextWindow: 1048560,
134
+ maxOutputTokens: 65536,
135
+ reasoning: true,
120
136
  },
121
137
  ],
122
138
  ],
@@ -133,78 +149,42 @@ const FAMILIES: [NeuralwattModelFamily, NeuralwattVariantSpec[]][] = [
133
149
  ],
134
150
  ],
135
151
  [
136
- GLM_5_2,
152
+ GLM_5_3,
137
153
  [
138
154
  {
139
- id: "glm-5.2",
140
- name: "GLM-5.2",
155
+ id: "glm-5.3",
156
+ name: "GLM 5.3",
141
157
  contextWindow: 1048560,
142
158
  maxOutputTokens: null,
143
159
  reasoning: true,
144
160
  },
145
161
  {
146
- // GLM-5.2 Fast pins thinking off by default, but keeps the parent's
147
- // full reasoning contract (`high`/`max`/`none`): sending
148
- // `reasoning_effort` re-enables thinking for that request.
149
- id: "glm-5.2-fast",
150
- name: "GLM-5.2 (fast)",
162
+ id: "glm-5.3-flex",
163
+ name: "GLM 5.3 (flex)",
151
164
  contextWindow: 1048560,
152
165
  maxOutputTokens: null,
153
166
  reasoning: true,
167
+ costMultiplier: 0.65,
154
168
  },
169
+ ],
170
+ ],
171
+ [
172
+ GLM_5_3_FLASH,
173
+ [
155
174
  {
156
- id: "glm-5.2-flex",
157
- name: "GLM-5.2 (flex)",
175
+ id: "glm-5.3-flash",
176
+ name: "GLM-5.3 Flash",
158
177
  contextWindow: 1048560,
159
178
  maxOutputTokens: null,
160
179
  reasoning: true,
161
- costMultiplier: FLEX_COST_MULTIPLIER,
162
180
  },
163
181
  {
164
- id: "glm-5.2-short",
165
- name: "GLM-5.2 Short",
166
- contextWindow: 199984,
167
- maxOutputTokens: 32000,
168
- reasoning: true,
169
- },
170
- {
171
- // Short/fast: pins thinking off but keeps the parent reasoning
172
- // contract, like glm-5.2-fast.
173
- id: "glm-5.2-short-fast",
174
- name: "GLM-5.2 (short, fast)",
175
- contextWindow: 199984,
176
- maxOutputTokens: 32000,
177
- reasoning: true,
178
- },
179
- {
180
- id: "glm-5.2-short-flex",
181
- name: "GLM-5.2 (short, flex)",
182
- contextWindow: 199984,
183
- maxOutputTokens: 32000,
184
- reasoning: true,
185
- costMultiplier: FLEX_COST_MULTIPLIER,
186
- },
187
- {
188
- // Short/fast/flex: pins thinking off but keeps the parent reasoning
189
- // contract, like glm-5.2-fast.
190
- id: "glm-5.2-short-fast-flex",
191
- name: "GLM-5.2 (short, fast, flex)",
192
- contextWindow: 199984,
193
- maxOutputTokens: 32000,
194
- reasoning: true,
195
- costMultiplier: FLEX_COST_MULTIPLIER,
196
- },
197
- ],
198
- ],
199
- [
200
- GLM_5_3,
201
- [
202
- {
203
- id: "glm-5.3",
204
- name: "GLM-5.3",
182
+ id: "glm-5.3-flash-flex",
183
+ name: "GLM-5.3 Flash (flex)",
205
184
  contextWindow: 1048560,
206
185
  maxOutputTokens: null,
207
186
  reasoning: true,
187
+ costMultiplier: 0.65,
208
188
  },
209
189
  ],
210
190
  ],
@@ -231,7 +211,7 @@ const FAMILIES: [NeuralwattModelFamily, NeuralwattVariantSpec[]][] = [
231
211
  contextWindow: 1048560,
232
212
  maxOutputTokens: null,
233
213
  reasoning: true,
234
- costMultiplier: FLEX_COST_MULTIPLIER,
214
+ costMultiplier: 0.65,
235
215
  },
236
216
  ],
237
217
  ],
@@ -260,7 +240,7 @@ const FAMILIES: [NeuralwattModelFamily, NeuralwattVariantSpec[]][] = [
260
240
  contextWindow: 262128,
261
241
  maxOutputTokens: null,
262
242
  reasoning: true,
263
- costMultiplier: FLEX_COST_MULTIPLIER,
243
+ costMultiplier: 0.65,
264
244
  },
265
245
  ],
266
246
  ],
@@ -281,15 +261,42 @@ const FAMILIES: [NeuralwattModelFamily, NeuralwattVariantSpec[]][] = [
281
261
  maxOutputTokens: null,
282
262
  reasoning: false,
283
263
  },
264
+ {
265
+ id: "qwen3.6-35b-flex",
266
+ name: "Qwen3.6 35B (flex)",
267
+ contextWindow: 131056,
268
+ maxOutputTokens: null,
269
+ reasoning: true,
270
+ costMultiplier: 0.65,
271
+ },
272
+ ],
273
+ ],
274
+ [
275
+ QWEN_3_8_27B,
276
+ [
277
+ {
278
+ id: "qwen-3.8-27b",
279
+ name: "Qwen 3.8 27B",
280
+ contextWindow: 262128,
281
+ maxOutputTokens: 131072,
282
+ reasoning: true,
283
+ },
284
+ {
285
+ id: "qwen-3.8-27b-flex",
286
+ name: "Qwen 3.8 27B (flex)",
287
+ contextWindow: 262128,
288
+ maxOutputTokens: 131072,
289
+ reasoning: true,
290
+ costMultiplier: 0.65,
291
+ },
284
292
  ],
285
293
  ],
286
294
  ];
287
295
 
288
296
  // `-flex` variants are the Flex tier: same model, context window, output cap,
289
- // and prompt cache as the standard variant, admitted on spare capacity.
290
- // The API now advertises flex variants but lists them at standard pricing;
291
- // the 35% Flex discount is a billing-time concept applied here via
292
- // `costMultiplier` rather than reflected in the catalog metadata.
297
+ // and prompt cache as the standard variant, admitted on spare capacity. The
298
+ // API lists them at discounted prices; the fallback mirrors that with
299
+ // `costMultiplier: 0.65` per variant.
293
300
  // https://docs.neuralwatt.com/guides/flex-tier.md
294
301
 
295
302
  export const NEURALWATT_MODELS: ProviderModelConfig[] = FAMILIES.flatMap(
@@ -1,10 +1,14 @@
1
- import type {
2
- Api,
3
- Model,
4
- Provider,
5
- ProviderStreamOptions,
6
- } from "@earendil-works/pi-ai";
7
- import { stream, streamSimple } from "@earendil-works/pi-ai/compat";
1
+ import type { Provider } from "@earendil-works/pi-ai";
2
+ import type { NeuralwattApi } from "../../src/config";
3
+ import { createAnthropicMessagesApi } from "./api/anthropic-messages";
4
+ import { createOpenAiCompletionsApi } from "./api/openai-completions";
5
+ import type { NeuralwattApiHandler } from "./api/types";
6
+ import {
7
+ NEURALWATT_API_KEY_ENV,
8
+ NEURALWATT_BASE_URL,
9
+ NEURALWATT_PROVIDER_ID,
10
+ NEURALWATT_REQUEST_HEADERS,
11
+ } from "./constants";
8
12
  import type { NeuralwattModel } from "./models/catalog";
9
13
  import {
10
14
  buildNeuralwattProviderModelsFromApi,
@@ -16,33 +20,39 @@ import {
16
20
  } from "./models/refresh";
17
21
  import type { AnyStreamSimple } from "./stream-simple";
18
22
 
19
- export const NEURALWATT_PROVIDER_ID = "neuralwatt";
20
- export const NEURALWATT_BASE_URL = "https://api.neuralwatt.com/v1";
21
- export const NEURALWATT_API_KEY_ENV = "NEURALWATT_API_KEY";
22
-
23
- const NEURALWATT_REQUEST_HEADERS = {
24
- Referer: "https://pi.dev",
25
- "X-Title": "npm:@aliou/pi-neuralwatt",
26
- };
23
+ export { NEURALWATT_API_KEY_ENV, NEURALWATT_BASE_URL, NEURALWATT_PROVIDER_ID };
27
24
 
28
- const API = "openai-completions" as const;
25
+ export interface NeuralwattProviderOptions {
26
+ /** Active API surface; resolved once. Changes need a `/reload`. */
27
+ api?: NeuralwattApi;
28
+ openAiStreamSimple?: AnyStreamSimple;
29
+ messagesStreamSimple?: AnyStreamSimple;
30
+ }
29
31
 
30
- function toProviderModels(models: NeuralwattModel[]): Model<Api>[] {
31
- return models.map((model) => ({
32
- ...model,
33
- api: model.api ?? API,
34
- provider: NEURALWATT_PROVIDER_ID,
35
- baseUrl: model.baseUrl ?? NEURALWATT_BASE_URL,
36
- headers: NEURALWATT_REQUEST_HEADERS,
37
- }));
32
+ function createApiHandler(
33
+ api: NeuralwattApi,
34
+ options?: NeuralwattProviderOptions,
35
+ ): NeuralwattApiHandler {
36
+ if (api === "anthropic-messages") {
37
+ return createAnthropicMessagesApi({
38
+ streamSimple: options?.messagesStreamSimple,
39
+ });
40
+ }
41
+ return createOpenAiCompletionsApi({
42
+ streamSimple: options?.openAiStreamSimple,
43
+ });
38
44
  }
39
45
 
40
46
  export function createNeuralwattProvider(
41
47
  staticModels: NeuralwattModel[],
42
48
  fetchApiModels: FetchNeuralwattApiModels,
43
- streamSimpleOverride?: AnyStreamSimple,
49
+ options?: NeuralwattProviderOptions,
44
50
  ): Provider {
45
- let liveModels = toProviderModels(staticModels);
51
+ const handler = createApiHandler(
52
+ options?.api ?? "openai-completions",
53
+ options,
54
+ );
55
+ let canonicalModels = staticModels;
46
56
  const refreshCatalog = createNeuralwattRefreshModels(
47
57
  staticModels,
48
58
  fetchApiModels,
@@ -92,17 +102,18 @@ export function createNeuralwattProvider(
92
102
  },
93
103
  },
94
104
  },
95
- getModels: () => liveModels,
105
+ getModels: () => handler.stampModels(canonicalModels),
96
106
  refreshModels: async (context) => {
97
107
  const models = await refreshCatalog(context);
98
108
  await context.publish({
99
109
  update: () => {
100
- liveModels = toProviderModels(models);
110
+ canonicalModels = models;
101
111
  },
102
112
  });
103
113
  },
104
- stream: (model, context, options) =>
105
- stream(model, context, options as ProviderStreamOptions | undefined),
106
- streamSimple: streamSimpleOverride ?? streamSimple,
114
+ stream: (model, context, streamOptions) =>
115
+ handler.stream(model, context, streamOptions as never),
116
+ streamSimple: (model, context, simpleOptions) =>
117
+ handler.streamSimple(model, context, simpleOptions),
107
118
  };
108
119
  }
@@ -31,13 +31,26 @@ export function updateQuotasFromSseComment(
31
31
  if (trimmed.startsWith(": cost ")) {
32
32
  const cost = JSON.parse(trimmed.slice(7)) as {
33
33
  request_cost_usd?: number;
34
+ allowance_remaining_usd?: number;
34
35
  };
35
36
  const requestCostUsd = cost.request_cost_usd ?? 0;
36
37
  if (requestCostUsd <= 0) return quotas;
37
- next.balance.credits_remaining_usd = Math.max(
38
- 0,
39
- next.balance.credits_remaining_usd - requestCostUsd,
40
- );
38
+ // The absolute allowance in the comment is fresher than the locally
39
+ // tracked total.
40
+ if (
41
+ typeof cost.allowance_remaining_usd === "number" &&
42
+ Number.isFinite(cost.allowance_remaining_usd)
43
+ ) {
44
+ next.balance.credits_remaining_usd = Math.max(
45
+ 0,
46
+ cost.allowance_remaining_usd,
47
+ );
48
+ } else {
49
+ next.balance.credits_remaining_usd = Math.max(
50
+ 0,
51
+ next.balance.credits_remaining_usd - requestCostUsd,
52
+ );
53
+ }
41
54
  next.balance.credits_used_usd += requestCostUsd;
42
55
  next.usage.current_month.cost_usd += requestCostUsd;
43
56
  next.usage.lifetime.cost_usd += requestCostUsd;
@@ -35,7 +35,7 @@ function headersToRecord(headers: Headers): Record<string, string> {
35
35
  return record;
36
36
  }
37
37
 
38
- function isProviderChatCompletionsUrl(
38
+ function isProviderStreamUrl(
39
39
  input: RequestInfo | URL,
40
40
  providerOrigin: string,
41
41
  ): boolean {
@@ -50,7 +50,8 @@ function isProviderChatCompletionsUrl(
50
50
  const url = new URL(rawUrl);
51
51
  return (
52
52
  url.origin === providerOrigin &&
53
- url.pathname.endsWith("/chat/completions")
53
+ (url.pathname.endsWith("/chat/completions") ||
54
+ url.pathname.endsWith("/messages"))
54
55
  );
55
56
  } catch {
56
57
  return false;
@@ -96,7 +97,7 @@ export function wrapNeuralwattStreamSimple(
96
97
  const wrappedFetch: typeof fetch = async (input, init) => {
97
98
  const response = await originalFetch(input, init);
98
99
 
99
- if (!isProviderChatCompletionsUrl(input, providerOrigin)) return response;
100
+ if (!isProviderStreamUrl(input, providerOrigin)) return response;
100
101
 
101
102
  const headers = headersToRecord(response.headers);
102
103
  if (response.status === 429) {
@@ -8,10 +8,37 @@ const COOLDOWN_MS = 60 * 60 * 1000; // 60 minutes
8
8
  const LOW_PCT = 25;
9
9
  const CRITICAL_PCT = 10;
10
10
 
11
- /** Per-kWh price once a subscription's included kWh are exhausted. */
12
- const OVERAGE_RATE_PER_KWH_SUBSCRIBED = 5;
13
- /** Per-kWh price when there is no active subscription (no included kWh). */
11
+ /** $/kWh by plan on a monthly interval. docs.neuralwatt.com/billing/faq */
12
+ const OVERAGE_RATES_MONTHLY = {
13
+ basic: 8.5,
14
+ standard: 8.0,
15
+ pro: 7.5,
16
+ max: 7.0,
17
+ } as const;
18
+ /** $/kWh by plan on an annual interval. */
19
+ const OVERAGE_RATES_ANNUAL = {
20
+ basic: 7.08,
21
+ standard: 6.67,
22
+ pro: 6.25,
23
+ max: 5.83,
24
+ } as const;
25
+ /** Pay-as-you-go, verified on portal.neuralwatt.com/pricing. */
14
26
  const OVERAGE_RATE_PER_KWH_UNSUBSCRIBED = 10;
27
+ /** Unknown plan on a subscription: fall back to the Standard monthly rate. */
28
+ const OVERAGE_RATE_PER_KWH_UNKNOWN_SUBSCRIBED = 8.0;
29
+
30
+ function resolveOverageRate(sub: NeuralwattQuotas["subscription"]): number {
31
+ if (!sub) return OVERAGE_RATE_PER_KWH_UNSUBSCRIBED;
32
+ const plan = sub.plan.toLowerCase();
33
+ const table =
34
+ sub.billing_interval === "year"
35
+ ? OVERAGE_RATES_ANNUAL
36
+ : OVERAGE_RATES_MONTHLY;
37
+ return (
38
+ (table as Record<string, number>)[plan] ??
39
+ OVERAGE_RATE_PER_KWH_UNKNOWN_SUBSCRIBED
40
+ );
41
+ }
15
42
 
16
43
  interface AlertState {
17
44
  lastSeverity: WarningSeverity;
@@ -112,9 +139,7 @@ export function computeOverageProgress(
112
139
  )
113
140
  : quotas.usage.current_month.energy_kwh;
114
141
 
115
- const rate = hasSub
116
- ? OVERAGE_RATE_PER_KWH_SUBSCRIBED
117
- : OVERAGE_RATE_PER_KWH_UNSUBSCRIBED;
142
+ const rate = resolveOverageRate(quotas.subscription);
118
143
  const costUsd = overageKwh * rate;
119
144
  const remainingUsd = Math.max(0, capUsd - costUsd);
120
145
  const pctRemaining = capUsd > 0 ? (remainingUsd / capUsd) * 100 : 0;
@@ -153,9 +178,12 @@ function overageWarning(progress: OverageProgress): PendingWarning {
153
178
  * no subscription, cap set → overage cap progress (all kWh billable)
154
179
  * no subscription, no cap → balance credits
155
180
  *
156
- * Overage cost is derived from kWh usage: subscribed pays $5/kWh for kWh
157
- * beyond the included quota; unsubscribed pays $10/kWh for all usage. There is
158
- * no overage-spent counter in the API, so progress is computed.
181
+ * Overage cost is derived from kWh usage: subscribed pays a per-plan rate
182
+ * ($7.00–$8.50/kWh by plan and billing interval) for kWh beyond the included
183
+ * quota; unsubscribed pays $10/kWh for all usage. There is no overage-spent
184
+ * counter in the API, so progress is computed. Note `kwh_used` is the
185
+ * *charged* energy — flex usage bills at 0.65× kWh — so the "kWh over" figure
186
+ * is billed kWh, not physical consumption.
159
187
  *
160
188
  * Usage totals (monthly/lifetime cost in USD) are deliberately not used as a
161
189
  * threshold basis — they are not directly tied to the subscription's kWh quota.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@aliou/pi-neuralwatt",
3
- "version": "0.15.3",
3
+ "version": "0.16.0",
4
4
  "license": "MIT",
5
5
  "type": "module",
6
6
  "private": false,
@@ -69,6 +69,7 @@
69
69
  "test": "vitest run",
70
70
  "test:watch": "vitest",
71
71
  "check:models": "tsx scripts/check-models.ts",
72
+ "check:changesets": "tsx scripts/check-changesets.ts",
72
73
  "gen:schema": "pi-settings-schema -p src/config/types.ts -t NeuralwattConfig -o schema.json --version 0.12.0",
73
74
  "check:schema": "pi-settings-schema -p src/config/types.ts -t NeuralwattConfig -o schema.json --version 0.12.0 --check",
74
75
  "check:lockfile": "pnpm install --frozen-lockfile --ignore-scripts",
package/schema.json CHANGED
@@ -20,6 +20,10 @@
20
20
  "$ref": "#/definitions/NeuralwattSubBarIntegrationConfig",
21
21
  "description": "Sub-bar/status-bar integration feature."
22
22
  },
23
+ "provider": {
24
+ "$ref": "#/definitions/NeuralwattProviderConfig",
25
+ "description": "Provider behavior (API surface)."
26
+ },
23
27
  "version": {
24
28
  "anyOf": [
25
29
  {
@@ -65,6 +69,24 @@
65
69
  }
66
70
  },
67
71
  "additionalProperties": false
72
+ },
73
+ "NeuralwattProviderConfig": {
74
+ "type": "object",
75
+ "properties": {
76
+ "api": {
77
+ "$ref": "#/definitions/NeuralwattApi",
78
+ "description": "Which API serves model requests."
79
+ }
80
+ },
81
+ "additionalProperties": false
82
+ },
83
+ "NeuralwattApi": {
84
+ "type": "string",
85
+ "enum": [
86
+ "openai-completions",
87
+ "anthropic-messages"
88
+ ],
89
+ "description": "Neuralwatt serves every chat model twice: on an OpenAI-compatible `chat/completions` endpoint and on a vLLM-backed Anthropic-compatible `POST /v1/messages` endpoint. Exactly one serves the provider at a time."
68
90
  }
69
91
  }
70
92
  }
@@ -10,4 +10,7 @@ export const DEFAULT_CONFIG: ResolvedNeuralwattConfig = {
10
10
  subBarIntegration: {
11
11
  enabled: true,
12
12
  },
13
+ provider: {
14
+ api: "openai-completions",
15
+ },
13
16
  };
@@ -1,4 +1,8 @@
1
1
  export { DEFAULT_CONFIG } from "./defaults";
2
- export { configLoader } from "./loader";
2
+ export { configLoader, resolveApi } from "./loader";
3
3
  export { migrations } from "./migration";
4
- export type { NeuralwattConfig, ResolvedNeuralwattConfig } from "./types";
4
+ export type {
5
+ NeuralwattApi,
6
+ NeuralwattConfig,
7
+ ResolvedNeuralwattConfig,
8
+ } from "./types";
@@ -2,7 +2,11 @@ import { buildSchemaUrl, ConfigLoader } from "@aliou/pi-utils-settings";
2
2
  import packageJson from "../../package.json";
3
3
  import { DEFAULT_CONFIG } from "./defaults";
4
4
  import { migrations } from "./migration";
5
- import type { NeuralwattConfig, ResolvedNeuralwattConfig } from "./types";
5
+ import type {
6
+ NeuralwattApi,
7
+ NeuralwattConfig,
8
+ ResolvedNeuralwattConfig,
9
+ } from "./types";
6
10
 
7
11
  /**
8
12
  * Fill in every field the rest of the code reads. Migrations already normalized
@@ -27,9 +31,19 @@ function normalizeResolvedConfig(
27
31
  config.subBarIntegration?.enabled ??
28
32
  DEFAULT_CONFIG.subBarIntegration.enabled,
29
33
  },
34
+ provider: {
35
+ api: resolveApi(config.provider?.api),
36
+ },
30
37
  };
31
38
  }
32
39
 
40
+ export function resolveApi(value: string | undefined): NeuralwattApi {
41
+ if (value === "anthropic-messages" || value === "openai-completions") {
42
+ return value;
43
+ }
44
+ return DEFAULT_CONFIG.provider.api;
45
+ }
46
+
33
47
  export const configLoader = new ConfigLoader<
34
48
  NeuralwattConfig,
35
49
  ResolvedNeuralwattConfig
@@ -6,11 +6,9 @@ export {
6
6
  flatToNestedConfigMigration,
7
7
  } from "./02-flat-to-nested-config";
8
8
  export { renameHiddenToEarlyAccessMigration } from "./03-rename-hidden-to-early-access";
9
- export { enableAliasesForLegacyUsersMigration } from "./04-enable-aliases-for-legacy-users";
10
9
 
11
10
  import { flatToNestedConfigMigration } from "./02-flat-to-nested-config";
12
11
  import { renameHiddenToEarlyAccessMigration } from "./03-rename-hidden-to-early-access";
13
- import { enableAliasesForLegacyUsersMigration } from "./04-enable-aliases-for-legacy-users";
14
12
 
15
13
  // Each migration is typed against its own historical input shape. The loader
16
14
  // applies them in sequence on the raw config record, so they are cast to the
@@ -18,5 +16,4 @@ import { enableAliasesForLegacyUsersMigration } from "./04-enable-aliases-for-le
18
16
  export const migrations = [
19
17
  flatToNestedConfigMigration,
20
18
  renameHiddenToEarlyAccessMigration,
21
- enableAliasesForLegacyUsersMigration,
22
19
  ] as unknown as Migration<NeuralwattConfig>[];
@@ -13,6 +13,18 @@ export interface NeuralwattSubBarIntegrationConfig {
13
13
  enabled?: boolean;
14
14
  }
15
15
 
16
+ /**
17
+ * Neuralwatt serves every chat model twice: on an OpenAI-compatible
18
+ * `chat/completions` endpoint and on a vLLM-backed Anthropic-compatible
19
+ * `POST /v1/messages` endpoint. Exactly one serves the provider at a time.
20
+ */
21
+ export type NeuralwattApi = "openai-completions" | "anthropic-messages";
22
+
23
+ export interface NeuralwattProviderConfig {
24
+ /** Which API serves model requests. */
25
+ api?: NeuralwattApi;
26
+ }
27
+
16
28
  export interface NeuralwattConfig {
17
29
  /** $schema URL for editor autocomplete. */
18
30
  $schema?: string;
@@ -25,6 +37,9 @@ export interface NeuralwattConfig {
25
37
 
26
38
  /** Sub-bar/status-bar integration feature. */
27
39
  subBarIntegration?: NeuralwattSubBarIntegrationConfig;
40
+
41
+ /** Provider behavior (API surface). */
42
+ provider?: NeuralwattProviderConfig;
28
43
  }
29
44
 
30
45
  export interface ResolvedNeuralwattConfig {
@@ -37,4 +52,7 @@ export interface ResolvedNeuralwattConfig {
37
52
  subBarIntegration: {
38
53
  enabled: boolean;
39
54
  };
55
+ provider: {
56
+ api: NeuralwattApi;
57
+ };
40
58
  }
@@ -5,6 +5,10 @@ export interface NeuralwattApiModelPricing {
5
5
  cached_output_per_million: number | null;
6
6
  currency: string;
7
7
  pricing_tbd: boolean;
8
+ /** Service tier the pricing applies to (e.g. "standard", "flex"). */
9
+ service_tier?: string;
10
+ /** Flex tier cost multiplier (e.g. 0.65); null/absent on standard pricing. */
11
+ flex_discount_multiplier?: number | null;
8
12
  }
9
13
 
10
14
  export interface NeuralwattApiModelCapabilities {
@@ -34,14 +38,11 @@ export type NeuralwattReasoningEffort =
34
38
  | "max";
35
39
 
36
40
  /**
37
- * Per-model reasoning contract from `/v1/models`.
38
- *
39
- * `supported_efforts` is authoritative for which Pi thinking levels to expose:
40
- * the Pi map is built by identity (a level is enabled iff it appears here),
41
- * see `buildThinkingLevelMap` in `extensions/provider/models/build.ts`.
42
- * `default_effort` and `effort_aliases` are typed for fidelity but are not
43
- * consumed — Pi has no default-reasoning field and we expose native efforts
44
- * rather than aliasing unsupported ones.
41
+ * Per-model reasoning contract from `/v1/models`. The openai-completions
42
+ * thinking map uses `supported_efforts` by identity; the anthropic-messages
43
+ * map additionally resolves through `effort_aliases` (vLLM's effort enum
44
+ * accepts only native values). `default_enabled`/`default_effort` are typed
45
+ * for fidelity but not consumed.
45
46
  */
46
47
  export interface NeuralwattApiModelReasoning {
47
48
  /** Whether the model reasons by default. */
@@ -54,7 +55,10 @@ export interface NeuralwattApiModelReasoning {
54
55
  accepted_efforts?: NeuralwattReasoningEffort[];
55
56
  /** Server-side default. Not consumed; Pi has no default-reasoning field. */
56
57
  default_effort: NeuralwattReasoningEffort;
57
- /** Wire-level aliases from accepted to supported efforts. Not consumed. */
58
+ /**
59
+ * Wire-level aliases from accepted to supported efforts. Consumed by the
60
+ * anthropic-messages thinking level map; ignored by openai-completions.
61
+ */
58
62
  effort_aliases?: Partial<
59
63
  Record<NeuralwattReasoningEffort, NeuralwattReasoningEffort>
60
64
  >;
@@ -28,6 +28,10 @@ export interface NeuralwattQuotas {
28
28
  total_credits_usd: number;
29
29
  credits_used_usd: number;
30
30
  accounting_method: string;
31
+ /** Legacy credit pool split (USD). Typed for fidelity, not consumed yet. */
32
+ legacy_credits_usd?: number;
33
+ /** New credit pool split (USD). Typed for fidelity, not consumed yet. */
34
+ new_credits_usd?: number;
31
35
  };
32
36
  usage: {
33
37
  lifetime: {
@@ -1,41 +0,0 @@
1
- import type { Migration } from "@aliou/pi-utils-settings";
2
-
3
- /** Nested config shape before aliases were split out (pre-0.11.0). */
4
- interface PreAliasNeuralwattConfig {
5
- $schema?: string;
6
- provider?: {
7
- includeLegacyModelIds?: boolean;
8
- includeAliasedModelIds?: boolean;
9
- includeEarlyAccessModels?: boolean;
10
- };
11
- quotaCommand?: { enabled?: boolean };
12
- quotaWarnings?: { enabled?: boolean };
13
- subBarIntegration?: { enabled?: boolean };
14
- }
15
-
16
- /**
17
- * Creator-scoped active model IDs were split out of the legacy model ID setting.
18
- * Preserve behavior for users who had explicitly enabled legacy model IDs.
19
- */
20
- export const enableAliasesForLegacyUsersMigration: Migration<PreAliasNeuralwattConfig> =
21
- {
22
- name: "enable-alias-model-ids-for-legacy-users",
23
- version: "0.11.0",
24
- shouldRun: (config) =>
25
- config.provider?.includeLegacyModelIds === true &&
26
- config.provider?.includeAliasedModelIds === undefined,
27
- message:
28
- "[neuralwatt] active model aliases now use `provider.includeAliasedModelIds`; it was enabled because legacy model IDs were enabled.",
29
- run: (config) => {
30
- const provider = config.provider;
31
- if (!provider) return config;
32
-
33
- return {
34
- ...config,
35
- provider: {
36
- ...provider,
37
- includeAliasedModelIds: true,
38
- },
39
- };
40
- },
41
- };