@aliou/pi-neuralwatt 0.12.1 → 0.13.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,4 +1,8 @@
1
1
  import type { ProviderModelConfig } from "@earendil-works/pi-coding-agent";
2
+ import type {
3
+ NeuralwattApiModelReasoning,
4
+ NeuralwattReasoningEffort,
5
+ } from "../../../src/types/models-api";
2
6
 
3
7
  export type ThinkingLevelMap = NonNullable<
4
8
  ProviderModelConfig["thinkingLevelMap"]
@@ -26,8 +30,11 @@ export interface NeuralwattCost {
26
30
  export interface NeuralwattModelFamily {
27
31
  cost: NeuralwattCost;
28
32
  vision: boolean;
29
- /** Thinking levels used by reasoning variants of this family. */
30
- thinkingLevelMap?: ThinkingLevelMap;
33
+ /**
34
+ * Reasoning contract snapshot from `/v1/models` for reasoning variants.
35
+ * `buildThinkingLevelMap` turns it into the Pi thinking level map.
36
+ */
37
+ reasoningMetadata?: NeuralwattReasoningMapSource;
31
38
  }
32
39
 
33
40
  export interface NeuralwattVariantSpec {
@@ -48,7 +55,54 @@ export interface NeuralwattVariantSpec {
48
55
  */
49
56
  costMultiplier?: number;
50
57
  vision?: boolean;
51
- thinkingLevelMap?: ThinkingLevelMap;
58
+ /** Override the family reasoning contract for this variant. */
59
+ reasoningMetadata?: NeuralwattReasoningMapSource;
60
+ }
61
+
62
+ /**
63
+ * Subset of the API reasoning block needed to build the Pi thinking level map.
64
+ * Kept narrow so public snapshots stay small and offline-friendly.
65
+ */
66
+ export type NeuralwattReasoningMapSource = Pick<
67
+ NeuralwattApiModelReasoning,
68
+ "supported_efforts" | "mandatory"
69
+ >;
70
+
71
+ /**
72
+ * Build the Pi thinking level map from the Neuralwatt reasoning contract.
73
+ *
74
+ * Pure identity mapping: a Pi level is enabled iff it appears in
75
+ * `supported_efforts` (mapped to its own name), `null` otherwise. `off` maps to
76
+ * `"none"` when the model permits disabling reasoning (`!mandatory` and
77
+ * `"none"` is supported).
78
+ *
79
+ * When the reasoning block is missing (e.g. Kimi K2.7 Code, whose API metadata
80
+ * exposes none), falls back to a conservative `high`-only map with `off: null`,
81
+ * matching the upstream binary thinking toggle.
82
+ *
83
+ * `default_effort` and `effort_aliases` are deliberately ignored: Pi has no
84
+ * default-reasoning field, and we expose native supported efforts rather than
85
+ * aliasing unsupported ones.
86
+ */
87
+ export function buildThinkingLevelMap(
88
+ reasoning: NeuralwattReasoningMapSource | undefined,
89
+ ): ThinkingLevelMap {
90
+ // Conservative fallback for models whose API metadata has no reasoning
91
+ // block. Expose one known-good level and forbid disabling reasoning.
92
+ const supported = new Set<NeuralwattReasoningEffort>(
93
+ reasoning?.supported_efforts ?? ["high"],
94
+ );
95
+ const mandatory = reasoning?.mandatory ?? true;
96
+
97
+ return {
98
+ off: !mandatory && supported.has("none") ? "none" : null,
99
+ minimal: supported.has("minimal") ? "minimal" : null,
100
+ low: supported.has("low") ? "low" : null,
101
+ medium: supported.has("medium") ? "medium" : null,
102
+ high: supported.has("high") ? "high" : null,
103
+ xhigh: supported.has("xhigh") ? "xhigh" : null,
104
+ max: supported.has("max") ? "max" : null,
105
+ };
52
106
  }
53
107
 
54
108
  /**
@@ -97,15 +151,14 @@ export function buildNeuralwattModel(
97
151
  };
98
152
 
99
153
  if (variant.reasoning) {
100
- const thinkingLevelMap =
101
- variant.thinkingLevelMap ?? family.thinkingLevelMap;
102
- if (!thinkingLevelMap) {
103
- throw new Error(
104
- `Missing thinkingLevelMap for reasoning model ${variant.id}`,
105
- );
106
- }
107
- // Clone so variants never share a family map instance.
108
- model.thinkingLevelMap = { ...thinkingLevelMap };
154
+ // Clone so variants never share a family map instance. The map is derived
155
+ // from the API reasoning contract; missing metadata falls back to a
156
+ // high-only map rather than throwing.
157
+ model.thinkingLevelMap = {
158
+ ...buildThinkingLevelMap(
159
+ variant.reasoningMetadata ?? family.reasoningMetadata,
160
+ ),
161
+ };
109
162
  }
110
163
 
111
164
  return model;
@@ -1,7 +1,7 @@
1
1
  import type { ProviderModelConfig } from "@earendil-works/pi-coding-agent";
2
2
  import { fetchNeuralwattModels } from "../../../src/lib/neuralwatt-api";
3
3
  import type { NeuralwattApiModel } from "../../../src/types/models-api";
4
- import { resolveMaxTokens } from "./build";
4
+ import { buildThinkingLevelMap, resolveMaxTokens } from "./build";
5
5
  import { NEURALWATT_MODELS } from "./public-models";
6
6
 
7
7
  // Pre-release models. Neuralwatt ships these to authorized accounts before they
@@ -13,10 +13,16 @@ export const EARLY_ACCESS_NEURALWATT_MODELS: ProviderModelConfig[] = [];
13
13
 
14
14
  // Per-ID overrides for known early-access models. The authenticated /v1/models
15
15
  // endpoint exposes pricing and capabilities, but some Pi-specific behavior
16
- // (thinking levels, compat flags) has to be supplied by hand.
16
+ // (compat flags, context window, max tokens) has to be supplied by hand.
17
+ // Reasoning config is always derived from the endpoint's `reasoning` block via
18
+ // `buildThinkingLevelMap`, so `reasoning` and `thinkingLevelMap` are not part
19
+ // of the override shape.
17
20
  // Models that have since gone public now live in public-models.ts.
18
21
  const EARLY_ACCESS_MODEL_OVERRIDES: Partial<
19
- Record<string, Partial<ProviderModelConfig>>
22
+ Record<
23
+ string,
24
+ Omit<Partial<ProviderModelConfig>, "reasoning" | "thinkingLevelMap">
25
+ >
20
26
  > = {};
21
27
 
22
28
  function buildEarlyAccessModel(
@@ -57,13 +63,10 @@ function buildEarlyAccessModel(
57
63
  };
58
64
 
59
65
  if (reasoning) {
60
- model.thinkingLevelMap = override?.thinkingLevelMap ?? {
61
- minimal: null,
62
- low: null,
63
- medium: "medium",
64
- high: null,
65
- xhigh: null,
66
- };
66
+ // Reasoning levels come straight from the endpoint's `reasoning` block:
67
+ // `supported_efforts` (identity) and `mandatory` (gates `off`). A missing
68
+ // block falls back to a high-only map inside `buildThinkingLevelMap`.
69
+ model.thinkingLevelMap = buildThinkingLevelMap(meta?.reasoning);
67
70
  }
68
71
 
69
72
  if (override) {
@@ -80,11 +83,7 @@ function applyEarlyAccessOverride(
80
83
  const result: ProviderModelConfig = { ...model };
81
84
 
82
85
  if (override.name !== undefined) result.name = override.name;
83
- if (override.reasoning !== undefined) result.reasoning = override.reasoning;
84
86
  if (override.input !== undefined) result.input = override.input;
85
- if (override.thinkingLevelMap !== undefined) {
86
- result.thinkingLevelMap = override.thinkingLevelMap;
87
- }
88
87
  if (override.contextWindow !== undefined) {
89
88
  result.contextWindow = override.contextWindow;
90
89
  }
@@ -96,6 +95,9 @@ function applyEarlyAccessOverride(
96
95
  result.compat = { ...model.compat, ...override.compat };
97
96
  }
98
97
 
98
+ // `reasoning` and `thinkingLevelMap` are intentionally not overridable:
99
+ // reasoning config is derived from the endpoint's `reasoning` block.
100
+
99
101
  return result;
100
102
  }
101
103
 
@@ -4,119 +4,83 @@ import {
4
4
  FLEX_COST_MULTIPLIER,
5
5
  type NeuralwattModelFamily,
6
6
  type NeuralwattVariantSpec,
7
- type ThinkingLevelMap,
8
7
  } from "./build";
9
8
 
10
9
  // Public models returned by https://api.neuralwatt.com/v1/models.
11
10
  // Pricing, capabilities, and limits are sourced from the API metadata fields;
12
11
  // `maxTokens` is `metadata.limits.max_output_tokens ?? max_model_len`.
13
12
  //
14
- // Models are declared per family so every variant (`-fast`, `-flex`, `-short`)
15
- // inherits the family's pricing, modalities, and thinking levels. See
16
- // `models.test.ts` for the drift check against the live catalog.
17
-
18
- // GLM natively supports `high` and `max` reasoning efforts. `xhigh` is an
19
- // unsupported hole between them. Pi added the `max` level in 0.80.6.
20
- const GLM_THINKING: ThinkingLevelMap = {
21
- off: "none",
22
- minimal: null,
23
- low: null,
24
- medium: null,
25
- high: "high",
26
- xhigh: null,
27
- max: "max",
28
- };
29
-
30
- // Binary thinking control (Kimi K2.x, Qwen3.x): no graded `reasoning_effort`
31
- // upstream, only a thinking on/off toggle. Expose a single known-good Pi
32
- // level; "high" stands in for standard full thinking.
33
- const BINARY_THINKING: ThinkingLevelMap = {
34
- minimal: null,
35
- low: null,
36
- medium: null,
37
- high: "high",
38
- xhigh: null,
39
- };
13
+ // Each reasoning family snapshots its `reasoning.supported_efforts` +
14
+ // `reasoning.mandatory` from the API; `buildThinkingLevelMap` turns that into
15
+ // the Pi thinking level map by identity (no aliasing). See `models.test.ts`
16
+ // for the drift check against the live catalog.
40
17
 
18
+ // DeepSeek V4 Flash: efforts max/high/none, not mandatory.
19
+ // https://api-docs.deepseek.com/guides/thinking_mode/
41
20
  const DEEPSEEK_V4_FLASH: NeuralwattModelFamily = {
42
21
  cost: { input: 0.14, output: 0.28, cacheRead: 0.028 },
43
22
  vision: false,
44
- // DeepSeek V4 Flash accepts reasoning_effort low/high/max (default high);
45
- // there is no "medium" tier, so Pi's low/high/max map directly and
46
- // minimal/medium/xhigh are unsupported holes.
47
- // https://api-docs.deepseek.com/guides/thinking_mode/
48
- thinkingLevelMap: {
49
- off: "none",
50
- minimal: null,
51
- low: "low",
52
- medium: null,
53
- high: "high",
54
- xhigh: null,
55
- max: "max",
23
+ reasoningMetadata: {
24
+ supported_efforts: ["max", "high", "none"],
25
+ mandatory: false,
56
26
  },
57
27
  };
58
28
 
59
29
  // Google, served from NVIDIA's NVFP4 checkpoint. Gemma 4's chat template
60
- // takes a boolean rather than an effort level, so it has a single reasoning
61
- // depth (`max`) plus thinking-off; every non-`none` value resolves to `max`.
30
+ // takes a boolean rather than an effort level, so the API only advertises
31
+ // `max` and `none`; every non-`none` request resolves to `max` upstream.
62
32
  // It does not reason by default (`default_enabled: false`), but the model
63
33
  // can produce reasoning traces when asked. See
64
34
  // https://portal.neuralwatt.com/docs/api/chat-completions#reasoning-effort
65
- const GEMMA_4_THINKING: ThinkingLevelMap = {
66
- off: "none",
67
- minimal: null,
68
- low: null,
69
- medium: null,
70
- high: null,
71
- xhigh: null,
72
- max: "max",
73
- };
74
-
75
35
  const GEMMA_4: NeuralwattModelFamily = {
76
36
  cost: { input: 0.144, output: 0.42, cacheRead: 0.0144 },
77
37
  vision: true,
78
- thinkingLevelMap: GEMMA_4_THINKING,
38
+ reasoningMetadata: {
39
+ supported_efforts: ["max", "none"],
40
+ mandatory: false,
41
+ },
79
42
  };
80
43
 
81
- // ZhipuAI.
44
+ // ZhipuAI. GLM-5.2 natively supports `high` and `max` reasoning efforts;
45
+ // `xhigh` is an unsupported hole between them. Pi's `max` level (0.80.6) maps
46
+ // to GLM's top tier.
82
47
  const GLM_5_2: NeuralwattModelFamily = {
83
48
  cost: { input: 1.45, output: 4.5, cacheRead: 0.145 },
84
49
  vision: false,
85
- thinkingLevelMap: GLM_THINKING,
50
+ reasoningMetadata: {
51
+ supported_efforts: ["max", "high", "none"],
52
+ mandatory: false,
53
+ },
86
54
  };
87
55
 
88
- // MoonshotAI. K3 is the largest open-weight model ever released, served in
89
- // preview with limited concurrency. K3 always reasons (thinking cannot be
90
- // disabled) and supports `reasoning_effort` values "low", "high", and "max"
91
- // (default "max"). There is no "medium" tier upstream, so Pi's low/high/max
92
- // map directly and `off`, `minimal`, `medium`, and `xhigh` are unsupported
93
- // holes. The `-fast` endpoint is a shorthand to set thinking to off.
56
+ // MoonshotAI. K3 supports reasoning efforts low/high/max (default max) and
57
+ // can be turned off (`mandatory: false`). The `-fast` endpoint is a shorthand
58
+ // to set thinking to off.
94
59
  const KIMI_K3: NeuralwattModelFamily = {
95
60
  cost: { input: 3, output: 15, cacheRead: 0.3 },
96
61
  vision: true,
97
- thinkingLevelMap: {
98
- off: null,
99
- minimal: null,
100
- low: "low",
101
- medium: null,
102
- high: "high",
103
- xhigh: null,
104
- max: "max",
62
+ reasoningMetadata: {
63
+ supported_efforts: ["max", "high", "low", "none"],
64
+ mandatory: false,
105
65
  },
106
66
  };
107
67
 
108
- // MoonshotAI.
68
+ // MoonshotAI. The K2.7 Code API exposes no `reasoning` block, so this family
69
+ // omits `reasoningMetadata`; `buildThinkingLevelMap` falls back to a high-only
70
+ // map with `off: null`, matching the upstream binary thinking toggle.
109
71
  const KIMI_K2_7_CODE: NeuralwattModelFamily = {
110
72
  cost: { input: 0.95, output: 4.0, cacheRead: 0.095 },
111
73
  vision: true,
112
- thinkingLevelMap: { off: null, ...BINARY_THINKING },
113
74
  };
114
75
 
115
- // Qwen.
76
+ // Qwen. Qwen3.6 35B only advertises `high` and `none`.
116
77
  const QWEN_3_6_35B: NeuralwattModelFamily = {
117
78
  cost: { input: 0.29, output: 1.15, cacheRead: 0.029 },
118
79
  vision: true,
119
- thinkingLevelMap: BINARY_THINKING,
80
+ reasoningMetadata: {
81
+ supported_efforts: ["high", "none"],
82
+ mandatory: false,
83
+ },
120
84
  };
121
85
 
122
86
  const FAMILIES: [NeuralwattModelFamily, NeuralwattVariantSpec[]][] = [
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@aliou/pi-neuralwatt",
3
- "version": "0.12.1",
3
+ "version": "0.13.0",
4
4
  "license": "MIT",
5
5
  "type": "module",
6
6
  "private": false,
@@ -18,6 +18,46 @@ export interface NeuralwattApiModelCapabilities {
18
18
  developer_role: boolean;
19
19
  }
20
20
 
21
+ /**
22
+ * Reasoning effort values Neuralwatt accepts on the wire. Mirrors Pi's
23
+ * `ModelThinkingLevel` (minus `off`, which the API spells `"none"`).
24
+ */
25
+ export type NeuralwattReasoningEffort =
26
+ | "none"
27
+ | "minimal"
28
+ | "low"
29
+ | "medium"
30
+ | "high"
31
+ | "xhigh"
32
+ | "max";
33
+
34
+ /**
35
+ * Per-model reasoning contract from `/v1/models`.
36
+ *
37
+ * `supported_efforts` is authoritative for which Pi thinking levels to expose:
38
+ * the Pi map is built by identity (a level is enabled iff it appears here),
39
+ * see `buildThinkingLevelMap` in `extensions/provider/models/build.ts`.
40
+ * `default_effort` and `effort_aliases` are typed for fidelity but are not
41
+ * consumed — Pi has no default-reasoning field and we expose native efforts
42
+ * rather than aliasing unsupported ones.
43
+ */
44
+ export interface NeuralwattApiModelReasoning {
45
+ /** Whether the model reasons by default. */
46
+ default_enabled: boolean;
47
+ /** Whether reasoning cannot be turned off. Forces `off: null` in the map. */
48
+ mandatory: boolean;
49
+ /** Efforts the model truly supports; drives the Pi thinking level map. */
50
+ supported_efforts: NeuralwattReasoningEffort[];
51
+ /** Efforts the API accepts but aliases onto a supported one. Not consumed. */
52
+ accepted_efforts?: NeuralwattReasoningEffort[];
53
+ /** Server-side default. Not consumed; Pi has no default-reasoning field. */
54
+ default_effort: NeuralwattReasoningEffort;
55
+ /** Wire-level aliases from accepted to supported efforts. Not consumed. */
56
+ effort_aliases?: Partial<
57
+ Record<NeuralwattReasoningEffort, NeuralwattReasoningEffort>
58
+ >;
59
+ }
60
+
21
61
  export interface NeuralwattApiModelLimits {
22
62
  max_context_length: number;
23
63
  max_output_tokens: number | null;
@@ -31,6 +71,7 @@ export interface NeuralwattApiModelMetadata {
31
71
  huggingface_id: string | null;
32
72
  pricing: NeuralwattApiModelPricing;
33
73
  capabilities: NeuralwattApiModelCapabilities;
74
+ reasoning?: NeuralwattApiModelReasoning;
34
75
  limits: NeuralwattApiModelLimits;
35
76
  deprecated: boolean;
36
77
  deprecated_message: string | null;