@aliou/pi-neuralwatt 0.14.2 → 0.15.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,158 +0,0 @@
1
- import type { ProviderModelConfig } from "@earendil-works/pi-coding-agent";
2
- import { fetchNeuralwattModels } from "../../../src/lib/neuralwatt-api";
3
- import type { NeuralwattApiModel } from "../../../src/types/models-api";
4
- import { buildThinkingLevelMap, resolveMaxTokens } from "./build";
5
- import { NEURALWATT_MODELS } from "./public-models";
6
-
7
- // Pre-release models. Neuralwatt ships these to authorized accounts before they
8
- // reach the public /v1/models response; most go public eventually. Keep them
9
- // gated by includeEarlyAccessModels and hardcode entries so they remain
10
- // available from the offline catalog.
11
- // Move an entry to public-models.ts once Neuralwatt advertises it publicly.
12
- export const EARLY_ACCESS_NEURALWATT_MODELS: ProviderModelConfig[] = [
13
- // Qwen3.8 27B FP8 — pre-release dense 27B VL model with MTP speculative
14
- // decoding, native 262K context, served on 2x H200. Returned by the
15
- // authenticated /v1/models catalog but absent from the public endpoint.
16
- // Binary thinking is toggled through `chat_template_kwargs.enable_thinking`
17
- // (verified: `enable_thinking: false` suppresses reasoning output); the API
18
- // exposes no `reasoning` block, so the thinking level map is the high-only
19
- // fallback with `off: null`.
20
- // https://huggingface.co/Qwen/Qwen3.8-27B-FP8
21
- {
22
- id: "Qwen/Qwen3.8-27B-FP8",
23
- name: "Qwen 3.8 27B FP8",
24
- reasoning: true,
25
- input: ["text", "image"],
26
- cost: { input: 0.45, output: 3.2, cacheRead: 0.25, cacheWrite: 0 },
27
- contextWindow: 262_128,
28
- maxTokens: 65_536,
29
- compat: {
30
- supportsDeveloperRole: false,
31
- maxTokensField: "max_tokens",
32
- requiresReasoningContentOnAssistantMessages: true,
33
- thinkingFormat: "chat-template",
34
- chatTemplateKwargs: {
35
- enable_thinking: { $var: "thinking.enabled" },
36
- },
37
- },
38
- thinkingLevelMap: { ...buildThinkingLevelMap(undefined) },
39
- },
40
- ];
41
-
42
- // Per-ID overrides for known early-access models. The authenticated /v1/models
43
- // endpoint exposes pricing and capabilities, but some Pi-specific behavior
44
- // (compat flags, context window, max tokens) has to be supplied by hand.
45
- // Reasoning config is always derived from the endpoint's `reasoning` block via
46
- // `buildThinkingLevelMap`, so `reasoning` and `thinkingLevelMap` are not part
47
- // of the override shape.
48
- // Models that have since gone public now live in public-models.ts.
49
- const EARLY_ACCESS_MODEL_OVERRIDES: Partial<
50
- Record<
51
- string,
52
- Omit<Partial<ProviderModelConfig>, "reasoning" | "thinkingLevelMap">
53
- >
54
- > = {};
55
-
56
- function buildEarlyAccessModel(
57
- apiModel: NeuralwattApiModel,
58
- ): ProviderModelConfig {
59
- const meta = apiModel.metadata;
60
- const reasoning = meta?.capabilities.reasoning ?? false;
61
- const override = EARLY_ACCESS_MODEL_OVERRIDES[apiModel.id];
62
-
63
- const compat: NonNullable<ProviderModelConfig["compat"]> = {
64
- supportsDeveloperRole: false,
65
- maxTokensField: "max_tokens",
66
- };
67
- if (reasoning) {
68
- compat.requiresReasoningContentOnAssistantMessages = true;
69
- }
70
-
71
- const model: ProviderModelConfig = {
72
- id: apiModel.id,
73
- name: meta?.display_name ?? apiModel.id,
74
- reasoning,
75
- input: (meta?.capabilities.vision ? ["text", "image"] : ["text"]) as (
76
- | "text"
77
- | "image"
78
- )[],
79
- cost: {
80
- input: meta?.pricing.input_per_million ?? 0,
81
- output: meta?.pricing.output_per_million ?? 0,
82
- cacheRead: meta?.pricing.cached_input_per_million ?? 0,
83
- cacheWrite: meta?.pricing.cached_output_per_million ?? 0,
84
- },
85
- contextWindow: apiModel.max_model_len,
86
- maxTokens: resolveMaxTokens(
87
- meta?.limits.max_output_tokens,
88
- apiModel.max_model_len,
89
- ),
90
- compat,
91
- };
92
-
93
- if (reasoning) {
94
- // Reasoning levels come straight from the endpoint's `reasoning` block:
95
- // `supported_efforts` (identity) and `mandatory` (gates `off`). A missing
96
- // block falls back to a high-only map inside `buildThinkingLevelMap`.
97
- model.thinkingLevelMap = buildThinkingLevelMap(meta?.reasoning);
98
- }
99
-
100
- if (override) {
101
- return applyEarlyAccessOverride(model, override);
102
- }
103
-
104
- return model;
105
- }
106
-
107
- function applyEarlyAccessOverride(
108
- model: ProviderModelConfig,
109
- override: Partial<ProviderModelConfig>,
110
- ): ProviderModelConfig {
111
- const result: ProviderModelConfig = { ...model };
112
-
113
- if (override.name !== undefined) result.name = override.name;
114
- if (override.input !== undefined) result.input = override.input;
115
- if (override.contextWindow !== undefined) {
116
- result.contextWindow = override.contextWindow;
117
- }
118
- if (override.maxTokens !== undefined) result.maxTokens = override.maxTokens;
119
- if (override.cost !== undefined) {
120
- result.cost = { ...model.cost, ...override.cost };
121
- }
122
- if (override.compat !== undefined) {
123
- result.compat = { ...model.compat, ...override.compat };
124
- }
125
-
126
- // `reasoning` and `thinkingLevelMap` are intentionally not overridable:
127
- // reasoning config is derived from the endpoint's `reasoning` block.
128
-
129
- return result;
130
- }
131
-
132
- /**
133
- * Load early-access models from the authenticated /v1/models endpoint.
134
- *
135
- * Early-access models are any models returned by the API that are not already
136
- * part of the public hardcoded list. If the API key is missing or the request
137
- * fails, an `undefined` distinguishes an unavailable/failed request from a
138
- * successful empty list, allowing refresh callers to preserve stale cache.
139
- */
140
- export async function loadEarlyAccessModels(
141
- apiKey: string,
142
- signal?: AbortSignal,
143
- ): Promise<ProviderModelConfig[] | undefined> {
144
- if (!apiKey) return undefined;
145
-
146
- const result = await fetchNeuralwattModels(apiKey, signal);
147
- if (!result.success) return undefined;
148
-
149
- const publicIds = new Set(NEURALWATT_MODELS.map((model) => model.id));
150
-
151
- return result.data
152
- .filter(
153
- (model) =>
154
- !model.metadata?.deprecated && !model.metadata?.pricing.pricing_tbd,
155
- )
156
- .filter((model) => !publicIds.has(model.id))
157
- .map(buildEarlyAccessModel);
158
- }
@@ -1,47 +0,0 @@
1
- import type { ProviderModelConfig } from "@earendil-works/pi-coding-agent";
2
- import { NEURALWATT_MODELS } from "./public-models";
3
-
4
- // Legacy model IDs that should resolve to a canonical public model.
5
- // These are phased out over time and are only included when `includeLegacyModelIds` is enabled.
6
- export const LEGACY_MODEL_ALIAS_MAP = {
7
- // GLM 5.1 → GLM 5.2
8
- "glm-5.1": "glm-5.2",
9
- "glm-5.1-fast": "glm-5.2-fast",
10
- "zai-org/GLM-5.1-FP8": "glm-5.2",
11
- // Kimi K2.5 → Kimi K2.7 Code (K2.6 was retired 8/3, redirected to K2.7)
12
- "moonshotai/Kimi-K2.5": "kimi-k2.7-code",
13
- "kimi-k2.5-fast": "kimi-k2.7-code-fast",
14
- // Kimi K2.6 retired 8/3, redirected to Kimi K2.7 Code
15
- "kimi-k2.6": "kimi-k2.7-code",
16
- "kimi-k2.6-fast": "kimi-k2.7-code-fast",
17
- "kimi-k2.6-flex": "kimi-k2.7-code-flex",
18
- "moonshotai/Kimi-K2.6": "kimi-k2.7-code",
19
- // Qwen 3.5 retired 8/3, redirected to Qwen 3.6
20
- "qwen3.5-397b": "qwen3.6-35b",
21
- "qwen3.5-397b-fast": "qwen3.6-35b-fast",
22
- "Qwen/Qwen3.5-397B-A17B-FP8": "qwen3.6-35b",
23
- } as const;
24
-
25
- export const LEGACY_NEURALWATT_MODEL_IDS = new Set<string>(
26
- Object.keys(LEGACY_MODEL_ALIAS_MAP),
27
- );
28
-
29
- export function buildLegacyNeuralwattModels(): ProviderModelConfig[] {
30
- return Object.entries(LEGACY_MODEL_ALIAS_MAP).map(
31
- ([legacyId, canonicalId]) => {
32
- const canonical = NEURALWATT_MODELS.find(
33
- (model) => model.id === canonicalId,
34
- );
35
-
36
- if (!canonical) {
37
- throw new Error(`Missing canonical model for legacy alias ${legacyId}`);
38
- }
39
-
40
- return {
41
- ...canonical,
42
- id: legacyId,
43
- name: `${canonical.name} (legacy ID)`,
44
- };
45
- },
46
- );
47
- }