@aliou/pi-neuralwatt 0.14.2 → 0.15.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/extensions/provider/commands/settings/index.ts +0 -73
- package/extensions/provider/index.ts +13 -35
- package/extensions/provider/models/catalog.ts +148 -0
- package/extensions/provider/models/index.ts +10 -37
- package/extensions/provider/models/refresh.ts +65 -183
- package/extensions/provider/provider.ts +19 -26
- package/package.json +1 -1
- package/schema.json +0 -22
- package/src/config/defaults.ts +0 -5
- package/src/config/loader.ts +0 -11
- package/src/config/migration/01-disable-legacy-model-ids-by-default.ts +15 -28
- package/src/config/migration/02-flat-to-nested-config.ts +35 -71
- package/src/config/migration/03-rename-hidden-to-early-access.ts +33 -38
- package/src/config/migration/04-enable-aliases-for-legacy-users.ts +16 -25
- package/src/config/migration/index.ts +5 -2
- package/src/config/types.ts +0 -19
- package/src/lib/neuralwatt-api.ts +7 -7
- package/extensions/provider/models/aliases.ts +0 -35
- package/extensions/provider/models/early-access.ts +0 -158
- package/extensions/provider/models/legacy.ts +0 -47
|
@@ -1,158 +0,0 @@
|
|
|
1
|
-
import type { ProviderModelConfig } from "@earendil-works/pi-coding-agent";
|
|
2
|
-
import { fetchNeuralwattModels } from "../../../src/lib/neuralwatt-api";
|
|
3
|
-
import type { NeuralwattApiModel } from "../../../src/types/models-api";
|
|
4
|
-
import { buildThinkingLevelMap, resolveMaxTokens } from "./build";
|
|
5
|
-
import { NEURALWATT_MODELS } from "./public-models";
|
|
6
|
-
|
|
7
|
-
// Pre-release models. Neuralwatt ships these to authorized accounts before they
|
|
8
|
-
// reach the public /v1/models response; most go public eventually. Keep them
|
|
9
|
-
// gated by includeEarlyAccessModels and hardcode entries so they remain
|
|
10
|
-
// available from the offline catalog.
|
|
11
|
-
// Move an entry to public-models.ts once Neuralwatt advertises it publicly.
|
|
12
|
-
export const EARLY_ACCESS_NEURALWATT_MODELS: ProviderModelConfig[] = [
|
|
13
|
-
// Qwen3.8 27B FP8 — pre-release dense 27B VL model with MTP speculative
|
|
14
|
-
// decoding, native 262K context, served on 2x H200. Returned by the
|
|
15
|
-
// authenticated /v1/models catalog but absent from the public endpoint.
|
|
16
|
-
// Binary thinking is toggled through `chat_template_kwargs.enable_thinking`
|
|
17
|
-
// (verified: `enable_thinking: false` suppresses reasoning output); the API
|
|
18
|
-
// exposes no `reasoning` block, so the thinking level map is the high-only
|
|
19
|
-
// fallback with `off: null`.
|
|
20
|
-
// https://huggingface.co/Qwen/Qwen3.8-27B-FP8
|
|
21
|
-
{
|
|
22
|
-
id: "Qwen/Qwen3.8-27B-FP8",
|
|
23
|
-
name: "Qwen 3.8 27B FP8",
|
|
24
|
-
reasoning: true,
|
|
25
|
-
input: ["text", "image"],
|
|
26
|
-
cost: { input: 0.45, output: 3.2, cacheRead: 0.25, cacheWrite: 0 },
|
|
27
|
-
contextWindow: 262_128,
|
|
28
|
-
maxTokens: 65_536,
|
|
29
|
-
compat: {
|
|
30
|
-
supportsDeveloperRole: false,
|
|
31
|
-
maxTokensField: "max_tokens",
|
|
32
|
-
requiresReasoningContentOnAssistantMessages: true,
|
|
33
|
-
thinkingFormat: "chat-template",
|
|
34
|
-
chatTemplateKwargs: {
|
|
35
|
-
enable_thinking: { $var: "thinking.enabled" },
|
|
36
|
-
},
|
|
37
|
-
},
|
|
38
|
-
thinkingLevelMap: { ...buildThinkingLevelMap(undefined) },
|
|
39
|
-
},
|
|
40
|
-
];
|
|
41
|
-
|
|
42
|
-
// Per-ID overrides for known early-access models. The authenticated /v1/models
|
|
43
|
-
// endpoint exposes pricing and capabilities, but some Pi-specific behavior
|
|
44
|
-
// (compat flags, context window, max tokens) has to be supplied by hand.
|
|
45
|
-
// Reasoning config is always derived from the endpoint's `reasoning` block via
|
|
46
|
-
// `buildThinkingLevelMap`, so `reasoning` and `thinkingLevelMap` are not part
|
|
47
|
-
// of the override shape.
|
|
48
|
-
// Models that have since gone public now live in public-models.ts.
|
|
49
|
-
const EARLY_ACCESS_MODEL_OVERRIDES: Partial<
|
|
50
|
-
Record<
|
|
51
|
-
string,
|
|
52
|
-
Omit<Partial<ProviderModelConfig>, "reasoning" | "thinkingLevelMap">
|
|
53
|
-
>
|
|
54
|
-
> = {};
|
|
55
|
-
|
|
56
|
-
function buildEarlyAccessModel(
|
|
57
|
-
apiModel: NeuralwattApiModel,
|
|
58
|
-
): ProviderModelConfig {
|
|
59
|
-
const meta = apiModel.metadata;
|
|
60
|
-
const reasoning = meta?.capabilities.reasoning ?? false;
|
|
61
|
-
const override = EARLY_ACCESS_MODEL_OVERRIDES[apiModel.id];
|
|
62
|
-
|
|
63
|
-
const compat: NonNullable<ProviderModelConfig["compat"]> = {
|
|
64
|
-
supportsDeveloperRole: false,
|
|
65
|
-
maxTokensField: "max_tokens",
|
|
66
|
-
};
|
|
67
|
-
if (reasoning) {
|
|
68
|
-
compat.requiresReasoningContentOnAssistantMessages = true;
|
|
69
|
-
}
|
|
70
|
-
|
|
71
|
-
const model: ProviderModelConfig = {
|
|
72
|
-
id: apiModel.id,
|
|
73
|
-
name: meta?.display_name ?? apiModel.id,
|
|
74
|
-
reasoning,
|
|
75
|
-
input: (meta?.capabilities.vision ? ["text", "image"] : ["text"]) as (
|
|
76
|
-
| "text"
|
|
77
|
-
| "image"
|
|
78
|
-
)[],
|
|
79
|
-
cost: {
|
|
80
|
-
input: meta?.pricing.input_per_million ?? 0,
|
|
81
|
-
output: meta?.pricing.output_per_million ?? 0,
|
|
82
|
-
cacheRead: meta?.pricing.cached_input_per_million ?? 0,
|
|
83
|
-
cacheWrite: meta?.pricing.cached_output_per_million ?? 0,
|
|
84
|
-
},
|
|
85
|
-
contextWindow: apiModel.max_model_len,
|
|
86
|
-
maxTokens: resolveMaxTokens(
|
|
87
|
-
meta?.limits.max_output_tokens,
|
|
88
|
-
apiModel.max_model_len,
|
|
89
|
-
),
|
|
90
|
-
compat,
|
|
91
|
-
};
|
|
92
|
-
|
|
93
|
-
if (reasoning) {
|
|
94
|
-
// Reasoning levels come straight from the endpoint's `reasoning` block:
|
|
95
|
-
// `supported_efforts` (identity) and `mandatory` (gates `off`). A missing
|
|
96
|
-
// block falls back to a high-only map inside `buildThinkingLevelMap`.
|
|
97
|
-
model.thinkingLevelMap = buildThinkingLevelMap(meta?.reasoning);
|
|
98
|
-
}
|
|
99
|
-
|
|
100
|
-
if (override) {
|
|
101
|
-
return applyEarlyAccessOverride(model, override);
|
|
102
|
-
}
|
|
103
|
-
|
|
104
|
-
return model;
|
|
105
|
-
}
|
|
106
|
-
|
|
107
|
-
function applyEarlyAccessOverride(
|
|
108
|
-
model: ProviderModelConfig,
|
|
109
|
-
override: Partial<ProviderModelConfig>,
|
|
110
|
-
): ProviderModelConfig {
|
|
111
|
-
const result: ProviderModelConfig = { ...model };
|
|
112
|
-
|
|
113
|
-
if (override.name !== undefined) result.name = override.name;
|
|
114
|
-
if (override.input !== undefined) result.input = override.input;
|
|
115
|
-
if (override.contextWindow !== undefined) {
|
|
116
|
-
result.contextWindow = override.contextWindow;
|
|
117
|
-
}
|
|
118
|
-
if (override.maxTokens !== undefined) result.maxTokens = override.maxTokens;
|
|
119
|
-
if (override.cost !== undefined) {
|
|
120
|
-
result.cost = { ...model.cost, ...override.cost };
|
|
121
|
-
}
|
|
122
|
-
if (override.compat !== undefined) {
|
|
123
|
-
result.compat = { ...model.compat, ...override.compat };
|
|
124
|
-
}
|
|
125
|
-
|
|
126
|
-
// `reasoning` and `thinkingLevelMap` are intentionally not overridable:
|
|
127
|
-
// reasoning config is derived from the endpoint's `reasoning` block.
|
|
128
|
-
|
|
129
|
-
return result;
|
|
130
|
-
}
|
|
131
|
-
|
|
132
|
-
/**
|
|
133
|
-
* Load early-access models from the authenticated /v1/models endpoint.
|
|
134
|
-
*
|
|
135
|
-
* Early-access models are any models returned by the API that are not already
|
|
136
|
-
* part of the public hardcoded list. If the API key is missing or the request
|
|
137
|
-
* fails, an `undefined` distinguishes an unavailable/failed request from a
|
|
138
|
-
* successful empty list, allowing refresh callers to preserve stale cache.
|
|
139
|
-
*/
|
|
140
|
-
export async function loadEarlyAccessModels(
|
|
141
|
-
apiKey: string,
|
|
142
|
-
signal?: AbortSignal,
|
|
143
|
-
): Promise<ProviderModelConfig[] | undefined> {
|
|
144
|
-
if (!apiKey) return undefined;
|
|
145
|
-
|
|
146
|
-
const result = await fetchNeuralwattModels(apiKey, signal);
|
|
147
|
-
if (!result.success) return undefined;
|
|
148
|
-
|
|
149
|
-
const publicIds = new Set(NEURALWATT_MODELS.map((model) => model.id));
|
|
150
|
-
|
|
151
|
-
return result.data
|
|
152
|
-
.filter(
|
|
153
|
-
(model) =>
|
|
154
|
-
!model.metadata?.deprecated && !model.metadata?.pricing.pricing_tbd,
|
|
155
|
-
)
|
|
156
|
-
.filter((model) => !publicIds.has(model.id))
|
|
157
|
-
.map(buildEarlyAccessModel);
|
|
158
|
-
}
|
|
@@ -1,47 +0,0 @@
|
|
|
1
|
-
import type { ProviderModelConfig } from "@earendil-works/pi-coding-agent";
|
|
2
|
-
import { NEURALWATT_MODELS } from "./public-models";
|
|
3
|
-
|
|
4
|
-
// Legacy model IDs that should resolve to a canonical public model.
|
|
5
|
-
// These are phased out over time and are only included when `includeLegacyModelIds` is enabled.
|
|
6
|
-
export const LEGACY_MODEL_ALIAS_MAP = {
|
|
7
|
-
// GLM 5.1 → GLM 5.2
|
|
8
|
-
"glm-5.1": "glm-5.2",
|
|
9
|
-
"glm-5.1-fast": "glm-5.2-fast",
|
|
10
|
-
"zai-org/GLM-5.1-FP8": "glm-5.2",
|
|
11
|
-
// Kimi K2.5 → Kimi K2.7 Code (K2.6 was retired 8/3, redirected to K2.7)
|
|
12
|
-
"moonshotai/Kimi-K2.5": "kimi-k2.7-code",
|
|
13
|
-
"kimi-k2.5-fast": "kimi-k2.7-code-fast",
|
|
14
|
-
// Kimi K2.6 retired 8/3, redirected to Kimi K2.7 Code
|
|
15
|
-
"kimi-k2.6": "kimi-k2.7-code",
|
|
16
|
-
"kimi-k2.6-fast": "kimi-k2.7-code-fast",
|
|
17
|
-
"kimi-k2.6-flex": "kimi-k2.7-code-flex",
|
|
18
|
-
"moonshotai/Kimi-K2.6": "kimi-k2.7-code",
|
|
19
|
-
// Qwen 3.5 retired 8/3, redirected to Qwen 3.6
|
|
20
|
-
"qwen3.5-397b": "qwen3.6-35b",
|
|
21
|
-
"qwen3.5-397b-fast": "qwen3.6-35b-fast",
|
|
22
|
-
"Qwen/Qwen3.5-397B-A17B-FP8": "qwen3.6-35b",
|
|
23
|
-
} as const;
|
|
24
|
-
|
|
25
|
-
export const LEGACY_NEURALWATT_MODEL_IDS = new Set<string>(
|
|
26
|
-
Object.keys(LEGACY_MODEL_ALIAS_MAP),
|
|
27
|
-
);
|
|
28
|
-
|
|
29
|
-
export function buildLegacyNeuralwattModels(): ProviderModelConfig[] {
|
|
30
|
-
return Object.entries(LEGACY_MODEL_ALIAS_MAP).map(
|
|
31
|
-
([legacyId, canonicalId]) => {
|
|
32
|
-
const canonical = NEURALWATT_MODELS.find(
|
|
33
|
-
(model) => model.id === canonicalId,
|
|
34
|
-
);
|
|
35
|
-
|
|
36
|
-
if (!canonical) {
|
|
37
|
-
throw new Error(`Missing canonical model for legacy alias ${legacyId}`);
|
|
38
|
-
}
|
|
39
|
-
|
|
40
|
-
return {
|
|
41
|
-
...canonical,
|
|
42
|
-
id: legacyId,
|
|
43
|
-
name: `${canonical.name} (legacy ID)`,
|
|
44
|
-
};
|
|
45
|
-
},
|
|
46
|
-
);
|
|
47
|
-
}
|