@aliou/pi-neuralwatt 0.15.2 → 0.15.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -1
- package/extensions/provider/models/build.ts +4 -2
- package/extensions/provider/models/catalog.ts +9 -9
- package/extensions/provider/models/public-models.ts +8 -17
- package/extensions/provider/models/refresh.ts +55 -16
- package/extensions/provider/rate-limit-error.ts +1 -1
- package/package.json +1 -1
- package/src/types/models-api.ts +2 -0
package/README.md
CHANGED
|
@@ -144,5 +144,5 @@ This repository uses [Changesets](https://github.com/changesets/changesets) for
|
|
|
144
144
|
## Links
|
|
145
145
|
|
|
146
146
|
- [Neuralwatt](https://portal.neuralwatt.com/auth/register?ref=NW-ALIOU-Q7MF)
|
|
147
|
-
- [Neuralwatt API Docs](https://neuralwatt.com/
|
|
147
|
+
- [Neuralwatt API Docs](https://docs.neuralwatt.com/quickstart.md)
|
|
148
148
|
- [Pi Documentation](https://buildwithpi.ai/)
|
|
@@ -13,7 +13,7 @@ export type ThinkingLevelMap = NonNullable<
|
|
|
13
13
|
* streams. A non-streaming request to a `-flex` model silently falls back to
|
|
14
14
|
* the standard tier and the standard price.
|
|
15
15
|
*
|
|
16
|
-
* https://
|
|
16
|
+
* https://docs.neuralwatt.com/guides/flex-tier.md
|
|
17
17
|
*/
|
|
18
18
|
export const FLEX_COST_MULTIPLIER = 0.65;
|
|
19
19
|
|
|
@@ -107,12 +107,14 @@ export function buildThinkingLevelMap(
|
|
|
107
107
|
|
|
108
108
|
/**
|
|
109
109
|
* Neuralwatt reports `max_output_tokens: null` for models whose output is only
|
|
110
|
-
* bounded by the context window.
|
|
110
|
+
* bounded by the context window. Some models incorrectly report 0; treat 0
|
|
111
|
+
* like null so we never emit maxTokens: 0.
|
|
111
112
|
*/
|
|
112
113
|
export function resolveMaxTokens(
|
|
113
114
|
maxOutputTokens: number | null | undefined,
|
|
114
115
|
contextWindow: number,
|
|
115
116
|
): number {
|
|
117
|
+
if (maxOutputTokens === 0) return contextWindow;
|
|
116
118
|
return maxOutputTokens ?? contextWindow;
|
|
117
119
|
}
|
|
118
120
|
|
|
@@ -10,12 +10,6 @@ import { NEURALWATT_MODELS } from "./public-models";
|
|
|
10
10
|
|
|
11
11
|
export type NeuralwattModel = ProviderModelConfig;
|
|
12
12
|
|
|
13
|
-
const CONTEXT_WINDOW_OVERRIDES: ReadonlyMap<string, number> = new Map([
|
|
14
|
-
["kimi-k3", 327_680],
|
|
15
|
-
["kimi-k3-fast", 327_680],
|
|
16
|
-
["kimi-k3-flex", 327_680],
|
|
17
|
-
]);
|
|
18
|
-
|
|
19
13
|
// Chat-template thinking: the API exposes a `reasoning` block, but the
|
|
20
14
|
// underlying mechanism is chat_template_kwargs, so Pi needs the mapping.
|
|
21
15
|
const COMPAT_OVERRIDES: Partial<
|
|
@@ -62,8 +56,7 @@ function apiModelToProviderModel(model: NeuralwattApiModel): NeuralwattModel {
|
|
|
62
56
|
if (reasoning) compat.requiresReasoningContentOnAssistantMessages = true;
|
|
63
57
|
Object.assign(compat, COMPAT_OVERRIDES[model.id]);
|
|
64
58
|
|
|
65
|
-
const contextWindow =
|
|
66
|
-
CONTEXT_WINDOW_OVERRIDES.get(model.id) ?? model.max_model_len;
|
|
59
|
+
const contextWindow = model.max_model_len;
|
|
67
60
|
|
|
68
61
|
const result: NeuralwattModel = {
|
|
69
62
|
id: model.id,
|
|
@@ -135,7 +128,14 @@ export function buildNeuralwattProviderModelsFromApi(
|
|
|
135
128
|
const models = apiModels
|
|
136
129
|
.filter(
|
|
137
130
|
(m) =>
|
|
138
|
-
m.metadata &&
|
|
131
|
+
m.metadata &&
|
|
132
|
+
!m.metadata.deprecated &&
|
|
133
|
+
!m.metadata.pricing.pricing_tbd &&
|
|
134
|
+
// Exclude non-chat models (e.g. embeddings) by task
|
|
135
|
+
!(
|
|
136
|
+
m.metadata.capabilities.task &&
|
|
137
|
+
!["chat", "completions"].includes(m.metadata.capabilities.task)
|
|
138
|
+
),
|
|
139
139
|
)
|
|
140
140
|
.map(apiModelToProviderModel);
|
|
141
141
|
return [...models, ...buildAliases(models, apiModels)];
|
|
@@ -31,7 +31,7 @@ const DEEPSEEK_V4_FLASH: NeuralwattModelFamily = {
|
|
|
31
31
|
// `max` and `none`; every non-`none` request resolves to `max` upstream.
|
|
32
32
|
// It does not reason by default (`default_enabled: false`), but the model
|
|
33
33
|
// can produce reasoning traces when asked. See
|
|
34
|
-
// https://
|
|
34
|
+
// https://docs.neuralwatt.com/api/chat-completions.md
|
|
35
35
|
const GEMMA_4: NeuralwattModelFamily = {
|
|
36
36
|
cost: { input: 0.144, output: 0.42, cacheRead: 0.0144 },
|
|
37
37
|
vision: true,
|
|
@@ -208,37 +208,28 @@ const FAMILIES: [NeuralwattModelFamily, NeuralwattVariantSpec[]][] = [
|
|
|
208
208
|
},
|
|
209
209
|
],
|
|
210
210
|
],
|
|
211
|
-
// The kimi-k3 endpoint rejects anything above 327,680 total tokens with
|
|
212
|
-
// `400: max_completion_tokens is too large … supports at most 327680
|
|
213
|
-
// completion tokens` (verified at runtime), even though the API advertises
|
|
214
|
-
// `max_model_len: 1048560` with a null output cap for the whole family.
|
|
215
|
-
// The -fast/-flex endpoints don't enforce any cap server-side yet (they
|
|
216
|
-
// accept max_completion_tokens beyond the advertised window), but they are
|
|
217
|
-
// the same K3 deployment and are expected to share the 327,680 limit, so
|
|
218
|
-
// all three variants are pinned to it. The drift check in models.test.ts
|
|
219
|
-
// whitelists this divergence via CONTEXT_WINDOW_OVERRIDES.
|
|
220
211
|
[
|
|
221
212
|
KIMI_K3,
|
|
222
213
|
[
|
|
223
214
|
{
|
|
224
215
|
id: "kimi-k3",
|
|
225
216
|
name: "Kimi K3",
|
|
226
|
-
contextWindow:
|
|
227
|
-
maxOutputTokens:
|
|
217
|
+
contextWindow: 1048560,
|
|
218
|
+
maxOutputTokens: null,
|
|
228
219
|
reasoning: true,
|
|
229
220
|
},
|
|
230
221
|
{
|
|
231
222
|
id: "kimi-k3-fast",
|
|
232
223
|
name: "Kimi K3 Fast",
|
|
233
|
-
contextWindow:
|
|
234
|
-
maxOutputTokens:
|
|
224
|
+
contextWindow: 1048560,
|
|
225
|
+
maxOutputTokens: null,
|
|
235
226
|
reasoning: false,
|
|
236
227
|
},
|
|
237
228
|
{
|
|
238
229
|
id: "kimi-k3-flex",
|
|
239
230
|
name: "Kimi K3 (flex)",
|
|
240
|
-
contextWindow:
|
|
241
|
-
maxOutputTokens:
|
|
231
|
+
contextWindow: 1048560,
|
|
232
|
+
maxOutputTokens: null,
|
|
242
233
|
reasoning: true,
|
|
243
234
|
costMultiplier: FLEX_COST_MULTIPLIER,
|
|
244
235
|
},
|
|
@@ -299,7 +290,7 @@ const FAMILIES: [NeuralwattModelFamily, NeuralwattVariantSpec[]][] = [
|
|
|
299
290
|
// The API now advertises flex variants but lists them at standard pricing;
|
|
300
291
|
// the 35% Flex discount is a billing-time concept applied here via
|
|
301
292
|
// `costMultiplier` rather than reflected in the catalog metadata.
|
|
302
|
-
// https://
|
|
293
|
+
// https://docs.neuralwatt.com/guides/flex-tier.md
|
|
303
294
|
|
|
304
295
|
export const NEURALWATT_MODELS: ProviderModelConfig[] = FAMILIES.flatMap(
|
|
305
296
|
([family, variants]) => buildNeuralwattFamily(family, variants),
|
|
@@ -12,6 +12,38 @@ import type {
|
|
|
12
12
|
|
|
13
13
|
export const MODEL_STORE_TTL_MS = 4 * 60 * 60 * 1000;
|
|
14
14
|
|
|
15
|
+
/**
|
|
16
|
+
* Scope a store entry applies to. The public catalog is a subset of any
|
|
17
|
+
* key-scoped catalog (preview, grant-gated, private models), so an entry
|
|
18
|
+
* stamped "public" must not shadow a keyed refresh — and a key-scoped entry
|
|
19
|
+
* must not be replayed for an anonymous user. Matches the anonymous-key
|
|
20
|
+
* convention in src/lib/neuralwatt-api.ts (authHeaders).
|
|
21
|
+
*/
|
|
22
|
+
const CATALOG_SCOPE_VERSION = "v1";
|
|
23
|
+
type CatalogScope = "public" | "key";
|
|
24
|
+
|
|
25
|
+
type ScopedModelsStoreEntry = ModelsStoreEntry & { catalogKey?: string };
|
|
26
|
+
|
|
27
|
+
function catalogScope(apiKey: string | undefined): CatalogScope {
|
|
28
|
+
return apiKey !== undefined && apiKey !== "" && apiKey !== "-"
|
|
29
|
+
? "key"
|
|
30
|
+
: "public";
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
function storedCatalogKey(entry: ScopedModelsStoreEntry): string | undefined {
|
|
34
|
+
return entry.catalogKey;
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
function catalogKeyMatches(
|
|
38
|
+
entry: ScopedModelsStoreEntry | undefined,
|
|
39
|
+
scope: CatalogScope,
|
|
40
|
+
): boolean {
|
|
41
|
+
return (
|
|
42
|
+
entry !== undefined &&
|
|
43
|
+
storedCatalogKey(entry) === `${scope} ${CATALOG_SCOPE_VERSION}`
|
|
44
|
+
);
|
|
45
|
+
}
|
|
46
|
+
|
|
15
47
|
export type FetchNeuralwattApiModels = (
|
|
16
48
|
apiKey: string | undefined,
|
|
17
49
|
signal?: AbortSignal,
|
|
@@ -25,6 +57,13 @@ function isFreshStoreEntry(
|
|
|
25
57
|
return Date.now() - checkedAt < MODEL_STORE_TTL_MS;
|
|
26
58
|
}
|
|
27
59
|
|
|
60
|
+
function isUsableStoreEntry(
|
|
61
|
+
entry: Readonly<ModelsStoreEntry> | undefined,
|
|
62
|
+
scope: CatalogScope,
|
|
63
|
+
): entry is ModelsStoreEntry {
|
|
64
|
+
return isFreshStoreEntry(entry) && catalogKeyMatches(entry, scope);
|
|
65
|
+
}
|
|
66
|
+
|
|
28
67
|
export function createNeuralwattRefreshModels(
|
|
29
68
|
staticModels: ReturnType<typeof buildNeuralwattProviderModels>,
|
|
30
69
|
fetchApiModels: FetchNeuralwattApiModels,
|
|
@@ -35,29 +74,29 @@ export function createNeuralwattRefreshModels(
|
|
|
35
74
|
context.signal.throwIfAborted();
|
|
36
75
|
const fallback = buildFromStore(staticModels);
|
|
37
76
|
try {
|
|
38
|
-
if (!context.allowNetwork) {
|
|
39
|
-
return context.stored
|
|
40
|
-
? buildFromStore(context.stored.models)
|
|
41
|
-
: fallback;
|
|
42
|
-
}
|
|
43
|
-
if (!context.force && isFreshStoreEntry(context.stored)) {
|
|
44
|
-
return buildFromStore(context.stored.models);
|
|
45
|
-
}
|
|
46
77
|
const apiKey =
|
|
47
78
|
context.credential?.type === "api_key"
|
|
48
79
|
? context.credential.key
|
|
49
80
|
: undefined;
|
|
81
|
+
const scope = catalogScope(apiKey);
|
|
82
|
+
const stored = context.stored as ScopedModelsStoreEntry | undefined;
|
|
83
|
+
if (!context.allowNetwork) {
|
|
84
|
+
return stored !== undefined && catalogKeyMatches(stored, scope)
|
|
85
|
+
? buildFromStore(stored.models)
|
|
86
|
+
: fallback;
|
|
87
|
+
}
|
|
88
|
+
if (!context.force && isUsableStoreEntry(stored, scope)) {
|
|
89
|
+
return buildFromStore(stored.models);
|
|
90
|
+
}
|
|
50
91
|
const apiModels = await fetchApiModels(apiKey, context.signal);
|
|
51
92
|
context.signal.throwIfAborted();
|
|
52
93
|
const models = buildFromApi(apiModels);
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
})
|
|
60
|
-
.catch(() => undefined);
|
|
94
|
+
const entry: ScopedModelsStoreEntry = {
|
|
95
|
+
models: models as unknown as ModelsStoreEntry["models"],
|
|
96
|
+
checkedAt: Date.now(),
|
|
97
|
+
catalogKey: `${scope} ${CATALOG_SCOPE_VERSION}`,
|
|
98
|
+
};
|
|
99
|
+
await context.publish({ persist: entry }).catch(() => undefined);
|
|
61
100
|
context.signal.throwIfAborted();
|
|
62
101
|
return models;
|
|
63
102
|
} catch (error) {
|
|
@@ -12,7 +12,7 @@ interface AssistantErrorLike {
|
|
|
12
12
|
* layer. Each sets unique headers so the client can tell which layer
|
|
13
13
|
* triggered the rejection.
|
|
14
14
|
*
|
|
15
|
-
* @see https://
|
|
15
|
+
* @see https://docs.neuralwatt.com/guides/rate-limits.md
|
|
16
16
|
*/
|
|
17
17
|
export interface NeuralwattRateLimitInfo {
|
|
18
18
|
/** Which rate-limit layer triggered the 429 */
|
package/package.json
CHANGED