pi-lilac-provider 1.9.0 → 1.9.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +7 -27
- package/deprecated-models.json +1 -0
- package/index.ts +34 -4
- package/models.json +2 -2
- package/package.json +3 -2
- package/scripts/update-models.js +63 -0
package/README.md
CHANGED
|
@@ -74,7 +74,7 @@ pi
|
|
|
74
74
|
| Model | Context | Vision | Reasoning | Input $/M | Cache Read $/M | Output $/M |
|
|
75
75
|
|-------|---------|--------|-----------|-----------|-----------------|------------|
|
|
76
76
|
| Gemma 4 | 262K | ✅ | ✅ | $0.11 | — | $0.35 |
|
|
77
|
-
| GLM 5.2 | 524K | ❌ | ✅ | $0.90 | $0.
|
|
77
|
+
| GLM 5.2 | 524K | ❌ | ✅ | $0.90 | $0.17 | $3.00 |
|
|
78
78
|
| Kimi K2.6 | 262K | ✅ | ✅ | $0.70 | $0.20 | $3.50 |
|
|
79
79
|
| MiniMax M3 | 1.0M | ✅ | ✅ | $0.28 | $0.05 | $1.10 |
|
|
80
80
|
|
|
@@ -102,10 +102,7 @@ pi --provider lilac --model moonshotai/kimi-k2.6
|
|
|
102
102
|
|
|
103
103
|
### Thinking Mode
|
|
104
104
|
|
|
105
|
-
All Lilac models toggle reasoning via `chat_template_kwargs`, but the key each
|
|
106
|
-
model's chat template honors differs per family. The provider uses pi's
|
|
107
|
-
`chat-template` thinkingFormat with per-model `chatTemplateKwargs` (configured in
|
|
108
|
-
`patch.json`) so the right key reaches each template:
|
|
105
|
+
All Lilac models toggle reasoning via `chat_template_kwargs`, but the key each model's chat template honors differs per family. The provider uses pi's `chat-template` thinkingFormat with per-model `chatTemplateKwargs` (configured in `patch.json`) so the right key reaches each template:
|
|
109
106
|
|
|
110
107
|
| Model | Reasoning key | Default |
|
|
111
108
|
|-------|---------------|---------|
|
|
@@ -116,21 +113,9 @@ model's chat template honors differs per family. The provider uses pi's
|
|
|
116
113
|
| MiniMax M2.7 | `thinking` + `enable_thinking` (bool) | on |
|
|
117
114
|
| MiniMax M3 | `thinking_mode` (`disabled`\|`adaptive`\|`enabled`) | adaptive (server) |
|
|
118
115
|
|
|
119
|
-
Kimi K2.6, GLM 5.1, Gemma 4, and MiniMax M2.7 use the forward-compatible form
|
|
120
|
-
|
|
121
|
-
|
|
122
|
-
`reasoning_effort` (`high` = lower-latency, `max` = deepest). MiniMax M3 uses
|
|
123
|
-
the `thinking_mode` enum, exposed as three pi thinking levels: `off` →
|
|
124
|
-
`disabled` (never think), `minimal` → `adaptive` (the model decides), `high` →
|
|
125
|
-
`enabled` (always think). Pi starts at `off` (`disabled`); cycle to `minimal`
|
|
126
|
-
for M3's adaptive "model decides" mode. (The selector/footer show pi's level
|
|
127
|
-
names — `minimal`/`high` — not the `thinking_mode` values; pi has no per-model
|
|
128
|
-
level-relabel hook.)
|
|
129
|
-
|
|
130
|
-
**Preserved thinking (full-history reasoning).** By default these templates
|
|
131
|
-
trim older assistant reasoning between turns (each vendor's default), which
|
|
132
|
-
degrades multi-turn recall. Three models opt into full-history preservation via
|
|
133
|
-
a template flag sent alongside the reasoning key:
|
|
116
|
+
Kimi K2.6, GLM 5.1, Gemma 4, and MiniMax M2.7 use the forward-compatible form that sends **both** `thinking` and `enable_thinking`, so whichever key the template honors is set. GLM 5.2 additionally maps pi's thinking levels to `reasoning_effort` (`high` = lower-latency, `max` = deepest). MiniMax M3 uses the `thinking_mode` enum, exposed as three pi thinking levels: `off` → `disabled` (never think), `minimal` → `adaptive` (the model decides), `high` → `enabled` (always think). Pi starts at `off` (`disabled`); cycle to `minimal` for M3's adaptive "model decides" mode. (The selector/footer show pi's level names — `minimal`/`high` — not the `thinking_mode` values; pi has no per-model level-relabel hook.)
|
|
117
|
+
|
|
118
|
+
**Preserved thinking (full-history reasoning).** By default these templates trim older assistant reasoning between turns (each vendor's default), which degrades multi-turn recall. Three models opt into full-history preservation via a template flag sent alongside the reasoning key:
|
|
134
119
|
|
|
135
120
|
| Model | Flag | Effect |
|
|
136
121
|
|-------|------|--------|
|
|
@@ -138,14 +123,9 @@ a template flag sent alongside the reasoning key:
|
|
|
138
123
|
| GLM 5.1 | `clear_thinking: false` | keeps reasoning for all turns (default: clears before the last user message) |
|
|
139
124
|
| GLM 5.2 | `clear_thinking: false` | keeps reasoning for all turns (default: clears before the last user message) |
|
|
140
125
|
|
|
141
|
-
Kimi K2.6 and GLM 5.2 are E2E-verified on the sibling neuralwatt provider via a
|
|
142
|
-
3-turn, two-20-digit-number recall test (Kimi 0/6 → 6/6, GLM 5.2 1/4 → 4/4);
|
|
143
|
-
GLM 5.1 uses the same `clear_thinking` mechanism (confirmed in its HuggingFace
|
|
144
|
-
chat template). Gemma 4 and MiniMax M2.7/M3 expose no family-wide preserve flag,
|
|
145
|
-
so their older assistant reasoning is trimmed per the template default.
|
|
126
|
+
Kimi K2.6 and GLM 5.2 are E2E-verified on the sibling neuralwatt provider via a 3-turn, two-20-digit-number recall test (Kimi 0/6 → 6/6, GLM 5.2 1/4 → 4/4); GLM 5.1 uses the same `clear_thinking` mechanism (confirmed in its HuggingFace chat template). Gemma 4 and MiniMax M2.7/M3 expose no family-wide preserve flag, so their older assistant reasoning is trimmed per the template default.
|
|
146
127
|
|
|
147
|
-
In pi, reasoning models automatically use the appropriate thinking format. Use
|
|
148
|
-
Shift+Tab to control thinking level.
|
|
128
|
+
In pi, reasoning models automatically use the appropriate thinking format. Use Shift+Tab to control thinking level.
|
|
149
129
|
|
|
150
130
|
### Vision
|
|
151
131
|
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{}
|
package/index.ts
CHANGED
|
@@ -88,6 +88,7 @@ import { getAgentDir, type ExtensionAPI, type ModelRegistry } from "@earendil-wo
|
|
|
88
88
|
import modelsData from "./models.json" with { type: "json" };
|
|
89
89
|
import customModelsData from "./custom-models.json" with { type: "json" };
|
|
90
90
|
import patchData from "./patch.json" with { type: "json" };
|
|
91
|
+
import deprecatedData from "./deprecated-models.json" with { type: "json" };
|
|
91
92
|
import fs from "fs";
|
|
92
93
|
import path from "path";
|
|
93
94
|
|
|
@@ -128,7 +129,7 @@ interface JsonModel {
|
|
|
128
129
|
id: string;
|
|
129
130
|
name: string;
|
|
130
131
|
reasoning: boolean;
|
|
131
|
-
input:
|
|
132
|
+
input: ("text" | "image")[];
|
|
132
133
|
cost: {
|
|
133
134
|
input: number;
|
|
134
135
|
output: number;
|
|
@@ -163,7 +164,7 @@ interface JsonModel {
|
|
|
163
164
|
interface PatchEntry {
|
|
164
165
|
name?: string;
|
|
165
166
|
reasoning?: boolean;
|
|
166
|
-
input?:
|
|
167
|
+
input?: ("text" | "image")[];
|
|
167
168
|
cost?: {
|
|
168
169
|
input?: number;
|
|
169
170
|
output?: number;
|
|
@@ -379,6 +380,35 @@ function applyPatch(model: JsonModel, patch: PatchEntry): JsonModel {
|
|
|
379
380
|
}
|
|
380
381
|
|
|
381
382
|
/** Full pipeline: base models → patch → custom → user modelOverrides → result */
|
|
383
|
+
// Grace period for delisted models. When the provider API stops listing a
|
|
384
|
+
// model, update-models.js moves its last-known definition into
|
|
385
|
+
// deprecated-models.json (stamped with deprecatedAt) instead of dropping it.
|
|
386
|
+
// For 14 days the model keeps working here so in-flight sessions and saved
|
|
387
|
+
// model settings do not break; afterwards it is evicted permanently.
|
|
388
|
+
const DEPRECATED_MODEL_TTL_MS = 14 * 24 * 60 * 60 * 1000;
|
|
389
|
+
|
|
390
|
+
// Grace-period deprecated models with deprecation metadata stripped.
|
|
391
|
+
function activeDeprecatedModels(): JsonModel[] {
|
|
392
|
+
const now = Date.now();
|
|
393
|
+
const result: JsonModel[] = [];
|
|
394
|
+
for (const entry of Object.values(deprecatedData as Record<string, JsonModel & { deprecatedAt?: string }>)) {
|
|
395
|
+
if (!entry?.id) continue;
|
|
396
|
+
const removedAt = Date.parse(entry.deprecatedAt ?? "");
|
|
397
|
+
if (Number.isNaN(removedAt) || now - removedAt > DEPRECATED_MODEL_TTL_MS) continue;
|
|
398
|
+
const model = { ...entry } as JsonModel & { deprecatedAt?: string };
|
|
399
|
+
delete model.deprecatedAt;
|
|
400
|
+
result.push(model);
|
|
401
|
+
}
|
|
402
|
+
return result;
|
|
403
|
+
}
|
|
404
|
+
|
|
405
|
+
// Append grace-period deprecated models the list does not already have (live data wins).
|
|
406
|
+
function withDeprecated(models: JsonModel[]): JsonModel[] {
|
|
407
|
+
const seen = new Set(models.map((m) => m.id));
|
|
408
|
+
const extras = activeDeprecatedModels().filter((m) => !seen.has(m.id));
|
|
409
|
+
return extras.length > 0 ? [...models, ...extras] : models;
|
|
410
|
+
}
|
|
411
|
+
|
|
382
412
|
function buildModels(
|
|
383
413
|
base: JsonModel[],
|
|
384
414
|
custom: JsonModel[],
|
|
@@ -424,7 +454,7 @@ function buildModels(
|
|
|
424
454
|
}
|
|
425
455
|
}
|
|
426
456
|
|
|
427
|
-
return Array.from(modelMap.values());
|
|
457
|
+
return withDeprecated(Array.from(modelMap.values()));
|
|
428
458
|
}
|
|
429
459
|
|
|
430
460
|
// ─── Stale-While-Revalidate Model Sync ────────────────────────────────────────
|
|
@@ -450,7 +480,7 @@ function transformApiModel(apiModel: any): JsonModel | null {
|
|
|
450
480
|
// multiply and preserves sub-cent cache prices like 0.003 ($/M).
|
|
451
481
|
const toPerM = (v: any) => Math.round((typeof v === "string" ? parseFloat(v) : (v || 0)) * 1_000_000 * 1e6) / 1e6;
|
|
452
482
|
|
|
453
|
-
const inputTypes:
|
|
483
|
+
const inputTypes: ("text" | "image")[] = ["text"];
|
|
454
484
|
if (hasImage) inputTypes.push("image");
|
|
455
485
|
// Video is sent as image frames, so we don't add a separate "video" input type
|
|
456
486
|
|
package/models.json
CHANGED
|
@@ -33,7 +33,7 @@
|
|
|
33
33
|
"cost": {
|
|
34
34
|
"input": 0.9,
|
|
35
35
|
"output": 3,
|
|
36
|
-
"cacheRead": 0.
|
|
36
|
+
"cacheRead": 0.17,
|
|
37
37
|
"cacheWrite": 0
|
|
38
38
|
},
|
|
39
39
|
"contextWindow": 524288,
|
|
@@ -56,7 +56,7 @@
|
|
|
56
56
|
"cost": {
|
|
57
57
|
"input": 0.7,
|
|
58
58
|
"output": 3.5,
|
|
59
|
-
"cacheRead": 0.
|
|
59
|
+
"cacheRead": 0.16,
|
|
60
60
|
"cacheWrite": 0
|
|
61
61
|
},
|
|
62
62
|
"contextWindow": 262144,
|
package/package.json
CHANGED
|
@@ -1,12 +1,13 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "pi-lilac-provider",
|
|
3
|
-
"version": "1.9.
|
|
3
|
+
"version": "1.9.2",
|
|
4
4
|
"description": "Lilac provider extension for pi - Access Kimi K2.6, GLM 5.1, and Gemma 4 models through Lilac's OpenAI-compatible API on idle GPUs",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"main": "index.ts",
|
|
7
7
|
"files": [
|
|
8
8
|
"index.ts",
|
|
9
9
|
"models.json",
|
|
10
|
+
"deprecated-models.json",
|
|
10
11
|
"patch.json",
|
|
11
12
|
"custom-models.json",
|
|
12
13
|
"scripts"
|
|
@@ -29,7 +30,7 @@
|
|
|
29
30
|
"@earendil-works/pi-coding-agent": ">=0.74.0"
|
|
30
31
|
},
|
|
31
32
|
"devDependencies": {
|
|
32
|
-
"@earendil-works/pi-coding-agent": "0.
|
|
33
|
+
"@earendil-works/pi-coding-agent": "0.83.0",
|
|
33
34
|
"typescript": "6.0.3",
|
|
34
35
|
"@types/node": "25.9.1",
|
|
35
36
|
"knip": "6.14.1"
|
package/scripts/update-models.js
CHANGED
|
@@ -295,6 +295,67 @@ function updateReadme(models) {
|
|
|
295
295
|
|
|
296
296
|
// ─── Main ────────────────────────────────────────────────────────────────────
|
|
297
297
|
|
|
298
|
+
// Grace period for delisted models: update-models.js moves models the API no
|
|
299
|
+
// longer lists into deprecated-models.json (stamped with deprecatedAt) instead
|
|
300
|
+
// of dropping them; the runtime appends them back so sessions and saved model
|
|
301
|
+
// settings keep working, and after 14 days they are evicted permanently.
|
|
302
|
+
const DEPRECATED_MODEL_TTL_MS = 14 * 24 * 60 * 60 * 1000;
|
|
303
|
+
|
|
304
|
+
/**
|
|
305
|
+
* Reconcile deprecated-models.json against the freshly fetched model list.
|
|
306
|
+
* - in old models.json but not the API: moved into the deprecated file
|
|
307
|
+
* (deprecatedAt = now; preserved on repeat runs so the grace clock is not reset)
|
|
308
|
+
* - back in the API: resurrected (dropped from the deprecated file)
|
|
309
|
+
* - deprecatedAt older than 14 days: evicted permanently
|
|
310
|
+
* Must run BEFORE the new models.json is written; it reads the old file itself.
|
|
311
|
+
*/
|
|
312
|
+
function updateDeprecatedModels(modelsJsonPath, newModels) {
|
|
313
|
+
const deprecatedPath = path.join(path.dirname(modelsJsonPath), 'deprecated-models.json');
|
|
314
|
+
|
|
315
|
+
let oldModels = [];
|
|
316
|
+
try {
|
|
317
|
+
const parsed = JSON.parse(fs.readFileSync(modelsJsonPath, 'utf8'));
|
|
318
|
+
if (Array.isArray(parsed)) oldModels = parsed;
|
|
319
|
+
} catch { /* first run: no previous models.json */ }
|
|
320
|
+
|
|
321
|
+
let deprecated = {};
|
|
322
|
+
try {
|
|
323
|
+
const parsed = JSON.parse(fs.readFileSync(deprecatedPath, 'utf8'));
|
|
324
|
+
if (parsed && typeof parsed === 'object' && !Array.isArray(parsed)) deprecated = parsed;
|
|
325
|
+
} catch { /* no graveyard yet */ }
|
|
326
|
+
|
|
327
|
+
const currentIds = new Set(newModels.map(m => m.id));
|
|
328
|
+
const now = new Date().toISOString();
|
|
329
|
+
const added = [];
|
|
330
|
+
const resurrected = [];
|
|
331
|
+
const evicted = [];
|
|
332
|
+
|
|
333
|
+
for (const old of oldModels) {
|
|
334
|
+
if (old && old.id && !currentIds.has(old.id) && !deprecated[old.id]) {
|
|
335
|
+
deprecated[old.id] = { ...old, deprecatedAt: now };
|
|
336
|
+
added.push(old.id);
|
|
337
|
+
}
|
|
338
|
+
}
|
|
339
|
+
|
|
340
|
+
for (const [id, entry] of Object.entries(deprecated)) {
|
|
341
|
+
if (currentIds.has(id)) {
|
|
342
|
+
delete deprecated[id];
|
|
343
|
+
resurrected.push(id);
|
|
344
|
+
continue;
|
|
345
|
+
}
|
|
346
|
+
const removedAt = Date.parse(entry && entry.deprecatedAt ? entry.deprecatedAt : '');
|
|
347
|
+
if (Number.isNaN(removedAt) || Date.now() - removedAt > DEPRECATED_MODEL_TTL_MS) {
|
|
348
|
+
delete deprecated[id];
|
|
349
|
+
evicted.push(id);
|
|
350
|
+
}
|
|
351
|
+
}
|
|
352
|
+
|
|
353
|
+
if (added.length > 0 || resurrected.length > 0 || evicted.length > 0) {
|
|
354
|
+
fs.writeFileSync(deprecatedPath, JSON.stringify(deprecated, null, 2) + '\n');
|
|
355
|
+
console.log('Updated deprecated-models.json ' + JSON.stringify({ added, resurrected, evicted }));
|
|
356
|
+
}
|
|
357
|
+
}
|
|
358
|
+
|
|
298
359
|
async function main() {
|
|
299
360
|
try {
|
|
300
361
|
const apiModels = await fetchModels();
|
|
@@ -318,6 +379,8 @@ async function main() {
|
|
|
318
379
|
models.sort((a, b) => a.name.localeCompare(b.name));
|
|
319
380
|
|
|
320
381
|
// Save models.json (pure API output, no patch/custom baked in)
|
|
382
|
+
// Move delisted models to deprecated-models.json BEFORE models.json is overwritten
|
|
383
|
+
updateDeprecatedModels(MODELS_JSON_PATH, models);
|
|
321
384
|
saveJson(MODELS_JSON_PATH, models);
|
|
322
385
|
|
|
323
386
|
// Build full model list for README: base → patch → custom
|