pi-lilac-provider 1.8.2 → 1.9.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +7 -27
- package/index.ts +54 -5
- package/models.json +2 -2
- package/package.json +6 -3
package/README.md
CHANGED
|
@@ -74,7 +74,7 @@ pi
|
|
|
74
74
|
| Model | Context | Vision | Reasoning | Input $/M | Cache Read $/M | Output $/M |
|
|
75
75
|
|-------|---------|--------|-----------|-----------|-----------------|------------|
|
|
76
76
|
| Gemma 4 | 262K | ✅ | ✅ | $0.11 | — | $0.35 |
|
|
77
|
-
| GLM 5.2 | 524K | ❌ | ✅ | $0.90 | $0.
|
|
77
|
+
| GLM 5.2 | 524K | ❌ | ✅ | $0.90 | $0.17 | $3.00 |
|
|
78
78
|
| Kimi K2.6 | 262K | ✅ | ✅ | $0.70 | $0.20 | $3.50 |
|
|
79
79
|
| MiniMax M3 | 1.0M | ✅ | ✅ | $0.28 | $0.05 | $1.10 |
|
|
80
80
|
|
|
@@ -102,10 +102,7 @@ pi --provider lilac --model moonshotai/kimi-k2.6
|
|
|
102
102
|
|
|
103
103
|
### Thinking Mode
|
|
104
104
|
|
|
105
|
-
All Lilac models toggle reasoning via `chat_template_kwargs`, but the key each
|
|
106
|
-
model's chat template honors differs per family. The provider uses pi's
|
|
107
|
-
`chat-template` thinkingFormat with per-model `chatTemplateKwargs` (configured in
|
|
108
|
-
`patch.json`) so the right key reaches each template:
|
|
105
|
+
All Lilac models toggle reasoning via `chat_template_kwargs`, but the key each model's chat template honors differs per family. The provider uses pi's `chat-template` thinkingFormat with per-model `chatTemplateKwargs` (configured in `patch.json`) so the right key reaches each template:
|
|
109
106
|
|
|
110
107
|
| Model | Reasoning key | Default |
|
|
111
108
|
|-------|---------------|---------|
|
|
@@ -116,21 +113,9 @@ model's chat template honors differs per family. The provider uses pi's
|
|
|
116
113
|
| MiniMax M2.7 | `thinking` + `enable_thinking` (bool) | on |
|
|
117
114
|
| MiniMax M3 | `thinking_mode` (`disabled`\|`adaptive`\|`enabled`) | adaptive (server) |
|
|
118
115
|
|
|
119
|
-
Kimi K2.6, GLM 5.1, Gemma 4, and MiniMax M2.7 use the forward-compatible form
|
|
120
|
-
|
|
121
|
-
|
|
122
|
-
`reasoning_effort` (`high` = lower-latency, `max` = deepest). MiniMax M3 uses
|
|
123
|
-
the `thinking_mode` enum, exposed as three pi thinking levels: `off` →
|
|
124
|
-
`disabled` (never think), `minimal` → `adaptive` (the model decides), `high` →
|
|
125
|
-
`enabled` (always think). Pi starts at `off` (`disabled`); cycle to `minimal`
|
|
126
|
-
for M3's adaptive "model decides" mode. (The selector/footer show pi's level
|
|
127
|
-
names — `minimal`/`high` — not the `thinking_mode` values; pi has no per-model
|
|
128
|
-
level-relabel hook.)
|
|
129
|
-
|
|
130
|
-
**Preserved thinking (full-history reasoning).** By default these templates
|
|
131
|
-
trim older assistant reasoning between turns (each vendor's default), which
|
|
132
|
-
degrades multi-turn recall. Three models opt into full-history preservation via
|
|
133
|
-
a template flag sent alongside the reasoning key:
|
|
116
|
+
Kimi K2.6, GLM 5.1, Gemma 4, and MiniMax M2.7 use the forward-compatible form that sends **both** `thinking` and `enable_thinking`, so whichever key the template honors is set. GLM 5.2 additionally maps pi's thinking levels to `reasoning_effort` (`high` = lower-latency, `max` = deepest). MiniMax M3 uses the `thinking_mode` enum, exposed as three pi thinking levels: `off` → `disabled` (never think), `minimal` → `adaptive` (the model decides), `high` → `enabled` (always think). Pi starts at `off` (`disabled`); cycle to `minimal` for M3's adaptive "model decides" mode. (The selector/footer show pi's level names — `minimal`/`high` — not the `thinking_mode` values; pi has no per-model level-relabel hook.)
|
|
117
|
+
|
|
118
|
+
**Preserved thinking (full-history reasoning).** By default these templates trim older assistant reasoning between turns (each vendor's default), which degrades multi-turn recall. Three models opt into full-history preservation via a template flag sent alongside the reasoning key:
|
|
134
119
|
|
|
135
120
|
| Model | Flag | Effect |
|
|
136
121
|
|-------|------|--------|
|
|
@@ -138,14 +123,9 @@ a template flag sent alongside the reasoning key:
|
|
|
138
123
|
| GLM 5.1 | `clear_thinking: false` | keeps reasoning for all turns (default: clears before the last user message) |
|
|
139
124
|
| GLM 5.2 | `clear_thinking: false` | keeps reasoning for all turns (default: clears before the last user message) |
|
|
140
125
|
|
|
141
|
-
Kimi K2.6 and GLM 5.2 are E2E-verified on the sibling neuralwatt provider via a
|
|
142
|
-
3-turn, two-20-digit-number recall test (Kimi 0/6 → 6/6, GLM 5.2 1/4 → 4/4);
|
|
143
|
-
GLM 5.1 uses the same `clear_thinking` mechanism (confirmed in its HuggingFace
|
|
144
|
-
chat template). Gemma 4 and MiniMax M2.7/M3 expose no family-wide preserve flag,
|
|
145
|
-
so their older assistant reasoning is trimmed per the template default.
|
|
126
|
+
Kimi K2.6 and GLM 5.2 are E2E-verified on the sibling neuralwatt provider via a 3-turn, two-20-digit-number recall test (Kimi 0/6 → 6/6, GLM 5.2 1/4 → 4/4); GLM 5.1 uses the same `clear_thinking` mechanism (confirmed in its HuggingFace chat template). Gemma 4 and MiniMax M2.7/M3 expose no family-wide preserve flag, so their older assistant reasoning is trimmed per the template default.
|
|
146
127
|
|
|
147
|
-
In pi, reasoning models automatically use the appropriate thinking format. Use
|
|
148
|
-
Shift+Tab to control thinking level.
|
|
128
|
+
In pi, reasoning models automatically use the appropriate thinking format. Use Shift+Tab to control thinking level.
|
|
149
129
|
|
|
150
130
|
### Vision
|
|
151
131
|
|
package/index.ts
CHANGED
|
@@ -128,7 +128,7 @@ interface JsonModel {
|
|
|
128
128
|
id: string;
|
|
129
129
|
name: string;
|
|
130
130
|
reasoning: boolean;
|
|
131
|
-
input:
|
|
131
|
+
input: ("text" | "image")[];
|
|
132
132
|
cost: {
|
|
133
133
|
input: number;
|
|
134
134
|
output: number;
|
|
@@ -163,7 +163,7 @@ interface JsonModel {
|
|
|
163
163
|
interface PatchEntry {
|
|
164
164
|
name?: string;
|
|
165
165
|
reasoning?: boolean;
|
|
166
|
-
input?:
|
|
166
|
+
input?: ("text" | "image")[];
|
|
167
167
|
cost?: {
|
|
168
168
|
input?: number;
|
|
169
169
|
output?: number;
|
|
@@ -214,7 +214,7 @@ function parseModelOverrides(raw: unknown): Record<string, ModelOverride> | unde
|
|
|
214
214
|
if (o.thinkingLevelMap && typeof o.thinkingLevelMap === "object") {
|
|
215
215
|
const m: Record<string, string | null> = {};
|
|
216
216
|
for (const [k, v] of Object.entries(o.thinkingLevelMap as Record<string, unknown>)) {
|
|
217
|
-
if (v === null || typeof v === "string") m[k] = v;
|
|
217
|
+
if (v === null || typeof v === "string") m[k] = v as string | null;
|
|
218
218
|
}
|
|
219
219
|
if (Object.keys(m).length > 0) parsed.thinkingLevelMap = m as ThinkingLevelMap;
|
|
220
220
|
}
|
|
@@ -450,7 +450,7 @@ function transformApiModel(apiModel: any): JsonModel | null {
|
|
|
450
450
|
// multiply and preserves sub-cent cache prices like 0.003 ($/M).
|
|
451
451
|
const toPerM = (v: any) => Math.round((typeof v === "string" ? parseFloat(v) : (v || 0)) * 1_000_000 * 1e6) / 1e6;
|
|
452
452
|
|
|
453
|
-
const inputTypes:
|
|
453
|
+
const inputTypes: ("text" | "image")[] = ["text"];
|
|
454
454
|
if (hasImage) inputTypes.push("image");
|
|
455
455
|
// Video is sent as image frames, so we don't add a separate "video" input type
|
|
456
456
|
|
|
@@ -1228,7 +1228,56 @@ export default function (pi: ExtensionAPI) {
|
|
|
1228
1228
|
description: "Configure Lilac: flex discount threshold + preserved thinking per model",
|
|
1229
1229
|
async handler(_args, ctx) {
|
|
1230
1230
|
if (ctx.mode !== "tui") {
|
|
1231
|
-
|
|
1231
|
+
// GUI/RPC fallback: a select -> select flow. The custom SettingsList is
|
|
1232
|
+
// TUI-only. One setting per run; per-model preserved thinking is a
|
|
1233
|
+
// pick-model -> pick-value pair.
|
|
1234
|
+
if (!ctx.hasUI) {
|
|
1235
|
+
ctx.ui.notify("/lilac-settings requires a UI (TUI or GUI).", "error");
|
|
1236
|
+
return;
|
|
1237
|
+
}
|
|
1238
|
+
const currentFlex = getConfig().flexThreshold ?? null;
|
|
1239
|
+
const guiItems = [
|
|
1240
|
+
{ id: "flex", label: "Flex threshold", current: currentFlex == null ? "off" : String(currentFlex), values: ["off", "50", "75"] },
|
|
1241
|
+
{ id: "preserved-thinking", label: "Preserved thinking", current: "configure" },
|
|
1242
|
+
];
|
|
1243
|
+
const pick = await ctx.ui.select("Lilac settings — pick a setting", guiItems.map((i) => `${i.label}: ${i.current}`));
|
|
1244
|
+
if (pick === undefined) return;
|
|
1245
|
+
const item = guiItems.find((i) => pick.startsWith(`${i.label}:`));
|
|
1246
|
+
if (!item) return;
|
|
1247
|
+
if (item.id === "flex") {
|
|
1248
|
+
const v = await ctx.ui.select("Flex threshold", ["off", "50", "75"]);
|
|
1249
|
+
if (v === undefined) return;
|
|
1250
|
+
applyFlexThreshold(v === "off" ? null : Number(v), ctx);
|
|
1251
|
+
} else {
|
|
1252
|
+
const fresh = collectPreserveState();
|
|
1253
|
+
if (fresh.length === 0) { ctx.ui.notify("No models support preserved thinking.", "info"); return; }
|
|
1254
|
+
const modelPick = await ctx.ui.select("Preserved thinking — pick a model", fresh.map((e) => `${e.name}: ${e.preserved ? "Preserve Thinking" : "Clear Thinking"}`));
|
|
1255
|
+
if (modelPick === undefined) return;
|
|
1256
|
+
const entry = fresh.find((e) => modelPick.startsWith(`${e.name}:`));
|
|
1257
|
+
if (!entry) return;
|
|
1258
|
+
const v = await ctx.ui.select(entry.name, ["Preserve Thinking", "Clear Thinking"]);
|
|
1259
|
+
if (v === undefined) return;
|
|
1260
|
+
const preservedOn = v === "Preserve Thinking";
|
|
1261
|
+
const flagValue = entry.flag === "clear_thinking" ? !preservedOn : preservedOn;
|
|
1262
|
+
updateConfig((cfg) => {
|
|
1263
|
+
const overrides = cfg.modelOverrides ?? (cfg.modelOverrides = {});
|
|
1264
|
+
const ov = overrides[entry.id] ?? (overrides[entry.id] = {});
|
|
1265
|
+
const compat = ov.compat ?? (ov.compat = {});
|
|
1266
|
+
const kwargs = compat.chatTemplateKwargs ?? (compat.chatTemplateKwargs = {});
|
|
1267
|
+
kwargs[entry.flag] = flagValue;
|
|
1268
|
+
return cfg;
|
|
1269
|
+
});
|
|
1270
|
+
listModelsCache = null;
|
|
1271
|
+
pi.registerProvider("lilac", {
|
|
1272
|
+
baseUrl: BASE_URL,
|
|
1273
|
+
apiKey: "$LILAC_API_KEY",
|
|
1274
|
+
api: "openai-completions",
|
|
1275
|
+
models: applyDiscounts(getListModels(), latestDiscounts),
|
|
1276
|
+
});
|
|
1277
|
+
syncStatus(ctx);
|
|
1278
|
+
ctx.ui.notify(`Preserved thinking ${preservedOn ? "on" : "off"} for ${entry.name} — takes effect now.`, "info");
|
|
1279
|
+
}
|
|
1280
|
+
ctx.ui.notify("Run /lilac-settings again for more.", "info");
|
|
1232
1281
|
return;
|
|
1233
1282
|
}
|
|
1234
1283
|
const { SettingsList, Container } = await import("@earendil-works/pi-tui");
|
package/models.json
CHANGED
|
@@ -33,7 +33,7 @@
|
|
|
33
33
|
"cost": {
|
|
34
34
|
"input": 0.9,
|
|
35
35
|
"output": 3,
|
|
36
|
-
"cacheRead": 0.
|
|
36
|
+
"cacheRead": 0.17,
|
|
37
37
|
"cacheWrite": 0
|
|
38
38
|
},
|
|
39
39
|
"contextWindow": 524288,
|
|
@@ -56,7 +56,7 @@
|
|
|
56
56
|
"cost": {
|
|
57
57
|
"input": 0.7,
|
|
58
58
|
"output": 3.5,
|
|
59
|
-
"cacheRead": 0.
|
|
59
|
+
"cacheRead": 0.16,
|
|
60
60
|
"cacheWrite": 0
|
|
61
61
|
},
|
|
62
62
|
"contextWindow": 262144,
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "pi-lilac-provider",
|
|
3
|
-
"version": "1.
|
|
3
|
+
"version": "1.9.1",
|
|
4
4
|
"description": "Lilac provider extension for pi - Access Kimi K2.6, GLM 5.1, and Gemma 4 models through Lilac's OpenAI-compatible API on idle GPUs",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"main": "index.ts",
|
|
@@ -29,7 +29,10 @@
|
|
|
29
29
|
"@earendil-works/pi-coding-agent": ">=0.74.0"
|
|
30
30
|
},
|
|
31
31
|
"devDependencies": {
|
|
32
|
-
"@earendil-works/pi-coding-agent": "0.80.3"
|
|
32
|
+
"@earendil-works/pi-coding-agent": "0.80.3",
|
|
33
|
+
"typescript": "6.0.3",
|
|
34
|
+
"@types/node": "25.9.1",
|
|
35
|
+
"knip": "6.14.1"
|
|
33
36
|
},
|
|
34
37
|
"pi": {
|
|
35
38
|
"extensions": [
|
|
@@ -39,7 +42,7 @@
|
|
|
39
42
|
"scripts": {
|
|
40
43
|
"clean": "echo 'nothing to clean'",
|
|
41
44
|
"build": "echo 'nothing to build'",
|
|
42
|
-
"check": "
|
|
45
|
+
"check": "tsc --noEmit && knip --no-gitignore",
|
|
43
46
|
"test": "node scripts/test-discounts.ts && node scripts/test-preserved-thinking.ts && node scripts/test-model-overrides.ts && node scripts/test-flex.ts",
|
|
44
47
|
"test:discounts": "node scripts/test-discounts.ts",
|
|
45
48
|
"test:thinking": "node scripts/test-preserved-thinking.ts",
|