pi-lilac-provider 1.8.0 → 1.8.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -74,10 +74,8 @@ pi
74
74
  | Model | Context | Vision | Reasoning | Input $/M | Cache Read $/M | Output $/M |
75
75
  |-------|---------|--------|-----------|-----------|-----------------|------------|
76
76
  | Gemma 4 | 262K | ✅ | ✅ | $0.11 | — | $0.35 |
77
- | GLM 5.1 | 203K | ❌ | ✅ | $0.90 | $0.27 | $3.00 |
78
77
  | GLM 5.2 | 524K | ❌ | ✅ | $0.90 | $0.27 | $3.00 |
79
78
  | Kimi K2.6 | 262K | ✅ | ✅ | $0.70 | $0.20 | $3.50 |
80
- | MiniMax M2.7 | 205K | ❌ | ✅ | $0.30 | $0.06 | $1.20 |
81
79
  | MiniMax M3 | 1.0M | ✅ | ✅ | $0.28 | $0.05 | $1.10 |
82
80
 
83
81
  *Costs are per million tokens. Prices subject to change — check [getlilac.com](https://getlilac.com/) for current pricing.*
@@ -121,7 +119,7 @@ model's chat template honors differs per family. The provider uses pi's
121
119
  Kimi K2.6, GLM 5.1, Gemma 4, and MiniMax M2.7 use the forward-compatible form
122
120
  that sends **both** `thinking` and `enable_thinking`, so whichever key the
123
121
  template honors is set. GLM 5.2 additionally maps pi's thinking levels to
124
- `reasoning_effort` (`high` = lower-latency, `xhigh` = `max`). MiniMax M3 uses
122
+ `reasoning_effort` (`high` = lower-latency, `max` = deepest). MiniMax M3 uses
125
123
  the `thinking_mode` enum, exposed as three pi thinking levels: `off` →
126
124
  `disabled` (never think), `minimal` → `adaptive` (the model decides), `high` →
127
125
  `enabled` (always think). Pi starts at `off` (`disabled`); cycle to `minimal`
package/index.ts CHANGED
@@ -21,7 +21,7 @@
21
21
  * MiniMax M3). The forward-compatible form sends BOTH `thinking` and
22
22
  * `enable_thinking` so whichever key the template honors is set:
23
23
  * { chat_template_kwargs: { thinking: <bool>, enable_thinking: <bool> } }
24
- * GLM 5.2 adds `reasoning_effort` (high = lower-latency, xhigh = max) via a
24
+ * GLM 5.2 adds `reasoning_effort` (high for lower-latency, max for deep) via a
25
25
  * thinkingLevelMap. MiniMax M3 maps to the `thinking_mode` enum as three pi
26
26
  * thinking levels — off→disabled, minimal→adaptive (model decides), high→enabled
27
27
  * — so adaptive is selectable via pi's Shift+Tab cycle (off→minimal→high). Pi
@@ -99,7 +99,7 @@ interface JsonDiscount {
99
99
  creditMultiplier: number;
100
100
  }
101
101
 
102
- // Maps pi's thinking levels (off, minimal, low, medium, high, xhigh) to the
102
+ // Maps pi's thinking levels (off, minimal, low, medium, high, xhigh, max) to the
103
103
  // provider-specific effort string sent on the wire. A `null` value marks a
104
104
  // level as unsupported — clampThinkingLevel skips it when resolving the
105
105
  // user's selection. Mirrors pi-ai's ThinkingLevelMap shape.
@@ -110,6 +110,7 @@ type ThinkingLevelMap = {
110
110
  medium?: string | null;
111
111
  high?: string | null;
112
112
  xhigh?: string | null;
113
+ max?: string | null;
113
114
  };
114
115
 
115
116
  // A chat_template_kwargs value, mirroring pi-ai's ChatTemplateKwargSchema. Scalar
@@ -1252,7 +1253,7 @@ export default function (pi: ExtensionAPI) {
1252
1253
  },
1253
1254
  {
1254
1255
  id: "preserved-thinking",
1255
- label: "Preserved thinking",
1256
+ label: "Preserved thinking ›",
1256
1257
  description: "Per-model Preserve Thinking / Clear Thinking (full-history reasoning). Preserve Thinking keeps all turns' reasoning; Clear Thinking lets the template drop older reasoning (saves tokens, can hurt multi-turn recall / cause overthinking).",
1257
1258
  currentValue: "configure",
1258
1259
  submenu: (_currentValue: string, subDone: (v?: string) => void) => {
package/models.json CHANGED
@@ -23,29 +23,6 @@
23
23
  "zaiToolStream": true
24
24
  }
25
25
  },
26
- {
27
- "id": "zai-org/glm-5.1",
28
- "name": "GLM 5.1",
29
- "reasoning": true,
30
- "input": [
31
- "text"
32
- ],
33
- "cost": {
34
- "input": 0.9,
35
- "output": 3,
36
- "cacheRead": 0.27,
37
- "cacheWrite": 0
38
- },
39
- "contextWindow": 202752,
40
- "maxTokens": 131072,
41
- "compat": {
42
- "supportsDeveloperRole": false,
43
- "supportsStore": false,
44
- "maxTokensField": "max_completion_tokens",
45
- "thinkingFormat": "qwen-chat-template",
46
- "zaiToolStream": true
47
- }
48
- },
49
26
  {
50
27
  "id": "zai-org/glm-5.2",
51
28
  "name": "GLM 5.2",
@@ -92,28 +69,6 @@
92
69
  "zaiToolStream": true
93
70
  }
94
71
  },
95
- {
96
- "id": "minimaxai/minimax-m2.7",
97
- "name": "MiniMax M2.7",
98
- "reasoning": true,
99
- "input": [
100
- "text"
101
- ],
102
- "cost": {
103
- "input": 0.3,
104
- "output": 1.2,
105
- "cacheRead": 0.055,
106
- "cacheWrite": 0
107
- },
108
- "contextWindow": 204800,
109
- "maxTokens": 204800,
110
- "compat": {
111
- "supportsDeveloperRole": true,
112
- "supportsStore": false,
113
- "maxTokensField": "max_completion_tokens",
114
- "thinkingFormat": "qwen-chat-template"
115
- }
116
- },
117
72
  {
118
73
  "id": "minimaxai/minimax-m3",
119
74
  "name": "MiniMax M3",
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "pi-lilac-provider",
3
- "version": "1.8.0",
3
+ "version": "1.8.2",
4
4
  "description": "Lilac provider extension for pi - Access Kimi K2.6, GLM 5.1, and Gemma 4 models through Lilac's OpenAI-compatible API on idle GPUs",
5
5
  "type": "module",
6
6
  "main": "index.ts",
package/patch.json CHANGED
@@ -53,7 +53,7 @@
53
53
  "low": "high",
54
54
  "medium": "high",
55
55
  "high": "high",
56
- "xhigh": "max"
56
+ "max": "max"
57
57
  }
58
58
  },
59
59
  "google/gemma-4-31b-it": {
@@ -66,7 +66,6 @@ function eq<T>(actual: T, expected: T, message: string) {
66
66
 
67
67
  const KIMI = "moonshotai/kimi-k2.6";
68
68
  const GLM52 = "zai-org/glm-5.2";
69
- const GLM51 = "zai-org/glm-5.1";
70
69
 
71
70
  // ─── applyModelOverride ────────────────────────────────────────────────────────
72
71
 
@@ -240,16 +239,6 @@ function find(models: any[], id: string): any {
240
239
  assert(glm.compat.chatTemplateKwargs.clear_thinking === false, "non-overridden clear_thinking survives a thinkingLevelMap-only override");
241
240
  }
242
241
 
243
- {
244
- // Override on glm-5.1 toggles clear_thinking (patch sets false); other compat survives
245
- const overrides = { [GLM51]: { compat: { chatTemplateKwargs: { clear_thinking: true } } } } as any;
246
- const models = buildModels(modelsData, customModelsData, patchData, overrides);
247
- const glm = find(models, GLM51);
248
- assert(glm.compat.chatTemplateKwargs.clear_thinking === true, "override wins over patch: glm-5.1 clear_thinking -> true");
249
- assert((glm.compat.chatTemplateKwargs as any).thinking?.$var === "thinking.enabled", "override deep-merges: glm-5.1 thinking $var key survives");
250
- assert(glm.compat.zaiToolStream === true, "override deep-merges: glm-5.1 zaiToolStream survives");
251
- }
252
-
253
242
  {
254
243
  // Override for an unknown id is a no-op (adds no models)
255
244
  const before = buildModels(modelsData, customModelsData, patchData, {});
@@ -194,25 +194,15 @@ console.log("\n=== GLM 5.2 on the wire (real pi-ai) ===");
194
194
  const high = await wire("zai-org/glm-5.2", "high");
195
195
  eq(high.chat_template_kwargs, { enable_thinking: true, reasoning_effort: "high", clear_thinking: false },
196
196
  "glm-5.2 @ high → enable_thinking+reasoning_effort AND clear_thinking false");
197
- const xhigh = await wire("zai-org/glm-5.2", "xhigh");
198
- eq(xhigh.chat_template_kwargs, { enable_thinking: true, reasoning_effort: "max", clear_thinking: false },
199
- "glm-5.2 @ xhigh → reasoning_effort max, clear_thinking false");
197
+ const max = await wire("zai-org/glm-5.2", "max");
198
+ eq(max.chat_template_kwargs, { enable_thinking: true, reasoning_effort: "max", clear_thinking: false },
199
+ "glm-5.2 @ max → reasoning_effort max, clear_thinking false");
200
200
  const off = await wire("zai-org/glm-5.2", "off");
201
201
  eq(off.chat_template_kwargs, { enable_thinking: false, clear_thinking: false },
202
202
  "glm-5.2 @ off → reasoning_effort omitted (omitWhenOff), clear_thinking false persists");
203
203
  }
204
204
 
205
- console.log("\n=== GLM 5.1 on the wire (real pi-ai) ===");
206
- {
207
- const high = await wire("zai-org/glm-5.1", "high");
208
- eq(high.chat_template_kwargs, { thinking: true, enable_thinking: true, clear_thinking: false },
209
- "glm-5.1 @ high → thinking+enable_thinking true AND clear_thinking false");
210
- const off = await wire("zai-org/glm-5.1", "off");
211
- eq(off.chat_template_kwargs, { thinking: false, enable_thinking: false, clear_thinking: false },
212
- "glm-5.1 @ off → thinking false but clear_thinking false persists");
213
- }
214
-
215
- console.log("\n=== Gemma 4 / MiniMax (no family-wide preserve flag — regression guard) ===");
205
+ console.log("\n=== Gemma / MiniMax (no family-wide preserve flag — regression guard) ===");
216
206
  {
217
207
  const gemma = await wire("google/gemma-4-31b-it", "high");
218
208
  eq(gemma.chat_template_kwargs, { thinking: true, enable_thinking: true },
@@ -224,11 +214,6 @@ console.log("\n=== Gemma 4 / MiniMax (no family-wide preserve flag — regressio
224
214
  eq(m3.chat_template_kwargs, { thinking_mode: "enabled" },
225
215
  "minimax-m3 @ high → only thinking_mode (no preserve/clear flag)");
226
216
  falsy(m3.chat_template_kwargs?.preserve_thinking, "minimax-m3 has no preserve_thinking");
227
-
228
- const m27 = await wire("minimaxai/minimax-m2.7", "high");
229
- eq(m27.chat_template_kwargs, { thinking: true, enable_thinking: true },
230
- "minimax-m2.7 @ high → only thinking/enable_thinking (no preserve/clear flag)");
231
- falsy(m27.chat_template_kwargs?.clear_thinking, "minimax-m2.7 has no clear_thinking");
232
217
  }
233
218
 
234
219
  globalThis.fetch = originalFetch;