pi-lilac-provider 1.4.1 → 1.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -128,6 +128,23 @@ for M3's adaptive "model decides" mode. (The selector/footer show pi's level
128
128
  names — `minimal`/`high` — not the `thinking_mode` values; pi has no per-model
129
129
  level-relabel hook.)
130
130
 
131
+ **Preserved thinking (full-history reasoning).** By default these templates
132
+ trim older assistant reasoning between turns (each vendor's default), which
133
+ degrades multi-turn recall. Three models opt into full-history preservation via
134
+ a template flag sent alongside the reasoning key:
135
+
136
+ | Model | Flag | Effect |
137
+ |-------|------|--------|
138
+ | Kimi K2.6 | `preserve_thinking: true` | keeps every assistant turn's reasoning (default: only the last) |
139
+ | GLM 5.1 | `clear_thinking: false` | keeps reasoning for all turns (default: clears before the last user message) |
140
+ | GLM 5.2 | `clear_thinking: false` | keeps reasoning for all turns (default: clears before the last user message) |
141
+
142
+ Kimi K2.6 and GLM 5.2 are E2E-verified on the sibling neuralwatt provider via a
143
+ 3-turn, two-20-digit-number recall test (Kimi 0/6 → 6/6, GLM 5.2 1/4 → 4/4);
144
+ GLM 5.1 uses the same `clear_thinking` mechanism (confirmed in its HuggingFace
145
+ chat template). Gemma 4 and MiniMax M2.7/M3 expose no family-wide preserve flag,
146
+ so their older assistant reasoning is trimmed per the template default.
147
+
131
148
  In pi, reasoning models automatically use the appropriate thinking format. Use
132
149
  Shift+Tab to control thinking level.
133
150
 
@@ -173,7 +190,7 @@ Add to your pi configuration for automatic loading:
173
190
 
174
191
  Lilac's API is OpenAI-compatible with these specifics:
175
192
 
176
- - **`thinkingFormat: "chat-template"`** — All reasoning models. Lilac's vLLM backend toggles reasoning via `chat_template_kwargs`, but the honored key differs per model family. Per-model `chatTemplateKwargs` in `patch.json` send the right key(s): `thinking`+`enable_thinking` (bool) for Kimi K2.6, GLM 5.1, Gemma 4, and MiniMax M2.7; `enable_thinking` + `reasoning_effort` for GLM 5.2; `thinking_mode` (adaptive|enabled|disabled) for MiniMax M3.
193
+ - **`thinkingFormat: "chat-template"`** — All reasoning models. Lilac's vLLM backend toggles reasoning via `chat_template_kwargs`, but the honored key differs per model family. Per-model `chatTemplateKwargs` in `patch.json` send the right key(s): `thinking`+`enable_thinking` (bool) for Kimi K2.6, GLM 5.1, Gemma 4, and MiniMax M2.7; `enable_thinking` + `reasoning_effort` for GLM 5.2; `thinking_mode` (adaptive|enabled|disabled) for MiniMax M3. Kimi K2.6, GLM 5.1, and GLM 5.2 additionally send a preservation flag (`preserve_thinking: true` / `clear_thinking: false`) to retain full reasoning history across turns — see [Preserved thinking](#thinking-mode) above.
177
194
  - **`maxTokensField: "max_completion_tokens"`** — All models. Lilac supports `max_completion_tokens` (preferred for reasoning models as it includes reasoning tokens).
178
195
  - **`supportsDeveloperRole: true`** — All models. Lilac's vLLM backend maps the developer role to system.
179
196
  - **`supportsStore: false`** — All models. Lilac doesn't support the `store` parameter.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "pi-lilac-provider",
3
- "version": "1.4.1",
3
+ "version": "1.5.0",
4
4
  "description": "Lilac provider extension for pi - Access Kimi K2.6, GLM 5.1, and Gemma 4 models through Lilac's OpenAI-compatible API on idle GPUs",
5
5
  "type": "module",
6
6
  "main": "index.ts",
@@ -35,6 +35,7 @@
35
35
  "build": "echo 'nothing to build'",
36
36
  "check": "echo 'nothing to check'",
37
37
  "test": "node scripts/test-discounts.ts",
38
+ "test:thinking": "node scripts/test-preserved-thinking.ts",
38
39
  "update-models": "node scripts/update-models.js"
39
40
  }
40
41
  }
package/patch.json CHANGED
@@ -10,7 +10,8 @@
10
10
  "thinkingFormat": "chat-template",
11
11
  "chatTemplateKwargs": {
12
12
  "thinking": { "$var": "thinking.enabled" },
13
- "enable_thinking": { "$var": "thinking.enabled" }
13
+ "enable_thinking": { "$var": "thinking.enabled" },
14
+ "preserve_thinking": true
14
15
  },
15
16
  "maxTokensField": "max_completion_tokens",
16
17
  "supportsDeveloperRole": false,
@@ -29,7 +30,8 @@
29
30
  "thinkingFormat": "chat-template",
30
31
  "chatTemplateKwargs": {
31
32
  "thinking": { "$var": "thinking.enabled" },
32
- "enable_thinking": { "$var": "thinking.enabled" }
33
+ "enable_thinking": { "$var": "thinking.enabled" },
34
+ "clear_thinking": false
33
35
  },
34
36
  "maxTokensField": "max_completion_tokens",
35
37
  "supportsDeveloperRole": false,
@@ -42,7 +44,8 @@
42
44
  "thinkingFormat": "chat-template",
43
45
  "chatTemplateKwargs": {
44
46
  "enable_thinking": { "$var": "thinking.enabled" },
45
- "reasoning_effort": { "$var": "thinking.effort", "omitWhenOff": true }
47
+ "reasoning_effort": { "$var": "thinking.effort", "omitWhenOff": true },
48
+ "clear_thinking": false
46
49
  }
47
50
  },
48
51
  "thinkingLevelMap": {
@@ -17,7 +17,7 @@
17
17
  * recomputed from list price so re-applied discounts never compound.
18
18
  * 11. before_provider_request mutates the bound (in-flight) model object so the
19
19
  * current turn's cost calc sees the discount in real time.
20
- * 12. session_start schedules a 10-minute background /status poll to cover idle
20
+ * 12. session_start schedules a 5-minute background /status poll to cover idle
21
21
  * sessions; session_shutdown clears it.
22
22
  */
23
23
 
@@ -700,12 +700,12 @@ assert(
700
700
  "deferred session_start paints the NEW lilac model's discount (glm), not the stale kimi capture",
701
701
  );
702
702
 
703
- // ─── Test 17: session_start schedules a 10-min idle poll; shutdown clears it ─
703
+ // ─── Test 17: session_start schedules a 5-min idle poll; shutdown clears it ─
704
704
 
705
705
  console.log("\n--- Test 17: session_start schedules idle poll; session_shutdown clears it ---");
706
706
 
707
- // Lilac refreshes discounts ~every 10 minutes. session_start must schedule a
708
- // background /status poll at that cadence to cover idle sessions (turn fetches
707
+ // Lilac refreshes discounts ~every 10 minutes, so session_start polls /status
708
+ // every 5 min (half the refresh window) to cover idle sessions (turn fetches
709
709
  // only run when the user sends a message), and session_shutdown must clear it so
710
710
  // it neither leaks nor keeps the process alive. Wrap the global timer APIs to
711
711
  // capture the scheduled delay + handle, then confirm shutdown clears it.
@@ -729,7 +729,7 @@ globalThis.clearInterval = ((handle: ReturnType<typeof setInterval>) => {
729
729
 
730
730
  try {
731
731
  // Benign fetch mock so the session_start fire-and-forget /models + /status
732
- // fetch doesn't hit the network. (The poll itself never fires — 10 min — so
732
+ // fetch doesn't hit the network. (The poll itself never fires — 5 min — so
733
733
  // only the startup fetch needs mocking here.)
734
734
  globalThis.fetch = mockFetch({
735
735
  "/models": { body: { data: [] } },
@@ -759,8 +759,8 @@ try {
759
759
  }
760
760
 
761
761
  assert(
762
- scheduledDelay === 10 * 60 * 1000,
763
- "session_start schedules a 10-minute (600000ms) /status poll for idle sessions",
762
+ scheduledDelay === 5 * 60 * 1000,
763
+ "session_start schedules a 5-minute (300000ms) /status poll for idle sessions",
764
764
  );
765
765
  assert(scheduledHandle !== null, "poll interval handle was captured");
766
766
  const capturedHandle = scheduledHandle;
@@ -0,0 +1,237 @@
1
+ #!/usr/bin/env node
2
+ /**
3
+ * Wire-level test for preserved-thinking (full-history reasoning) flags.
4
+ *
5
+ * Verifies, against the REAL pi-ai streamSimple + a stubbed fetch, that the
6
+ * `chat_template_kwargs` each Lilac reasoning model puts on the wire include the
7
+ * preservation flags configured in patch.json — and that those static flags
8
+ * coexist with the { $var }-resolved thinking keys at every thinking level.
9
+ *
10
+ * Background: reasoning models trim older assistant reasoning across turns by
11
+ * default (each vendor's template default). Two template-level flags opt into
12
+ * full-history preservation (confirmed in each model's HuggingFace chat
13
+ * template; Kimi K2.6 and GLM 5.2 additionally E2E-verified on the sibling
14
+ * neuralwatt provider's 3-turn / two-20-digit-number recall test):
15
+ *
16
+ * Kimi K2.6 → preserve_thinking: true (template: "if preserve_thinking, keep -1 ... retain reasoning")
17
+ * GLM 5.1 → clear_thinking: false (same clear_thinking mechanism as GLM 5.2)
18
+ * GLM 5.2 → clear_thinking: false (template: keep reasoning when "clear_thinking is defined and not clear_thinking")
19
+ *
20
+ * Lilac uses pi-ai's `chat-template` thinkingFormat, which calls
21
+ * buildChatTemplateKwargs → resolveChatTemplateKwargValue. Static primitive
22
+ * kwargs pass through verbatim; only { $var } objects are resolved against the
23
+ * turn's thinking state. So the preserve flags ride onto the wire as plain
24
+ * booleans next to the { $var } thinking/enable_thinking keys — no onPayload
25
+ * hook needed (unlike neuralwatt, which drives the openai reasoning_effort path
26
+ * and must inject via onPayload because the two paths are mutually exclusive).
27
+ *
28
+ * Gemma 4 and MiniMax M2.7/M3 expose NO family-wide preserve flag (their HF
29
+ * templates read only enable_thinking / thinking_mode + reasoning_content), so
30
+ * no flag is added for them here — asserted as a regression guard.
31
+ *
32
+ * Run: node scripts/test-preserved-thinking.ts
33
+ */
34
+ import fs from "fs";
35
+ import path from "path";
36
+ import { pathToFileURL } from "url";
37
+
38
+ const HERE = path.dirname(new URL(import.meta.url).pathname);
39
+
40
+ // ─── Faithful replica of index.ts applyPatch / buildModels ────────────────────
41
+ // Mirrors the provider's model pipeline so the test exercises the same objects
42
+ // the runtime registers. (Kept inline so the test has no TS-import dependency
43
+ // on index.ts, which pulls in pi-coding-agent types.)
44
+
45
+ function applyPatch(model, patch) {
46
+ const result = { ...model };
47
+ if (patch.name !== undefined) result.name = patch.name;
48
+ if (patch.reasoning !== undefined) result.reasoning = patch.reasoning;
49
+ if (patch.input !== undefined) result.input = patch.input;
50
+ if (patch.contextWindow !== undefined) result.contextWindow = patch.contextWindow;
51
+ if (patch.maxTokens !== undefined) result.maxTokens = patch.maxTokens;
52
+ if (patch.cost) {
53
+ result.cost = {
54
+ input: patch.cost.input ?? result.cost.input,
55
+ output: patch.cost.output ?? result.cost.output,
56
+ cacheRead: patch.cost.cacheRead ?? result.cost.cacheRead,
57
+ cacheWrite: patch.cost.cacheWrite ?? result.cost.cacheWrite,
58
+ };
59
+ }
60
+ if (patch.compat) result.compat = { ...(result.compat || {}), ...patch.compat };
61
+ if (patch.thinkingLevelMap !== undefined) result.thinkingLevelMap = patch.thinkingLevelMap;
62
+ if (!result.reasoning && result.compat?.thinkingFormat) delete result.compat.thinkingFormat;
63
+ if (!result.reasoning && result.thinkingLevelMap) delete result.thinkingLevelMap;
64
+ if (result.compat && Object.keys(result.compat).length === 0) delete result.compat;
65
+ return result;
66
+ }
67
+
68
+ function buildModels(base, custom, patch) {
69
+ const map = new Map();
70
+ for (const m of base) map.set(m.id, m);
71
+ for (const [id, p] of Object.entries(patch)) {
72
+ const ex = map.get(id);
73
+ if (ex) map.set(id, applyPatch(ex, p));
74
+ }
75
+ for (const m of custom) {
76
+ const ex = map.get(m.id);
77
+ const p = patch[m.id];
78
+ if (ex && p) map.set(m.id, applyPatch(m, p));
79
+ else if (ex) map.set(m.id, m);
80
+ else if (p) map.set(m.id, applyPatch(m, p));
81
+ else map.set(m.id, m);
82
+ }
83
+ return Array.from(map.values());
84
+ }
85
+
86
+ // ─── Load REAL pi-ai streamSimple from the global pi install ──────────────────
87
+ // Imported by absolute path so its relative deps (openai, ../models.js, ...) and
88
+ // the `openai` SDK resolve from the global pi node_modules tree.
89
+ const PI_AI_API = path.join(
90
+ os_home_pi_ai(),
91
+ "dist/api/openai-completions.js",
92
+ );
93
+ function os_home_pi_ai() {
94
+ // Resolve the pi-ai package shipped inside the globally-installed pi agent.
95
+ const candidates = [
96
+ "/Users/monotykamary/.npm-global/lib/node_modules/@earendil-works/pi-coding-agent/node_modules/@earendil-works/pi-ai",
97
+ ];
98
+ for (const c of candidates) if (fs.existsSync(c)) return c;
99
+ throw new Error("Could not locate pi-ai package in the global pi install.");
100
+ }
101
+
102
+ const { streamSimple } = await import(pathToFileURL(PI_AI_API).href);
103
+
104
+ // ─── Build the exact model objects the provider registers ────────────────────
105
+ const embedded = JSON.parse(fs.readFileSync(path.join(HERE, "..", "models.json"), "utf8"));
106
+ const custom = JSON.parse(fs.readFileSync(path.join(HERE, "..", "custom-models.json"), "utf8"));
107
+ const patch = JSON.parse(fs.readFileSync(path.join(HERE, "..", "patch.json"), "utf8"));
108
+ const models = new Map(buildModels(embedded, custom, patch).map((m) => [m.id, m]));
109
+
110
+ // ─── Stub fetch to capture the request body (fires before response parsing) ──
111
+ const captured = [];
112
+ const originalFetch = globalThis.fetch;
113
+ globalThis.fetch = async (url, init) => {
114
+ captured.push({ url: String(url), body: init?.body ?? null });
115
+ // Minimal SSE response that ends immediately — enough for the OpenAI SDK to
116
+ // construct the stream; we only need the request body, captured above.
117
+ return new Response(new ReadableStream({ start(c) { c.close(); } }), {
118
+ headers: { "content-type": "text/event-stream" },
119
+ });
120
+ };
121
+
122
+ async function wire(modelId, reasoning) {
123
+ const model = models.get(modelId);
124
+ if (!model) throw new Error(`unknown model: ${modelId}`);
125
+ captured.length = 0;
126
+ const ctx = { messages: [{ role: "user", content: [{ type: "text", text: "hi" }] }] };
127
+ const s = streamSimple(
128
+ { ...model, provider: "lilac", api: "openai-completions", baseUrl: "https://api.getlilac.com/v1" },
129
+ ctx,
130
+ { apiKey: "sk-test", reasoning },
131
+ );
132
+ // The fetch fires inside stream()'s async IIFE; poll for it.
133
+ const deadline = Date.now() + 2000;
134
+ while (captured.length === 0 && Date.now() < deadline) await new Promise((r) => setTimeout(r, 20));
135
+ try { s.end?.(); } catch {}
136
+ const hit = captured.find((c) => c.url.includes("/chat/completions"));
137
+ if (!hit) throw new Error(`no /chat/completions request captured for ${modelId} (${reasoning})`);
138
+ return JSON.parse(hit.body);
139
+ }
140
+
141
+ // ─── Assertions ───────────────────────────────────────────────────────────────
142
+ let failures = 0;
143
+ function eq(actual, expected, msg) {
144
+ const a = JSON.stringify(actual);
145
+ const e = JSON.stringify(expected);
146
+ const ok = a === e;
147
+ console.log(`${ok ? "✓" : "✗"} ${msg}`);
148
+ if (!ok) {
149
+ failures++;
150
+ console.log(` expected ${e}`);
151
+ console.log(` actual ${a}`);
152
+ }
153
+ }
154
+ function truthy(actual, msg) {
155
+ const ok = !!actual;
156
+ console.log(`${ok ? "✓" : "✗"} ${msg}`);
157
+ if (!ok) {
158
+ failures++;
159
+ console.log(` expected truthy, got ${JSON.stringify(actual)}`);
160
+ }
161
+ }
162
+ function falsy(actual, msg) {
163
+ const ok = !actual;
164
+ console.log(`${ok ? "✓" : "✗"} ${msg}`);
165
+ if (!ok) {
166
+ failures++;
167
+ console.log(` expected falsy/undefined, got ${JSON.stringify(actual)}`);
168
+ }
169
+ }
170
+
171
+ console.log("\n=== patch.json data ===");
172
+ const kimiPatch = patch["moonshotai/kimi-k2.6"]?.compat?.chatTemplateKwargs;
173
+ eq(kimiPatch?.preserve_thinking, true, "kimi-k2.6 patch sets preserve_thinking: true");
174
+ const glm51Patch = patch["zai-org/glm-5.1"]?.compat?.chatTemplateKwargs;
175
+ eq(glm51Patch?.clear_thinking, false, "glm-5.1 patch sets clear_thinking: false");
176
+ const glm52Patch = patch["zai-org/glm-5.2"]?.compat?.chatTemplateKwargs;
177
+ eq(glm52Patch?.clear_thinking, false, "glm-5.2 patch sets clear_thinking: false");
178
+
179
+ console.log("\n=== Kimi K2.6 on the wire (real pi-ai) ===");
180
+ {
181
+ const high = await wire("moonshotai/kimi-k2.6", "high");
182
+ eq(high.chat_template_kwargs, { thinking: true, enable_thinking: true, preserve_thinking: true },
183
+ "kimi @ high → thinking+enable_thinking true AND preserve_thinking true");
184
+ const off = await wire("moonshotai/kimi-k2.6", "off");
185
+ eq(off.chat_template_kwargs, { thinking: false, enable_thinking: false, preserve_thinking: true },
186
+ "kimi @ off → thinking false but preserve_thinking still true (level-independent)");
187
+ const minimal = await wire("moonshotai/kimi-k2.6", "minimal");
188
+ eq(minimal.chat_template_kwargs, { thinking: true, enable_thinking: true, preserve_thinking: true },
189
+ "kimi @ minimal → preserve_thinking present at every level");
190
+ }
191
+
192
+ console.log("\n=== GLM 5.2 on the wire (real pi-ai) ===");
193
+ {
194
+ const high = await wire("zai-org/glm-5.2", "high");
195
+ eq(high.chat_template_kwargs, { enable_thinking: true, reasoning_effort: "high", clear_thinking: false },
196
+ "glm-5.2 @ high → enable_thinking+reasoning_effort AND clear_thinking false");
197
+ const xhigh = await wire("zai-org/glm-5.2", "xhigh");
198
+ eq(xhigh.chat_template_kwargs, { enable_thinking: true, reasoning_effort: "max", clear_thinking: false },
199
+ "glm-5.2 @ xhigh → reasoning_effort max, clear_thinking false");
200
+ const off = await wire("zai-org/glm-5.2", "off");
201
+ eq(off.chat_template_kwargs, { enable_thinking: false, clear_thinking: false },
202
+ "glm-5.2 @ off → reasoning_effort omitted (omitWhenOff), clear_thinking false persists");
203
+ }
204
+
205
+ console.log("\n=== GLM 5.1 on the wire (real pi-ai) ===");
206
+ {
207
+ const high = await wire("zai-org/glm-5.1", "high");
208
+ eq(high.chat_template_kwargs, { thinking: true, enable_thinking: true, clear_thinking: false },
209
+ "glm-5.1 @ high → thinking+enable_thinking true AND clear_thinking false");
210
+ const off = await wire("zai-org/glm-5.1", "off");
211
+ eq(off.chat_template_kwargs, { thinking: false, enable_thinking: false, clear_thinking: false },
212
+ "glm-5.1 @ off → thinking false but clear_thinking false persists");
213
+ }
214
+
215
+ console.log("\n=== Gemma 4 / MiniMax (no family-wide preserve flag — regression guard) ===");
216
+ {
217
+ const gemma = await wire("google/gemma-4-31b-it", "high");
218
+ eq(gemma.chat_template_kwargs, { thinking: true, enable_thinking: true },
219
+ "gemma-4 @ high → only thinking/enable_thinking (no preserve/clear flag)");
220
+ falsy(gemma.chat_template_kwargs?.preserve_thinking, "gemma-4 has no preserve_thinking");
221
+ falsy(gemma.chat_template_kwargs?.clear_thinking, "gemma-4 has no clear_thinking");
222
+
223
+ const m3 = await wire("minimaxai/minimax-m3", "high");
224
+ eq(m3.chat_template_kwargs, { thinking_mode: "enabled" },
225
+ "minimax-m3 @ high → only thinking_mode (no preserve/clear flag)");
226
+ falsy(m3.chat_template_kwargs?.preserve_thinking, "minimax-m3 has no preserve_thinking");
227
+
228
+ const m27 = await wire("minimaxai/minimax-m2.7", "high");
229
+ eq(m27.chat_template_kwargs, { thinking: true, enable_thinking: true },
230
+ "minimax-m2.7 @ high → only thinking/enable_thinking (no preserve/clear flag)");
231
+ falsy(m27.chat_template_kwargs?.clear_thinking, "minimax-m2.7 has no clear_thinking");
232
+ }
233
+
234
+ globalThis.fetch = originalFetch;
235
+
236
+ console.log(`\n${failures === 0 ? "ALL PASS" : `${failures} FAILURE(S)`}`);
237
+ if (failures > 0) process.exit(1);