pi-lilac-provider 1.4.1 → 1.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +18 -1
- package/package.json +2 -1
- package/patch.json +6 -3
- package/scripts/test-discounts.ts +7 -7
- package/scripts/test-preserved-thinking.ts +237 -0
package/README.md
CHANGED
|
@@ -128,6 +128,23 @@ for M3's adaptive "model decides" mode. (The selector/footer show pi's level
|
|
|
128
128
|
names — `minimal`/`high` — not the `thinking_mode` values; pi has no per-model
|
|
129
129
|
level-relabel hook.)
|
|
130
130
|
|
|
131
|
+
**Preserved thinking (full-history reasoning).** By default these templates
|
|
132
|
+
trim older assistant reasoning between turns (each vendor's default), which
|
|
133
|
+
degrades multi-turn recall. Three models opt into full-history preservation via
|
|
134
|
+
a template flag sent alongside the reasoning key:
|
|
135
|
+
|
|
136
|
+
| Model | Flag | Effect |
|
|
137
|
+
|-------|------|--------|
|
|
138
|
+
| Kimi K2.6 | `preserve_thinking: true` | keeps every assistant turn's reasoning (default: only the last) |
|
|
139
|
+
| GLM 5.1 | `clear_thinking: false` | keeps reasoning for all turns (default: clears before the last user message) |
|
|
140
|
+
| GLM 5.2 | `clear_thinking: false` | keeps reasoning for all turns (default: clears before the last user message) |
|
|
141
|
+
|
|
142
|
+
Kimi K2.6 and GLM 5.2 are E2E-verified on the sibling neuralwatt provider via a
|
|
143
|
+
3-turn, two-20-digit-number recall test (Kimi 0/6 → 6/6, GLM 5.2 1/4 → 4/4);
|
|
144
|
+
GLM 5.1 uses the same `clear_thinking` mechanism (confirmed in its HuggingFace
|
|
145
|
+
chat template). Gemma 4 and MiniMax M2.7/M3 expose no family-wide preserve flag,
|
|
146
|
+
so their older assistant reasoning is trimmed per the template default.
|
|
147
|
+
|
|
131
148
|
In pi, reasoning models automatically use the appropriate thinking format. Use
|
|
132
149
|
Shift+Tab to control thinking level.
|
|
133
150
|
|
|
@@ -173,7 +190,7 @@ Add to your pi configuration for automatic loading:
|
|
|
173
190
|
|
|
174
191
|
Lilac's API is OpenAI-compatible with these specifics:
|
|
175
192
|
|
|
176
|
-
- **`thinkingFormat: "chat-template"`** — All reasoning models. Lilac's vLLM backend toggles reasoning via `chat_template_kwargs`, but the honored key differs per model family. Per-model `chatTemplateKwargs` in `patch.json` send the right key(s): `thinking`+`enable_thinking` (bool) for Kimi K2.6, GLM 5.1, Gemma 4, and MiniMax M2.7; `enable_thinking` + `reasoning_effort` for GLM 5.2; `thinking_mode` (adaptive|enabled|disabled) for MiniMax M3.
|
|
193
|
+
- **`thinkingFormat: "chat-template"`** — All reasoning models. Lilac's vLLM backend toggles reasoning via `chat_template_kwargs`, but the honored key differs per model family. Per-model `chatTemplateKwargs` in `patch.json` send the right key(s): `thinking`+`enable_thinking` (bool) for Kimi K2.6, GLM 5.1, Gemma 4, and MiniMax M2.7; `enable_thinking` + `reasoning_effort` for GLM 5.2; `thinking_mode` (adaptive|enabled|disabled) for MiniMax M3. Kimi K2.6, GLM 5.1, and GLM 5.2 additionally send a preservation flag (`preserve_thinking: true` / `clear_thinking: false`) to retain full reasoning history across turns — see [Preserved thinking](#thinking-mode) above.
|
|
177
194
|
- **`maxTokensField: "max_completion_tokens"`** — All models. Lilac supports `max_completion_tokens` (preferred for reasoning models as it includes reasoning tokens).
|
|
178
195
|
- **`supportsDeveloperRole: true`** — All models. Lilac's vLLM backend maps the developer role to system.
|
|
179
196
|
- **`supportsStore: false`** — All models. Lilac doesn't support the `store` parameter.
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "pi-lilac-provider",
|
|
3
|
-
"version": "1.
|
|
3
|
+
"version": "1.5.0",
|
|
4
4
|
"description": "Lilac provider extension for pi - Access Kimi K2.6, GLM 5.1, and Gemma 4 models through Lilac's OpenAI-compatible API on idle GPUs",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"main": "index.ts",
|
|
@@ -35,6 +35,7 @@
|
|
|
35
35
|
"build": "echo 'nothing to build'",
|
|
36
36
|
"check": "echo 'nothing to check'",
|
|
37
37
|
"test": "node scripts/test-discounts.ts",
|
|
38
|
+
"test:thinking": "node scripts/test-preserved-thinking.ts",
|
|
38
39
|
"update-models": "node scripts/update-models.js"
|
|
39
40
|
}
|
|
40
41
|
}
|
package/patch.json
CHANGED
|
@@ -10,7 +10,8 @@
|
|
|
10
10
|
"thinkingFormat": "chat-template",
|
|
11
11
|
"chatTemplateKwargs": {
|
|
12
12
|
"thinking": { "$var": "thinking.enabled" },
|
|
13
|
-
"enable_thinking": { "$var": "thinking.enabled" }
|
|
13
|
+
"enable_thinking": { "$var": "thinking.enabled" },
|
|
14
|
+
"preserve_thinking": true
|
|
14
15
|
},
|
|
15
16
|
"maxTokensField": "max_completion_tokens",
|
|
16
17
|
"supportsDeveloperRole": false,
|
|
@@ -29,7 +30,8 @@
|
|
|
29
30
|
"thinkingFormat": "chat-template",
|
|
30
31
|
"chatTemplateKwargs": {
|
|
31
32
|
"thinking": { "$var": "thinking.enabled" },
|
|
32
|
-
"enable_thinking": { "$var": "thinking.enabled" }
|
|
33
|
+
"enable_thinking": { "$var": "thinking.enabled" },
|
|
34
|
+
"clear_thinking": false
|
|
33
35
|
},
|
|
34
36
|
"maxTokensField": "max_completion_tokens",
|
|
35
37
|
"supportsDeveloperRole": false,
|
|
@@ -42,7 +44,8 @@
|
|
|
42
44
|
"thinkingFormat": "chat-template",
|
|
43
45
|
"chatTemplateKwargs": {
|
|
44
46
|
"enable_thinking": { "$var": "thinking.enabled" },
|
|
45
|
-
"reasoning_effort": { "$var": "thinking.effort", "omitWhenOff": true }
|
|
47
|
+
"reasoning_effort": { "$var": "thinking.effort", "omitWhenOff": true },
|
|
48
|
+
"clear_thinking": false
|
|
46
49
|
}
|
|
47
50
|
},
|
|
48
51
|
"thinkingLevelMap": {
|
|
@@ -17,7 +17,7 @@
|
|
|
17
17
|
* recomputed from list price so re-applied discounts never compound.
|
|
18
18
|
* 11. before_provider_request mutates the bound (in-flight) model object so the
|
|
19
19
|
* current turn's cost calc sees the discount in real time.
|
|
20
|
-
* 12. session_start schedules a
|
|
20
|
+
* 12. session_start schedules a 5-minute background /status poll to cover idle
|
|
21
21
|
* sessions; session_shutdown clears it.
|
|
22
22
|
*/
|
|
23
23
|
|
|
@@ -700,12 +700,12 @@ assert(
|
|
|
700
700
|
"deferred session_start paints the NEW lilac model's discount (glm), not the stale kimi capture",
|
|
701
701
|
);
|
|
702
702
|
|
|
703
|
-
// ─── Test 17: session_start schedules a
|
|
703
|
+
// ─── Test 17: session_start schedules a 5-min idle poll; shutdown clears it ─
|
|
704
704
|
|
|
705
705
|
console.log("\n--- Test 17: session_start schedules idle poll; session_shutdown clears it ---");
|
|
706
706
|
|
|
707
|
-
// Lilac refreshes discounts ~every 10 minutes
|
|
708
|
-
//
|
|
707
|
+
// Lilac refreshes discounts ~every 10 minutes, so session_start polls /status
|
|
708
|
+
// every 5 min (half the refresh window) to cover idle sessions (turn fetches
|
|
709
709
|
// only run when the user sends a message), and session_shutdown must clear it so
|
|
710
710
|
// it neither leaks nor keeps the process alive. Wrap the global timer APIs to
|
|
711
711
|
// capture the scheduled delay + handle, then confirm shutdown clears it.
|
|
@@ -729,7 +729,7 @@ globalThis.clearInterval = ((handle: ReturnType<typeof setInterval>) => {
|
|
|
729
729
|
|
|
730
730
|
try {
|
|
731
731
|
// Benign fetch mock so the session_start fire-and-forget /models + /status
|
|
732
|
-
// fetch doesn't hit the network. (The poll itself never fires —
|
|
732
|
+
// fetch doesn't hit the network. (The poll itself never fires — 5 min — so
|
|
733
733
|
// only the startup fetch needs mocking here.)
|
|
734
734
|
globalThis.fetch = mockFetch({
|
|
735
735
|
"/models": { body: { data: [] } },
|
|
@@ -759,8 +759,8 @@ try {
|
|
|
759
759
|
}
|
|
760
760
|
|
|
761
761
|
assert(
|
|
762
|
-
scheduledDelay ===
|
|
763
|
-
"session_start schedules a
|
|
762
|
+
scheduledDelay === 5 * 60 * 1000,
|
|
763
|
+
"session_start schedules a 5-minute (300000ms) /status poll for idle sessions",
|
|
764
764
|
);
|
|
765
765
|
assert(scheduledHandle !== null, "poll interval handle was captured");
|
|
766
766
|
const capturedHandle = scheduledHandle;
|
|
@@ -0,0 +1,237 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
/**
|
|
3
|
+
* Wire-level test for preserved-thinking (full-history reasoning) flags.
|
|
4
|
+
*
|
|
5
|
+
* Verifies, against the REAL pi-ai streamSimple + a stubbed fetch, that the
|
|
6
|
+
* `chat_template_kwargs` each Lilac reasoning model puts on the wire include the
|
|
7
|
+
* preservation flags configured in patch.json — and that those static flags
|
|
8
|
+
* coexist with the { $var }-resolved thinking keys at every thinking level.
|
|
9
|
+
*
|
|
10
|
+
* Background: reasoning models trim older assistant reasoning across turns by
|
|
11
|
+
* default (each vendor's template default). Two template-level flags opt into
|
|
12
|
+
* full-history preservation (confirmed in each model's HuggingFace chat
|
|
13
|
+
* template; Kimi K2.6 and GLM 5.2 additionally E2E-verified on the sibling
|
|
14
|
+
* neuralwatt provider's 3-turn / two-20-digit-number recall test):
|
|
15
|
+
*
|
|
16
|
+
* Kimi K2.6 → preserve_thinking: true (template: "if preserve_thinking, keep -1 ... retain reasoning")
|
|
17
|
+
* GLM 5.1 → clear_thinking: false (same clear_thinking mechanism as GLM 5.2)
|
|
18
|
+
* GLM 5.2 → clear_thinking: false (template: keep reasoning when "clear_thinking is defined and not clear_thinking")
|
|
19
|
+
*
|
|
20
|
+
* Lilac uses pi-ai's `chat-template` thinkingFormat, which calls
|
|
21
|
+
* buildChatTemplateKwargs → resolveChatTemplateKwargValue. Static primitive
|
|
22
|
+
* kwargs pass through verbatim; only { $var } objects are resolved against the
|
|
23
|
+
* turn's thinking state. So the preserve flags ride onto the wire as plain
|
|
24
|
+
* booleans next to the { $var } thinking/enable_thinking keys — no onPayload
|
|
25
|
+
* hook needed (unlike neuralwatt, which drives the openai reasoning_effort path
|
|
26
|
+
* and must inject via onPayload because the two paths are mutually exclusive).
|
|
27
|
+
*
|
|
28
|
+
* Gemma 4 and MiniMax M2.7/M3 expose NO family-wide preserve flag (their HF
|
|
29
|
+
* templates read only enable_thinking / thinking_mode + reasoning_content), so
|
|
30
|
+
* no flag is added for them here — asserted as a regression guard.
|
|
31
|
+
*
|
|
32
|
+
* Run: node scripts/test-preserved-thinking.ts
|
|
33
|
+
*/
|
|
34
|
+
import fs from "fs";
|
|
35
|
+
import path from "path";
|
|
36
|
+
import { pathToFileURL } from "url";
|
|
37
|
+
|
|
38
|
+
const HERE = path.dirname(new URL(import.meta.url).pathname);
|
|
39
|
+
|
|
40
|
+
// ─── Faithful replica of index.ts applyPatch / buildModels ────────────────────
|
|
41
|
+
// Mirrors the provider's model pipeline so the test exercises the same objects
|
|
42
|
+
// the runtime registers. (Kept inline so the test has no TS-import dependency
|
|
43
|
+
// on index.ts, which pulls in pi-coding-agent types.)
|
|
44
|
+
|
|
45
|
+
function applyPatch(model, patch) {
|
|
46
|
+
const result = { ...model };
|
|
47
|
+
if (patch.name !== undefined) result.name = patch.name;
|
|
48
|
+
if (patch.reasoning !== undefined) result.reasoning = patch.reasoning;
|
|
49
|
+
if (patch.input !== undefined) result.input = patch.input;
|
|
50
|
+
if (patch.contextWindow !== undefined) result.contextWindow = patch.contextWindow;
|
|
51
|
+
if (patch.maxTokens !== undefined) result.maxTokens = patch.maxTokens;
|
|
52
|
+
if (patch.cost) {
|
|
53
|
+
result.cost = {
|
|
54
|
+
input: patch.cost.input ?? result.cost.input,
|
|
55
|
+
output: patch.cost.output ?? result.cost.output,
|
|
56
|
+
cacheRead: patch.cost.cacheRead ?? result.cost.cacheRead,
|
|
57
|
+
cacheWrite: patch.cost.cacheWrite ?? result.cost.cacheWrite,
|
|
58
|
+
};
|
|
59
|
+
}
|
|
60
|
+
if (patch.compat) result.compat = { ...(result.compat || {}), ...patch.compat };
|
|
61
|
+
if (patch.thinkingLevelMap !== undefined) result.thinkingLevelMap = patch.thinkingLevelMap;
|
|
62
|
+
if (!result.reasoning && result.compat?.thinkingFormat) delete result.compat.thinkingFormat;
|
|
63
|
+
if (!result.reasoning && result.thinkingLevelMap) delete result.thinkingLevelMap;
|
|
64
|
+
if (result.compat && Object.keys(result.compat).length === 0) delete result.compat;
|
|
65
|
+
return result;
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
function buildModels(base, custom, patch) {
|
|
69
|
+
const map = new Map();
|
|
70
|
+
for (const m of base) map.set(m.id, m);
|
|
71
|
+
for (const [id, p] of Object.entries(patch)) {
|
|
72
|
+
const ex = map.get(id);
|
|
73
|
+
if (ex) map.set(id, applyPatch(ex, p));
|
|
74
|
+
}
|
|
75
|
+
for (const m of custom) {
|
|
76
|
+
const ex = map.get(m.id);
|
|
77
|
+
const p = patch[m.id];
|
|
78
|
+
if (ex && p) map.set(m.id, applyPatch(m, p));
|
|
79
|
+
else if (ex) map.set(m.id, m);
|
|
80
|
+
else if (p) map.set(m.id, applyPatch(m, p));
|
|
81
|
+
else map.set(m.id, m);
|
|
82
|
+
}
|
|
83
|
+
return Array.from(map.values());
|
|
84
|
+
}
|
|
85
|
+
|
|
86
|
+
// ─── Load REAL pi-ai streamSimple from the global pi install ──────────────────
|
|
87
|
+
// Imported by absolute path so its relative deps (openai, ../models.js, ...) and
|
|
88
|
+
// the `openai` SDK resolve from the global pi node_modules tree.
|
|
89
|
+
const PI_AI_API = path.join(
|
|
90
|
+
os_home_pi_ai(),
|
|
91
|
+
"dist/api/openai-completions.js",
|
|
92
|
+
);
|
|
93
|
+
function os_home_pi_ai() {
|
|
94
|
+
// Resolve the pi-ai package shipped inside the globally-installed pi agent.
|
|
95
|
+
const candidates = [
|
|
96
|
+
"/Users/monotykamary/.npm-global/lib/node_modules/@earendil-works/pi-coding-agent/node_modules/@earendil-works/pi-ai",
|
|
97
|
+
];
|
|
98
|
+
for (const c of candidates) if (fs.existsSync(c)) return c;
|
|
99
|
+
throw new Error("Could not locate pi-ai package in the global pi install.");
|
|
100
|
+
}
|
|
101
|
+
|
|
102
|
+
const { streamSimple } = await import(pathToFileURL(PI_AI_API).href);
|
|
103
|
+
|
|
104
|
+
// ─── Build the exact model objects the provider registers ────────────────────
|
|
105
|
+
const embedded = JSON.parse(fs.readFileSync(path.join(HERE, "..", "models.json"), "utf8"));
|
|
106
|
+
const custom = JSON.parse(fs.readFileSync(path.join(HERE, "..", "custom-models.json"), "utf8"));
|
|
107
|
+
const patch = JSON.parse(fs.readFileSync(path.join(HERE, "..", "patch.json"), "utf8"));
|
|
108
|
+
const models = new Map(buildModels(embedded, custom, patch).map((m) => [m.id, m]));
|
|
109
|
+
|
|
110
|
+
// ─── Stub fetch to capture the request body (fires before response parsing) ──
|
|
111
|
+
const captured = [];
|
|
112
|
+
const originalFetch = globalThis.fetch;
|
|
113
|
+
globalThis.fetch = async (url, init) => {
|
|
114
|
+
captured.push({ url: String(url), body: init?.body ?? null });
|
|
115
|
+
// Minimal SSE response that ends immediately — enough for the OpenAI SDK to
|
|
116
|
+
// construct the stream; we only need the request body, captured above.
|
|
117
|
+
return new Response(new ReadableStream({ start(c) { c.close(); } }), {
|
|
118
|
+
headers: { "content-type": "text/event-stream" },
|
|
119
|
+
});
|
|
120
|
+
};
|
|
121
|
+
|
|
122
|
+
async function wire(modelId, reasoning) {
|
|
123
|
+
const model = models.get(modelId);
|
|
124
|
+
if (!model) throw new Error(`unknown model: ${modelId}`);
|
|
125
|
+
captured.length = 0;
|
|
126
|
+
const ctx = { messages: [{ role: "user", content: [{ type: "text", text: "hi" }] }] };
|
|
127
|
+
const s = streamSimple(
|
|
128
|
+
{ ...model, provider: "lilac", api: "openai-completions", baseUrl: "https://api.getlilac.com/v1" },
|
|
129
|
+
ctx,
|
|
130
|
+
{ apiKey: "sk-test", reasoning },
|
|
131
|
+
);
|
|
132
|
+
// The fetch fires inside stream()'s async IIFE; poll for it.
|
|
133
|
+
const deadline = Date.now() + 2000;
|
|
134
|
+
while (captured.length === 0 && Date.now() < deadline) await new Promise((r) => setTimeout(r, 20));
|
|
135
|
+
try { s.end?.(); } catch {}
|
|
136
|
+
const hit = captured.find((c) => c.url.includes("/chat/completions"));
|
|
137
|
+
if (!hit) throw new Error(`no /chat/completions request captured for ${modelId} (${reasoning})`);
|
|
138
|
+
return JSON.parse(hit.body);
|
|
139
|
+
}
|
|
140
|
+
|
|
141
|
+
// ─── Assertions ───────────────────────────────────────────────────────────────
|
|
142
|
+
let failures = 0;
|
|
143
|
+
function eq(actual, expected, msg) {
|
|
144
|
+
const a = JSON.stringify(actual);
|
|
145
|
+
const e = JSON.stringify(expected);
|
|
146
|
+
const ok = a === e;
|
|
147
|
+
console.log(`${ok ? "✓" : "✗"} ${msg}`);
|
|
148
|
+
if (!ok) {
|
|
149
|
+
failures++;
|
|
150
|
+
console.log(` expected ${e}`);
|
|
151
|
+
console.log(` actual ${a}`);
|
|
152
|
+
}
|
|
153
|
+
}
|
|
154
|
+
function truthy(actual, msg) {
|
|
155
|
+
const ok = !!actual;
|
|
156
|
+
console.log(`${ok ? "✓" : "✗"} ${msg}`);
|
|
157
|
+
if (!ok) {
|
|
158
|
+
failures++;
|
|
159
|
+
console.log(` expected truthy, got ${JSON.stringify(actual)}`);
|
|
160
|
+
}
|
|
161
|
+
}
|
|
162
|
+
function falsy(actual, msg) {
|
|
163
|
+
const ok = !actual;
|
|
164
|
+
console.log(`${ok ? "✓" : "✗"} ${msg}`);
|
|
165
|
+
if (!ok) {
|
|
166
|
+
failures++;
|
|
167
|
+
console.log(` expected falsy/undefined, got ${JSON.stringify(actual)}`);
|
|
168
|
+
}
|
|
169
|
+
}
|
|
170
|
+
|
|
171
|
+
console.log("\n=== patch.json data ===");
|
|
172
|
+
const kimiPatch = patch["moonshotai/kimi-k2.6"]?.compat?.chatTemplateKwargs;
|
|
173
|
+
eq(kimiPatch?.preserve_thinking, true, "kimi-k2.6 patch sets preserve_thinking: true");
|
|
174
|
+
const glm51Patch = patch["zai-org/glm-5.1"]?.compat?.chatTemplateKwargs;
|
|
175
|
+
eq(glm51Patch?.clear_thinking, false, "glm-5.1 patch sets clear_thinking: false");
|
|
176
|
+
const glm52Patch = patch["zai-org/glm-5.2"]?.compat?.chatTemplateKwargs;
|
|
177
|
+
eq(glm52Patch?.clear_thinking, false, "glm-5.2 patch sets clear_thinking: false");
|
|
178
|
+
|
|
179
|
+
console.log("\n=== Kimi K2.6 on the wire (real pi-ai) ===");
|
|
180
|
+
{
|
|
181
|
+
const high = await wire("moonshotai/kimi-k2.6", "high");
|
|
182
|
+
eq(high.chat_template_kwargs, { thinking: true, enable_thinking: true, preserve_thinking: true },
|
|
183
|
+
"kimi @ high → thinking+enable_thinking true AND preserve_thinking true");
|
|
184
|
+
const off = await wire("moonshotai/kimi-k2.6", "off");
|
|
185
|
+
eq(off.chat_template_kwargs, { thinking: false, enable_thinking: false, preserve_thinking: true },
|
|
186
|
+
"kimi @ off → thinking false but preserve_thinking still true (level-independent)");
|
|
187
|
+
const minimal = await wire("moonshotai/kimi-k2.6", "minimal");
|
|
188
|
+
eq(minimal.chat_template_kwargs, { thinking: true, enable_thinking: true, preserve_thinking: true },
|
|
189
|
+
"kimi @ minimal → preserve_thinking present at every level");
|
|
190
|
+
}
|
|
191
|
+
|
|
192
|
+
console.log("\n=== GLM 5.2 on the wire (real pi-ai) ===");
|
|
193
|
+
{
|
|
194
|
+
const high = await wire("zai-org/glm-5.2", "high");
|
|
195
|
+
eq(high.chat_template_kwargs, { enable_thinking: true, reasoning_effort: "high", clear_thinking: false },
|
|
196
|
+
"glm-5.2 @ high → enable_thinking+reasoning_effort AND clear_thinking false");
|
|
197
|
+
const xhigh = await wire("zai-org/glm-5.2", "xhigh");
|
|
198
|
+
eq(xhigh.chat_template_kwargs, { enable_thinking: true, reasoning_effort: "max", clear_thinking: false },
|
|
199
|
+
"glm-5.2 @ xhigh → reasoning_effort max, clear_thinking false");
|
|
200
|
+
const off = await wire("zai-org/glm-5.2", "off");
|
|
201
|
+
eq(off.chat_template_kwargs, { enable_thinking: false, clear_thinking: false },
|
|
202
|
+
"glm-5.2 @ off → reasoning_effort omitted (omitWhenOff), clear_thinking false persists");
|
|
203
|
+
}
|
|
204
|
+
|
|
205
|
+
console.log("\n=== GLM 5.1 on the wire (real pi-ai) ===");
|
|
206
|
+
{
|
|
207
|
+
const high = await wire("zai-org/glm-5.1", "high");
|
|
208
|
+
eq(high.chat_template_kwargs, { thinking: true, enable_thinking: true, clear_thinking: false },
|
|
209
|
+
"glm-5.1 @ high → thinking+enable_thinking true AND clear_thinking false");
|
|
210
|
+
const off = await wire("zai-org/glm-5.1", "off");
|
|
211
|
+
eq(off.chat_template_kwargs, { thinking: false, enable_thinking: false, clear_thinking: false },
|
|
212
|
+
"glm-5.1 @ off → thinking false but clear_thinking false persists");
|
|
213
|
+
}
|
|
214
|
+
|
|
215
|
+
console.log("\n=== Gemma 4 / MiniMax (no family-wide preserve flag — regression guard) ===");
|
|
216
|
+
{
|
|
217
|
+
const gemma = await wire("google/gemma-4-31b-it", "high");
|
|
218
|
+
eq(gemma.chat_template_kwargs, { thinking: true, enable_thinking: true },
|
|
219
|
+
"gemma-4 @ high → only thinking/enable_thinking (no preserve/clear flag)");
|
|
220
|
+
falsy(gemma.chat_template_kwargs?.preserve_thinking, "gemma-4 has no preserve_thinking");
|
|
221
|
+
falsy(gemma.chat_template_kwargs?.clear_thinking, "gemma-4 has no clear_thinking");
|
|
222
|
+
|
|
223
|
+
const m3 = await wire("minimaxai/minimax-m3", "high");
|
|
224
|
+
eq(m3.chat_template_kwargs, { thinking_mode: "enabled" },
|
|
225
|
+
"minimax-m3 @ high → only thinking_mode (no preserve/clear flag)");
|
|
226
|
+
falsy(m3.chat_template_kwargs?.preserve_thinking, "minimax-m3 has no preserve_thinking");
|
|
227
|
+
|
|
228
|
+
const m27 = await wire("minimaxai/minimax-m2.7", "high");
|
|
229
|
+
eq(m27.chat_template_kwargs, { thinking: true, enable_thinking: true },
|
|
230
|
+
"minimax-m2.7 @ high → only thinking/enable_thinking (no preserve/clear flag)");
|
|
231
|
+
falsy(m27.chat_template_kwargs?.clear_thinking, "minimax-m2.7 has no clear_thinking");
|
|
232
|
+
}
|
|
233
|
+
|
|
234
|
+
globalThis.fetch = originalFetch;
|
|
235
|
+
|
|
236
|
+
console.log(`\n${failures === 0 ? "ALL PASS" : `${failures} FAILURE(S)`}`);
|
|
237
|
+
if (failures > 0) process.exit(1);
|