pi-makora-provider 1.4.0 → 1.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +38 -0
- package/death-loop-guard.ts +398 -0
- package/index.ts +11 -0
- package/models.json +21 -0
- package/package.json +4 -1
package/README.md
CHANGED
|
@@ -20,6 +20,7 @@ _DeepSeek V4, Kimi K2.7 Code, GLM 5.2, Qwen 3.6 for [pi](https://github.com/eare
|
|
|
20
20
|
|-------|----|-----------|-------|
|
|
21
21
|
| DeepSeek V4 Flash | `deepseek-ai/DeepSeek-V4-Flash` | Yes | returns `reasoning` field |
|
|
22
22
|
| DeepSeek V4 Pro | `deepseek-ai/DeepSeek-V4-Pro` | Yes | returns `reasoning` field |
|
|
23
|
+
| Gemma 4 26B A4B | `google/gemma-4-26B-A4B` | No | |
|
|
23
24
|
| GLM 5.2 FP8 | `zai-org/GLM-5.2-FP8` | Yes | `enable_thinking` via `qwen-chat-template`; effort via `reasoning_effort` (only `high`/`max` distinct, per vLLM GLM-5.2 recipe); thinking levels aligned with neuralwatt GLM 5.2; returns `reasoning` field |
|
|
24
25
|
| GLM 5.2 NVFP4 | `zai-org/GLM-5.2-NVFP4` | Yes | `enable_thinking` via `qwen-chat-template`; effort via `reasoning_effort` (only `high`/`max` distinct, per vLLM GLM-5.2 recipe); returns `reasoning` field |
|
|
25
26
|
| Kimi K2.7 Code | `moonshotai/Kimi-K2.7-Code` | Yes | Reasoning on by default (thinking-only model); `preserve_thinking` via `chatTemplateKwargs` for multi-turn continuity; returns `reasoning` field |
|
|
@@ -156,3 +157,40 @@ Do **not** edit `models.json` directly — it is auto-generated from the API. To
|
|
|
156
157
|
- The API is OpenAI-compatible (chat completions format)
|
|
157
158
|
- All models are hosted on vLLM
|
|
158
159
|
- The `developer` role is not supported (prompts are silently dropped); `supportsDeveloperRole` is set to `false` for all models
|
|
160
|
+
|
|
161
|
+
## Death-Loop Guard
|
|
162
|
+
|
|
163
|
+
GLM 5.2 (NVFP4 / FP8) occasionally degenerates into an unbroken `!` repetition
|
|
164
|
+
loop (`!!!!...`) that eats the whole response. This extension ships a guard
|
|
165
|
+
that watches the streamed assistant output (both the visible answer and the
|
|
166
|
+
reasoning trace) and, when it detects a long run of `!` characters, **aborts
|
|
167
|
+
the runaway generation and resumes the agentic loop invisibly** — no new user
|
|
168
|
+
message is injected, using the
|
|
169
|
+
[pi-invisible-continue](https://github.com/monotykamary/pi-invisible-continue)
|
|
170
|
+
pattern (`agent.prompt([])`).
|
|
171
|
+
|
|
172
|
+
On abort, pi finalizes the in-flight assistant message **with its accumulated
|
|
173
|
+
`!!!` content** into the transcript, so the guard drops that partial message
|
|
174
|
+
before resuming — otherwise the model would just re-read its own `!!!` and loop
|
|
175
|
+
again. Recovery retries **indefinitely with exponential backoff (2s→60s,
|
|
176
|
+
2×)**, like [pi-retry](https://github.com/monotykamary/pi-retry) — long-horizon
|
|
177
|
+
agent work can trip the loop many times in a session, so a retry cap would
|
|
178
|
+
strand the agent mid-task. The loop exits only on a clean turn, a user abort
|
|
179
|
+
(Esc), or a session change (`/new`, `/resume`). Backoff is interruptible
|
|
180
|
+
(polls every 100ms) so Esc and `/new` take effect within 100ms instead of
|
|
181
|
+
waiting out the full delay.
|
|
182
|
+
|
|
183
|
+
The guard is scoped to the Makora GLM 5.2 family by default and is tunable via
|
|
184
|
+
constants at the top of [`death-loop-guard.ts`](./death-loop-guard.ts):
|
|
185
|
+
|
|
186
|
+
| Constant | Default | Meaning |
|
|
187
|
+
|---|---|---|
|
|
188
|
+
| `GUARDED_MODEL_IDS` | `zai-org/GLM-5.2-NVFP4`, `zai-org/GLM-5.2-FP8` | Which model IDs to guard. Add `'*'` to guard every Makora model. |
|
|
189
|
+
| `BANG_THRESHOLD` | `40` | Consecutive `!` characters that trip the guard. 40 is far above anything normal prose/code produces. |
|
|
190
|
+
| `BACKOFF_BASE_MS` | `2000` | Initial recovery backoff delay. |
|
|
191
|
+
| `BACKOFF_MAX_MS` | `60000` | Maximum backoff delay (cap). |
|
|
192
|
+
| `BACKOFF_MULTIPLIER` | `2` | Backoff growth factor per retry. |
|
|
193
|
+
|
|
194
|
+
It is only active for the `makora` provider, so it never interferes when you
|
|
195
|
+
switch to another provider. If you also run `pi-invisible-continue`, the two
|
|
196
|
+
coexist — both chain the `Agent.prototype.subscribe` patch.
|
|
@@ -0,0 +1,398 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Death-loop guard for Makora reasoning models.
|
|
3
|
+
*
|
|
4
|
+
* Some Makora models (notably GLM 5.2 NVFP4 / FP8) occasionally fall into a
|
|
5
|
+
* degenerate repetition loop, emitting an unbroken run of '!' characters
|
|
6
|
+
* (e.g. "!!!!...") that consumes the whole response. This guard watches the
|
|
7
|
+
* streamed assistant output (both the visible answer and the reasoning
|
|
8
|
+
* trace); when that run is detected it aborts the runaway generation, drops
|
|
9
|
+
* the partial (toxic) assistant message from the transcript, and resumes the
|
|
10
|
+
* agentic loop invisibly via agent.prompt([]) — the same pattern
|
|
11
|
+
* pi-invisible-continue / pi-retry uses, so no new user message pollutes the
|
|
12
|
+
* context.
|
|
13
|
+
*
|
|
14
|
+
* Infinite retries with exponential backoff. Long-horizon agent work can trip
|
|
15
|
+
* the loop many times in one session; capping retries would strand the agent
|
|
16
|
+
* mid-task. Instead the recovery loops indefinitely (like pi-retry) until the
|
|
17
|
+
* model produces a clean turn, the user aborts (Esc), or the session changes
|
|
18
|
+
* (/new, /resume). Backoff (2s→60s, 2×) paces retries and gives the user a
|
|
19
|
+
* window to intervene; interruptible sleep polls every 100ms so Esc and /new
|
|
20
|
+
* take effect within 100ms instead of waiting out the full delay.
|
|
21
|
+
*
|
|
22
|
+
* Why trim the aborted message: on abort, pi finalizes the in-flight
|
|
23
|
+
* assistant message WITH its accumulated '!!!' content and stopReason
|
|
24
|
+
* "aborted" into the transcript. Resuming from that context would re-feed
|
|
25
|
+
* the toxic text to the model and likely re-trigger the loop. Dropping the
|
|
26
|
+
* last (aborted) assistant message leaves the context ending at the prior
|
|
27
|
+
* user/toolResult message — a clean continuation point.
|
|
28
|
+
*
|
|
29
|
+
* Distinguishing our abort from a user Esc: both call agent.abort(), so both
|
|
30
|
+
* surface as stopReason "aborted" at turn_end. The handler sets _weAborted
|
|
31
|
+
* before aborting; turn_end only treats an abort as user-initiated (setting
|
|
32
|
+
* _userAborted to exit the loop) when _weAborted is false. Caveat: if the
|
|
33
|
+
* user hits Esc in the ~ms between our trip and the abort completing, the
|
|
34
|
+
* abort is mis-attributed to us and one unwanted retry fires. The backoff
|
|
35
|
+
* window lets the user Esc again, which then works normally.
|
|
36
|
+
*
|
|
37
|
+
* Module resolution: @earendil-works/pi-agent-core is a devDependency only
|
|
38
|
+
* (types + test resolution). At runtime pi's extension loader aliases that
|
|
39
|
+
* specifier to its bundled copy, so the Agent class patched below is the
|
|
40
|
+
* SAME class AgentSession uses. A static import is required — jiti's alias
|
|
41
|
+
* applies to static imports (which it rewrites to its own resolver) but not
|
|
42
|
+
* to native dynamic import() calls.
|
|
43
|
+
*/
|
|
44
|
+
|
|
45
|
+
import type { ExtensionAPI } from "@earendil-works/pi-coding-agent";
|
|
46
|
+
import { Agent } from "@earendil-works/pi-agent-core";
|
|
47
|
+
|
|
48
|
+
const PROVIDER_ID = "makora";
|
|
49
|
+
|
|
50
|
+
/** Makora model IDs to guard. Add ids to widen coverage, or include "*"
|
|
51
|
+
* to guard every Makora model. Defaults to the GLM 5.2 family (the known
|
|
52
|
+
* offender). Kept exported so tests and downstream forks can introspect. */
|
|
53
|
+
export const GUARDED_MODEL_IDS = new Set<string>([
|
|
54
|
+
"zai-org/GLM-5.2-NVFP4",
|
|
55
|
+
"zai-org/GLM-5.2-FP8",
|
|
56
|
+
]);
|
|
57
|
+
|
|
58
|
+
/** Trip after this many consecutive '!' characters in the streamed answer.
|
|
59
|
+
* 40 is far above anything normal prose or code produces. */
|
|
60
|
+
export const BANG_THRESHOLD = 40;
|
|
61
|
+
|
|
62
|
+
/** Exponential backoff for recovery retries. Mirrors pi-retry's defaults:
|
|
63
|
+
* 2s base, 60s cap, 2× multiplier. Tunable via these constants. */
|
|
64
|
+
export const BACKOFF_BASE_MS = 2000;
|
|
65
|
+
export const BACKOFF_MAX_MS = 60_000;
|
|
66
|
+
export const BACKOFF_MULTIPLIER = 2;
|
|
67
|
+
|
|
68
|
+
export interface BackoffConfig {
|
|
69
|
+
baseDelayMs: number;
|
|
70
|
+
maxDelayMs: number;
|
|
71
|
+
multiplier: number;
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
export const DEFAULT_BACKOFF_CONFIG: BackoffConfig = {
|
|
75
|
+
baseDelayMs: BACKOFF_BASE_MS,
|
|
76
|
+
maxDelayMs: BACKOFF_MAX_MS,
|
|
77
|
+
multiplier: BACKOFF_MULTIPLIER,
|
|
78
|
+
};
|
|
79
|
+
|
|
80
|
+
export interface GuardedMessage {
|
|
81
|
+
role: string;
|
|
82
|
+
stopReason?: string;
|
|
83
|
+
content?: unknown;
|
|
84
|
+
}
|
|
85
|
+
|
|
86
|
+
export interface GuardedAgent {
|
|
87
|
+
abort(): void;
|
|
88
|
+
waitForIdle(): Promise<void>;
|
|
89
|
+
prompt(input: unknown[] | string): Promise<void>;
|
|
90
|
+
state: { messages: GuardedMessage[] };
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
let _agent: GuardedAgent | null = null;
|
|
94
|
+
|
|
95
|
+
/** Mutex: only one recovery loop may be in-flight at a time. The
|
|
96
|
+
* message_update handler still aborts mid-stream while this is held, but
|
|
97
|
+
* only the loop driver re-issues prompt([]). */
|
|
98
|
+
let _recovering = false;
|
|
99
|
+
|
|
100
|
+
/** Trailing '!' run length in the current text block. */
|
|
101
|
+
let _trailingBangs = 0;
|
|
102
|
+
|
|
103
|
+
/** Latch: already tripped for the current assistant message. */
|
|
104
|
+
let _tripped = false;
|
|
105
|
+
|
|
106
|
+
/** True when WE aborted the current turn (death-loop), false on user Esc.
|
|
107
|
+
* Reset on message_start of each new assistant turn. */
|
|
108
|
+
let _weAborted = false;
|
|
109
|
+
|
|
110
|
+
/** True when the user cancelled (Esc). Set in turn_end, cleared on
|
|
111
|
+
* session_start and fresh successful turns. Stops the recovery loop. */
|
|
112
|
+
let _userAborted = false;
|
|
113
|
+
|
|
114
|
+
/** Session generation counter: incremented on every session_start. The
|
|
115
|
+
* recovery loop captures it on entry and exits when it changes (/new,
|
|
116
|
+
* /resume), so a stale loop never drives a new session. */
|
|
117
|
+
let _sessionGeneration = 0;
|
|
118
|
+
|
|
119
|
+
/** notify() captured fresh from the most recent event ctx, since
|
|
120
|
+
* ctx.ui.notify isn't available inside the loop driver. Mirrors pi-retry. */
|
|
121
|
+
let _notifyFn:
|
|
122
|
+
| ((message: string, level: "info" | "warning" | "error") => void)
|
|
123
|
+
| null = null;
|
|
124
|
+
|
|
125
|
+
export function isGuardedModel(
|
|
126
|
+
model: { provider?: string; id?: string } | undefined | null,
|
|
127
|
+
): boolean {
|
|
128
|
+
if (!model) return false;
|
|
129
|
+
if (model.provider !== PROVIDER_ID) return false;
|
|
130
|
+
if (GUARDED_MODEL_IDS.has("*")) return true;
|
|
131
|
+
return model.id != null && GUARDED_MODEL_IDS.has(model.id);
|
|
132
|
+
}
|
|
133
|
+
|
|
134
|
+
/** Update the trailing-'!' run length given a new text delta. O(len(delta)).
|
|
135
|
+
* Trailing run depends only on the delta's suffix: if the delta contains any
|
|
136
|
+
* non-'!' char, the prior run is cut off at that char; if the delta is all
|
|
137
|
+
* '!', it extends the prior run. */
|
|
138
|
+
export function nextTrailingBangs(prev: number, delta: string): number {
|
|
139
|
+
const len = delta.length;
|
|
140
|
+
if (len === 0) return prev;
|
|
141
|
+
let i = len - 1;
|
|
142
|
+
while (i >= 0 && delta.charCodeAt(i) === 0x21) i--; // '!' === 0x21
|
|
143
|
+
const trailingInDelta = len - 1 - i;
|
|
144
|
+
return i >= 0 ? trailingInDelta : prev + len;
|
|
145
|
+
}
|
|
146
|
+
|
|
147
|
+
export function extractText(content: unknown): string {
|
|
148
|
+
if (typeof content === "string") return content;
|
|
149
|
+
if (Array.isArray(content)) {
|
|
150
|
+
let out = "";
|
|
151
|
+
for (const block of content) {
|
|
152
|
+
if (
|
|
153
|
+
block &&
|
|
154
|
+
typeof block === "object" &&
|
|
155
|
+
(block as { type?: string }).type === "text" &&
|
|
156
|
+
typeof (block as { text?: unknown }).text === "string"
|
|
157
|
+
) {
|
|
158
|
+
out += (block as { text: string }).text;
|
|
159
|
+
}
|
|
160
|
+
}
|
|
161
|
+
return out;
|
|
162
|
+
}
|
|
163
|
+
return "";
|
|
164
|
+
}
|
|
165
|
+
|
|
166
|
+
/** Trailing '!' run length of a finalized message's text content. */
|
|
167
|
+
export function messageTrailingBangs(msg: GuardedMessage): number {
|
|
168
|
+
const text = extractText(msg.content);
|
|
169
|
+
let i = text.length - 1;
|
|
170
|
+
while (i >= 0 && text.charCodeAt(i) === 0x21) i--;
|
|
171
|
+
return text.length - 1 - i;
|
|
172
|
+
}
|
|
173
|
+
|
|
174
|
+
/** Exponential backoff delay, capped at maxDelayMs. Mirrors pi-retry. */
|
|
175
|
+
export function calculateDelay(
|
|
176
|
+
attempt: number,
|
|
177
|
+
config: BackoffConfig = DEFAULT_BACKOFF_CONFIG,
|
|
178
|
+
): number {
|
|
179
|
+
const delay = config.baseDelayMs * Math.pow(config.multiplier, attempt - 1);
|
|
180
|
+
return Math.min(delay, config.maxDelayMs);
|
|
181
|
+
}
|
|
182
|
+
|
|
183
|
+
/** Format a duration for user-facing messages. Mirrors pi-retry. */
|
|
184
|
+
export function formatDuration(ms: number): string {
|
|
185
|
+
if (ms < 1000) return `${ms}ms`;
|
|
186
|
+
if (ms < 60000) return `${(ms / 1000).toFixed(1)}s`;
|
|
187
|
+
const minutes = Math.floor(ms / 60000);
|
|
188
|
+
const seconds = ((ms % 60000) / 1000).toFixed(0);
|
|
189
|
+
return `${minutes}m ${seconds}s`;
|
|
190
|
+
}
|
|
191
|
+
|
|
192
|
+
/** Interruptible sleep: polls _userAborted and _sessionGeneration every
|
|
193
|
+
* 100ms. Returns true if interrupted (Esc or /new), false if the full delay
|
|
194
|
+
* elapsed. Mirrors pi-retry. */
|
|
195
|
+
function interruptibleSleep(ms: number, generation: number): Promise<boolean> {
|
|
196
|
+
if (ms <= 0) return Promise.resolve(false);
|
|
197
|
+
return new Promise((resolve) => {
|
|
198
|
+
const checkInterval = 100;
|
|
199
|
+
let elapsed = 0;
|
|
200
|
+
const timer = setInterval(() => {
|
|
201
|
+
elapsed += checkInterval;
|
|
202
|
+
if (_userAborted || _sessionGeneration !== generation) {
|
|
203
|
+
clearInterval(timer);
|
|
204
|
+
resolve(true);
|
|
205
|
+
} else if (elapsed >= ms) {
|
|
206
|
+
clearInterval(timer);
|
|
207
|
+
resolve(false);
|
|
208
|
+
}
|
|
209
|
+
}, checkInterval);
|
|
210
|
+
});
|
|
211
|
+
}
|
|
212
|
+
|
|
213
|
+
/** Remove a trailing aborted/death-loop assistant message so the next
|
|
214
|
+
* prompt([]) sends a clean context ending at the prior user/toolResult. */
|
|
215
|
+
function trimAbortedDeathLoop(agent: GuardedAgent): void {
|
|
216
|
+
const msgs = agent.state.messages;
|
|
217
|
+
const last = msgs[msgs.length - 1];
|
|
218
|
+
if (
|
|
219
|
+
last &&
|
|
220
|
+
last.role === "assistant" &&
|
|
221
|
+
(last.stopReason === "aborted" ||
|
|
222
|
+
messageTrailingBangs(last) >= BANG_THRESHOLD)
|
|
223
|
+
) {
|
|
224
|
+
agent.state.messages = msgs.slice(0, -1);
|
|
225
|
+
}
|
|
226
|
+
}
|
|
227
|
+
|
|
228
|
+
/** Recovery loop driver — the core. Loops until a clean turn, user abort,
|
|
229
|
+
* or session change. Backoff sleep happens AFTER each prompt([]) settles,
|
|
230
|
+
* so it does not block the agent; Esc and /new take effect within 100ms. */
|
|
231
|
+
async function triggerRecovery(): Promise<void> {
|
|
232
|
+
if (!_agent) return;
|
|
233
|
+
if (_userAborted) return;
|
|
234
|
+
if (_recovering) return;
|
|
235
|
+
_recovering = true;
|
|
236
|
+
|
|
237
|
+
const myGeneration = _sessionGeneration;
|
|
238
|
+
|
|
239
|
+
try {
|
|
240
|
+
await _agent.waitForIdle();
|
|
241
|
+
if (_userAborted || _sessionGeneration !== myGeneration) return;
|
|
242
|
+
|
|
243
|
+
// Trim the aborted !!! message from the initial trip.
|
|
244
|
+
trimAbortedDeathLoop(_agent);
|
|
245
|
+
|
|
246
|
+
let attempt = 0;
|
|
247
|
+
// Loop until success, user abort, or session change.
|
|
248
|
+
while (true) {
|
|
249
|
+
if (_userAborted || _sessionGeneration !== myGeneration) return;
|
|
250
|
+
|
|
251
|
+
attempt++;
|
|
252
|
+
const delay = calculateDelay(attempt);
|
|
253
|
+
if (_notifyFn) {
|
|
254
|
+
_notifyFn(
|
|
255
|
+
`Makora death-loop guard: resuming after runaway '!' output ` +
|
|
256
|
+
`(retry ${attempt}, backoff ${formatDuration(delay)})...`,
|
|
257
|
+
"warning",
|
|
258
|
+
);
|
|
259
|
+
}
|
|
260
|
+
|
|
261
|
+
// Interruptible sleep with backoff BEFORE the retry. Lets Esc and
|
|
262
|
+
// /new take effect within 100ms instead of waiting the full delay.
|
|
263
|
+
const interrupted = await interruptibleSleep(delay, myGeneration);
|
|
264
|
+
if (interrupted) return;
|
|
265
|
+
if (_userAborted || _sessionGeneration !== myGeneration) return;
|
|
266
|
+
|
|
267
|
+
// _weAborted is also reset on message_start, but reset here too in
|
|
268
|
+
// case message_start already fired before we reached this point.
|
|
269
|
+
_weAborted = false;
|
|
270
|
+
try {
|
|
271
|
+
await _agent.prompt([]);
|
|
272
|
+
} catch {
|
|
273
|
+
// "Agent is already processing" or other transient error — bail.
|
|
274
|
+
return;
|
|
275
|
+
}
|
|
276
|
+
if (_userAborted || _sessionGeneration !== myGeneration) return;
|
|
277
|
+
|
|
278
|
+
// Did the resumed turn death-loop again? If we aborted it, trim and
|
|
279
|
+
// loop; otherwise it completed cleanly (or the user aborted) — exit.
|
|
280
|
+
if (_weAborted) {
|
|
281
|
+
trimAbortedDeathLoop(_agent);
|
|
282
|
+
continue;
|
|
283
|
+
}
|
|
284
|
+
return;
|
|
285
|
+
}
|
|
286
|
+
} finally {
|
|
287
|
+
// Only release the mutex if the session hasn't changed. If /new fired,
|
|
288
|
+
// a new recovery loop may already own it — resetting here would clobber.
|
|
289
|
+
if (_sessionGeneration === myGeneration) {
|
|
290
|
+
_recovering = false;
|
|
291
|
+
}
|
|
292
|
+
}
|
|
293
|
+
}
|
|
294
|
+
|
|
295
|
+
export function registerDeathLoopGuard(pi: ExtensionAPI): void {
|
|
296
|
+
// Capture the live Agent instance by chaining Agent.prototype.subscribe.
|
|
297
|
+
// subscribe() fires when AgentSession attaches — on every fresh session
|
|
298
|
+
// and every resume — so _agent always points at the active Agent. Chain
|
|
299
|
+
// any prior patch (e.g. pi-invisible-continue, pi-retry) so all coexist.
|
|
300
|
+
const proto = Agent.prototype as unknown as {
|
|
301
|
+
subscribe: (this: GuardedAgent, ...args: unknown[]) => unknown;
|
|
302
|
+
};
|
|
303
|
+
const origSubscribe = proto.subscribe;
|
|
304
|
+
proto.subscribe = function (this: GuardedAgent, ...args: unknown[]) {
|
|
305
|
+
_agent = this;
|
|
306
|
+
return origSubscribe.apply(this, args);
|
|
307
|
+
};
|
|
308
|
+
|
|
309
|
+
pi.on("session_start", () => {
|
|
310
|
+
// Bump generation so any in-flight recovery loop from a previous session
|
|
311
|
+
// exits within 100ms (during backoff) or right after prompt([]) returns.
|
|
312
|
+
_sessionGeneration++;
|
|
313
|
+
_recovering = false;
|
|
314
|
+
_trailingBangs = 0;
|
|
315
|
+
_tripped = false;
|
|
316
|
+
_weAborted = false;
|
|
317
|
+
_userAborted = false;
|
|
318
|
+
});
|
|
319
|
+
|
|
320
|
+
// before_agent_start fires only for user prompts (the AgentSession path),
|
|
321
|
+
// not for the recovery's direct agent.prompt([]), so these resets bound
|
|
322
|
+
// detection to the current user prompt without clearing mid-recovery.
|
|
323
|
+
pi.on("before_agent_start", () => {
|
|
324
|
+
_trailingBangs = 0;
|
|
325
|
+
_tripped = false;
|
|
326
|
+
_weAborted = false;
|
|
327
|
+
// A new user prompt is fresh activity — clear a stale user-abort flag.
|
|
328
|
+
_userAborted = false;
|
|
329
|
+
});
|
|
330
|
+
|
|
331
|
+
pi.on("message_start", (event) => {
|
|
332
|
+
if (event.message?.role === "assistant") {
|
|
333
|
+
_trailingBangs = 0;
|
|
334
|
+
_tripped = false;
|
|
335
|
+
_weAborted = false;
|
|
336
|
+
}
|
|
337
|
+
});
|
|
338
|
+
|
|
339
|
+
pi.on("message_update", (event, ctx) => {
|
|
340
|
+
// Refresh notify on every handler so it stays current after session
|
|
341
|
+
// switches (a stale ctx goes invalid). Same approach as pi-retry.
|
|
342
|
+
_notifyFn = (message, level) => ctx.ui.notify(message, level);
|
|
343
|
+
|
|
344
|
+
if (_recovering || _tripped) return;
|
|
345
|
+
const ame = event.assistantMessageEvent;
|
|
346
|
+
if (ame.type === "text_start" || ame.type === "thinking_start") {
|
|
347
|
+
// New content block (answer or reasoning) — trailing run starts fresh.
|
|
348
|
+
_trailingBangs = 0;
|
|
349
|
+
return;
|
|
350
|
+
}
|
|
351
|
+
// Watch both the visible answer and the reasoning trace: GLM 5.2 is a
|
|
352
|
+
// reasoning model, and the loop can surface in either.
|
|
353
|
+
if (ame.type !== "text_delta" && ame.type !== "thinking_delta") return;
|
|
354
|
+
if (!isGuardedModel(ctx.model)) return;
|
|
355
|
+
|
|
356
|
+
_trailingBangs = nextTrailingBangs(_trailingBangs, ame.delta);
|
|
357
|
+
if (_trailingBangs < BANG_THRESHOLD) return;
|
|
358
|
+
|
|
359
|
+
// Trip: abort the runaway stream and kick off the recovery loop (once,
|
|
360
|
+
// mutex-gated). The loop driver handles re-detection on resumed turns.
|
|
361
|
+
_tripped = true;
|
|
362
|
+
_weAborted = true;
|
|
363
|
+
const agent = _agent;
|
|
364
|
+
if (!agent) {
|
|
365
|
+
ctx.ui.notify(
|
|
366
|
+
"Makora death-loop guard: runaway '!' output detected but the Agent " +
|
|
367
|
+
"instance was not captured; cannot recover automatically.",
|
|
368
|
+
"warning",
|
|
369
|
+
);
|
|
370
|
+
return;
|
|
371
|
+
}
|
|
372
|
+
agent.abort();
|
|
373
|
+
// Detach: the handler must return so the run unwinds to idle before the
|
|
374
|
+
// loop driver awaits waitForIdle(). Only kick off the loop if it isn't
|
|
375
|
+
// already running; otherwise the running loop catches this abort.
|
|
376
|
+
if (!_recovering) {
|
|
377
|
+
void triggerRecovery();
|
|
378
|
+
}
|
|
379
|
+
});
|
|
380
|
+
|
|
381
|
+
// Detect user aborts via turn_end. Our own death-loop abort also surfaces
|
|
382
|
+
// as stopReason "aborted", so _weAborted gates this: only a non-we-aborted
|
|
383
|
+
// aborted turn is treated as a user Esc and stops the recovery loop.
|
|
384
|
+
pi.on("turn_end", (event) => {
|
|
385
|
+
const msg = event.message as GuardedMessage | undefined;
|
|
386
|
+
if (msg?.role === "assistant" && msg.stopReason === "aborted" && !_weAborted) {
|
|
387
|
+
_userAborted = true;
|
|
388
|
+
}
|
|
389
|
+
});
|
|
390
|
+
|
|
391
|
+
// Also refresh notify on turn_end so the loop driver has a fresh fn even
|
|
392
|
+
// if message_update hasn't fired yet this session.
|
|
393
|
+
pi.on("turn_end", (_event, ctx) => {
|
|
394
|
+
if (!_notifyFn) {
|
|
395
|
+
_notifyFn = (message, level) => ctx.ui.notify(message, level);
|
|
396
|
+
}
|
|
397
|
+
});
|
|
398
|
+
}
|
package/index.ts
CHANGED
|
@@ -20,6 +20,12 @@
|
|
|
20
20
|
* - Qwen 3.6 models: returns `reasoning` field.
|
|
21
21
|
* - Llama 3.3 70B: not a reasoning model.
|
|
22
22
|
*
|
|
23
|
+
* A death-loop guard (see ./death-loop-guard.ts) is registered alongside the
|
|
24
|
+
* provider. It watches the assistant text stream on the GLM 5.2 family and,
|
|
25
|
+
* if the model falls into an unbroken '!' repetition loop, aborts the runaway
|
|
26
|
+
* generation and resumes the agentic loop invisibly via agent.prompt([]) (the
|
|
27
|
+
* pi-invisible-continue pattern) — no new user message is injected.
|
|
28
|
+
*
|
|
23
29
|
* Developer role is NOT supported by any of the chat templates on Makora's
|
|
24
30
|
* vLLM deployment (prompts with role: "developer" are silently dropped).
|
|
25
31
|
* supportsDeveloperRole is set to false for all models.
|
|
@@ -42,6 +48,7 @@ import type { ExtensionAPI } from "@earendil-works/pi-coding-agent";
|
|
|
42
48
|
import modelsData from "./models.json" with { type: "json" };
|
|
43
49
|
import customModelsData from "./custom-models.json" with { type: "json" };
|
|
44
50
|
import patchData from "./patch.json" with { type: "json" };
|
|
51
|
+
import { registerDeathLoopGuard } from "./death-loop-guard.js";
|
|
45
52
|
|
|
46
53
|
// Types
|
|
47
54
|
|
|
@@ -205,4 +212,8 @@ export default function (pi: ExtensionAPI) {
|
|
|
205
212
|
api: "openai-completions",
|
|
206
213
|
models,
|
|
207
214
|
});
|
|
215
|
+
|
|
216
|
+
// Abort runaway '!' repetition loops on the GLM 5.2 family and resume the
|
|
217
|
+
// agentic loop invisibly (no new user message). See ./death-loop-guard.ts.
|
|
218
|
+
registerDeathLoopGuard(pi);
|
|
208
219
|
}
|
package/models.json
CHANGED
|
@@ -41,6 +41,27 @@
|
|
|
41
41
|
"maxTokensField": "max_completion_tokens"
|
|
42
42
|
}
|
|
43
43
|
},
|
|
44
|
+
{
|
|
45
|
+
"id": "google/gemma-4-26B-A4B",
|
|
46
|
+
"name": "Gemma 4 26B A4B",
|
|
47
|
+
"reasoning": false,
|
|
48
|
+
"input": [
|
|
49
|
+
"text"
|
|
50
|
+
],
|
|
51
|
+
"cost": {
|
|
52
|
+
"input": 0,
|
|
53
|
+
"output": 0,
|
|
54
|
+
"cacheRead": 0,
|
|
55
|
+
"cacheWrite": 0
|
|
56
|
+
},
|
|
57
|
+
"contextWindow": 262144,
|
|
58
|
+
"maxTokens": 0,
|
|
59
|
+
"compat": {
|
|
60
|
+
"supportsDeveloperRole": false,
|
|
61
|
+
"supportsStore": false,
|
|
62
|
+
"maxTokensField": "max_completion_tokens"
|
|
63
|
+
}
|
|
64
|
+
},
|
|
44
65
|
{
|
|
45
66
|
"id": "meta-llama/Llama-3.3-70B-Instruct",
|
|
46
67
|
"name": "Llama 3.3 70B Instruct",
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "pi-makora-provider",
|
|
3
|
-
"version": "1.
|
|
3
|
+
"version": "1.6.0",
|
|
4
4
|
"description": "Makora provider extension for pi - Access DeepSeek V4, GLM 5.2, Kimi K2.7 Code, Llama 3.3, Qwen 3.6, and more through the Makora inference API",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"main": "index.ts",
|
|
@@ -21,6 +21,7 @@
|
|
|
21
21
|
"license": "MIT",
|
|
22
22
|
"files": [
|
|
23
23
|
"index.ts",
|
|
24
|
+
"death-loop-guard.ts",
|
|
24
25
|
"models.json",
|
|
25
26
|
"custom-models.json",
|
|
26
27
|
"patch.json",
|
|
@@ -33,6 +34,8 @@
|
|
|
33
34
|
]
|
|
34
35
|
},
|
|
35
36
|
"devDependencies": {
|
|
37
|
+
"@earendil-works/pi-agent-core": "^0.80.2",
|
|
38
|
+
"@earendil-works/pi-coding-agent": "^0.80.2",
|
|
36
39
|
"vitest": "^4.1.9"
|
|
37
40
|
},
|
|
38
41
|
"scripts": {
|