pi-antiloop 1.5.1 → 1.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -6,15 +6,16 @@
6
6
 
7
7
  # Antiloop — Loop Detection and Break for pi
8
8
 
9
- **Antiloop watches every assistant message, tool call and thinking block, and forces the model out of reasoning loops before they eat your context and your patience.** Four simultaneous detection strategies (text similarity, tool-call sequences, thinking content, structural openings) find loops that humans miss — and progressive intervention (warning → force break → abort) tells the model to take a different approach, without you having to babysit it.
9
+ **Antiloop watches every assistant message, tool call and thinking block, and forces the model out of reasoning loops before they eat your context and your patience.** Seven simultaneous detection strategies (text similarity, tool-call sequences, thinking content, structural openings, degenerate repetition, block/narration repetition, no-progress outcome runs) find loops that humans miss — and progressive intervention (warning → force break → abort) tells the model to take a different approach, without you having to babysit it.
10
10
 
11
11
  ---
12
12
 
13
13
  ## Features
14
14
 
15
- - **Four detection strategies** — text repetition (trigram Jaccard + Levenshtein), tool-call sequences (name + near-identical arguments + same outcome — result-aware, so retries that make progress don't false-positive), thinking blocks, and structural opening-phrase patterns
15
+ - **Seven detection strategies** — text repetition (trigram Jaccard + Levenshtein), tool-call sequences (name + near-identical arguments + same outcome — result-aware, so retries that make progress don't false-positive), thinking blocks, structural opening-phrase patterns, **degenerate repetition** (a single message/call stuck repeating one word hundreds of times — the `noguerol ×5145` meltdown — caught at `message_end` with no repeated peer needed, and degenerate bash commands blocked before they execute), **block repetition** (ONE message replaying whole sentences/phrases — the narration loop: `Let me start by checking the environment…` ×5 — same `message_end` timing and strong turn weight), and **no-progress outcome runs** (the NFS `test A…QQQ` class: dozens of near-identical re-runs of the same experiment, every one ending in the *same failing outcome* — args mutate so tool-loop can't see it; the repeated failure signature can)
16
+ - **Stays alive in long sessions** — tracked messages carry a monotonic sequence number, so trimming the sliding window can never make the dedupe guard skip later messages (a bug that silently blinded antiloop after ~15 tracked messages)
16
17
  - **Task-stream recognition (batch work)** — when another extension (e.g. `punched` appending lines to pi.md, or `plan` adding tasks) makes the model call the *same* tool many times with *different* content, antiloop recognizes it as N distinct tasks of one type and stays silent — no warning, no force break. A genuine loop (the *same* call repeated verbatim) is still caught
17
- - **Progressive intervention** — `warning` reminds the model to vary its approach; `force break` injects explicit anti-loop instructions and modifies context; `abort` stops the run entirely
18
+ - **Progressive intervention** — `warning` reminds the model to vary its approach; `force break` steers a real break message into the running agent before its next LLM call; `abort` stops the run entirely
18
19
  - **Configurable thresholds** — independent dials for similarity cutoff, warning/force-break/abort counts, detection window, and which strategies are on
19
20
  - **Sliding window** — only the last N messages are compared, so detection is O(N) in the window size, not in the full session
20
21
  - **Live footer indicator** — `🔄 antiloop(on|off)` in the footer, per spec, with the current level (`⚠️/🛑/🚨`) and consecutive count; an interactive TUI footer adds a keyboard toggle (`esc+a` by default, configurable/off, honored only while idle) and preserves the built-in footer's pwd/branch/context/model info
@@ -98,10 +99,16 @@ Detection strategies:
98
99
  Text loops: ✅
99
100
  Tool loops: ✅
100
101
  Thinking loops: ✅
102
+ Degenerate: ✅
103
+ Block repetition: ✅
104
+ Outcome (no-progress): ✅
101
105
 
102
106
  Recent detections:
103
107
  [text] Text similarity 85% with message 3 (2m ago)
104
108
  [tool] repeated 3x: bash (5m ago)
109
+ [degenerate] bash command: degenerate repetition — "noguerol" ×5145 (5m ago)
110
+ [block] message text: block repetition — 100% of 450 tokens replay repeated 5-word phrases (5m ago)
111
+ [outcome] no progress: 9 near-identical bash attempts with the same failing outcome (5m ago)
105
112
  ```
106
113
 
107
114
  ### `/antiloop config`
@@ -123,6 +130,11 @@ Grouped interactive menu showing the current value in each option:
123
130
  - **🔧 call similarity** — `99 / 95 / 90 / 80%` — how identical tool-call *arguments* must be to count as the same call (default 95%: only near-identical repeats loop)
124
131
  - **🔁 call repeats** — `1 / 2 / 3` — how many times the same call must repeat before it flags (default 2)
125
132
  - **🧾 result similarity** — `95 / 80 / 60%` — how similar captured results must be to count as the *same outcome*; a repeated command that starts producing a different result is progress, not a loop (default 80%)
133
+ - **🌀 degenerate run** — `8 / 16 / 24 / 48` — identical words in a row inside ONE message/call before it counts as stuck generation (default 16; the `noguerol ×5145` class)
134
+ - **🔁 block repetition** — on/off — detect ONE message replaying whole sentences/phrases (the narration loop: `Let me start by checking the environment…` ×5) with no repeated peer needed
135
+ - **🔁 block share** — `70 / 85 / 95%` — share of replayed 5-word phrases in one message before it counts as a replay (default 85%)
136
+ - **⛔ block repetitive bash** — on/off — refuse a degenerate or replayed bash command before it executes, feeding the reason back to the model (default on)
137
+ - **📉 no-progress after** — `4 / 6 / 8 / 12` — same failing outcome repeated this many times (near-identical args) before flagging (default 8; the NFS `test A…QQQ` class)
126
138
 
127
139
  **📋 Task streams** — batch work (punched_log, plan_manager, …) is N tasks of one type, not a loop
128
140
  - **📋 task streams** — on/off — recognize that batch work and stay silent
@@ -133,6 +145,9 @@ Grouped interactive menu showing the current value in each option:
133
145
  - **📝 text** — on/off — detect repeated text messages
134
146
  - **🔧 tools** — on/off — detect repeated tool calls
135
147
  - **🧠 thinking** — on/off — detect repeated internal reasoning
148
+ - **🌀 degenerate** — on/off — detect one message stuck repeating a single word/token (no repeated peer needed)
149
+ - **🔁 block** — on/off — detect one message replaying whole sentences/phrases (narration loop, no repeated peer needed)
150
+ - **📉 outcome** — on/off — detect many near-identical attempts all ending in the same failing outcome (no progress)
136
151
 
137
152
  **🧹 reset state** — clear all counters and history
138
153
 
@@ -160,13 +175,26 @@ batch no detections → silent (exp silent — 98.9% args would match without
160
175
  loop still detected → tool (exp tool — identical repeats are NOT a stream) ✅
161
176
  loop survives batch gate→ tool (exp tool — bash repeats are real) ✅
162
177
  stream needs ≥3 calls → no stream (exp no stream at 2 calls) ✅
178
+ degenerate first sight → degenerate (bash command: "noguerol" ×402 …) ✅ ← v1.6: ONE meltdown message, no peer
179
+ degenerate legit cmd → ok (exp ok) ✅
180
+ degenerate interleaved → flag (noguerol ×150) (exp flag — freq/share clause) ✅
181
+ degenerate glued token → flag (noguerol ×300) (exp flag — perfect power) ✅
182
+ degenerate turn weight → degenerate 2, text 1 (exp 2, 1) ✅
183
+ block S0 narration loop → block (message text: 100% of 450 tokens replay repeated 5-word phrases) ✅ ← v1.7: ONE message replaying sentences, no peer
184
+ block 3 cycles / 2 cycles→ flag (100%) / no (exp flag / no — needs ≥3 repeats) ✅
185
+ block legit prose/logs → ok (exp ok) ✅
186
+ outcome fires on 9th → outcome (9 near-identical bash attempts, same failing outcome) ✅ ← v1.6.1: NFS test-A…QQQ class
187
+ outcome needs 8 prior → silent (exp silent at 5 attempts) ✅
188
+ outcome converging sweep → silent (exp silent — outcomes differ = progress) ✅
189
+ outcome diff failures → silent (exp silent — error changed = progress) ✅
190
+ outcome identical OKs → silent (exp silent — success repeats ≠ loop) ✅
163
191
  ```
164
192
 
165
193
  ## How It Works
166
194
 
167
195
  ### Detection pipeline
168
196
 
169
- After every assistant `message_end` event, antiloop extracts the new content (text, thinking, tool calls — including their ids) and pushes it onto a sliding window of the last `detectionWindow + 5` messages. Detection itself runs at `turn_end`, once the tool results are known: results are fingerprinted and attached to the tracked calls, then the active detection strategies run against the window:
197
+ After every assistant `message_end` event, antiloop extracts the new content (text, thinking, tool calls — including their ids) and pushes it onto a sliding window of the last `detectionWindow + 5` messages. Detection itself runs at `turn_end`, once the tool results are known: results are fingerprinted and attached to the tracked calls, then the active detection strategies run against the window. The exceptions are the **intra-message** strategies — **degenerate** and **block**. A single stuck message needs no peer and no tool result, so both are evaluated right at `message_end` — the only point before the message's own tool calls execute — and repetitive `bash` calls are also blocked at the `tool_call` hook.
170
198
 
171
199
  | Strategy | What it compares | Algorithm |
172
200
  |----------|------------------|-----------|
@@ -174,9 +202,12 @@ After every assistant `message_end` event, antiloop extracts the new content (te
174
202
  | Tool | Tool name + arguments (+ captured result) | Sequence match + near-identical args (≥ `toolSimilarityThreshold`, default 95%) *and* ≥ `minToolRepeatCount` prior recurrences. **Result veto:** if both runs captured a result and the outcomes differ, it's progress, not a loop |
175
203
  | Thinking | Internal reasoning/thinking blocks | Same as text |
176
204
  | Structural | First 10 words of each message | Opening-phrase similarity ≥ 90% across ≥ 3 messages |
205
+ | Degenerate | One single message/call (no peer needed) | Run-length + frequency of identical words inside the payload: ≥ `degenerateMaxRun` (default 16) consecutive identical words, or one word ≥ `degenerateMaxFreq`× at ≥ `degenerateMaxShare` of all tokens; plus a perfect-power check for glued no-space tokens. Scanned at `message_end` — before the tool calls execute — and on every `bash` `tool_call` (blocking gate) |
206
+ | Block | One single message (no peer needed) | Sliding `blockNgram`-word n-grams over the normalized payload: a payload of ≥ `blockMinTokens` (120) words is a replay when ≥ `blockRepeatShare` (default 85%) of the n-gram positions recur AND the most repeated n-gram appears ≥ `blockMinRepeats` (3) times. Catches a message replaying whole sentences/phrases — the narration loop (`Let me start by checking the environment…` ×5) that every cross-message detector misses because there is no peer. Scanned at `message_end` and on every `bash` `tool_call` (blocking gate) |
207
+ | Outcome | Single tool calls across the window, after the last user input | ≥ `outcomeMinRepeats` (default 8) PRIOR attempts with args ≥ `outcomeArgSimilarity` (0.85) similar AND the same *failing* outcome (failure signatures compared at ≥ `outcomeSigThreshold`, 0.7; identical OK results never count — they're the norm for batches). Catches mutated re-run loops the tool detector can't see (labels/permutations change every turn) |
177
208
  | Task stream | Same tool, many calls | When a tool appears ≥ `taskStreamMinCalls` times (default 3) in the window and *no two* calls are near-identical (`taskStreamTwinThreshold`, default 99%), the tool is an active batch: N different tasks of one type (e.g. `punched_log` appends, `plan_manager` task adds). Those calls are exempt from tool-loop detection, and text/thinking/structural patterns that only involve those batch messages are suppressed too. If even one call pair is a twin (the same task repeated), the tool is *not* a stream and detection proceeds normally |
178
209
 
179
- Each detected pair becomes a `LoopDetection { type, similarity, messageIndices, description }` and the consecutive counter increases.
210
+ Each detected pair becomes a `LoopDetection { type, similarity, messageIndices, description }` and the consecutive counter increases (a degenerate or block turn counts `degenerateTurnWeight`, default 2 — warning on first sight).
180
211
 
181
212
  ### Intervention levels
182
213
 
@@ -184,11 +215,13 @@ Each detected pair becomes a `LoopDetection { type, similarity, messageIndices,
184
215
  |-------|---------|----------|
185
216
  | 0 (no loop) | — | Silent — passes the message through |
186
217
  | 1 (warning) | `consecutiveDetections >= warningThreshold` | Notifies the user (`⚠️`) — no message is injected into the conversation, so the model's generation is never interrupted by the warning itself |
187
- | 2 (force break) | `consecutiveDetections >= forceBreakThreshold` | Injects mandatory anti-loop instructions + appends a context message to the last assistant message |
188
- | 3 (abort) | `consecutiveDetections >= abortThreshold` | (Disabled by default) Surfaces an error asking the user for new instructions |
218
+ | 2 (force break) | `consecutiveDetections >= forceBreakThreshold` | Steers a real break message into the running agent (`pi.sendUserMessage`, delivered right before its next LLM call) telling it to stop repeating and change approach |
219
+ | 3 (abort) | `consecutiveDetections >= abortThreshold` | (Disabled by default) Stops the run outright via `ctx.abort()` |
189
220
 
190
221
  The level never de-escalates during an active loop; user input decays the consecutive counter naturally so a fresh prompt can break the cycle.
191
222
 
223
+ **Degenerate and block turns are handled earlier than the ladder:** they are detected at `message_end` (the same moment the text/tool-call payload is complete, *before* pi preflights and executes its tools). A single such turn already adds `degenerateTurnWeight` (2) consecutive points → warning on first sight; the second consecutive one → force break steer; after the steer, further degenerate/blocked output counts against `ignoredSteerLimit` → hard stop. Repetitive `bash` calls are additionally refused by the `tool_call` gate (`blockDegenerateBash`) — the command never runs, and the block reason is fed back to the model as the tool error so it can still change approach.
224
+
192
225
  ### Similarity scoring
193
226
 
194
227
  ```
@@ -228,6 +261,7 @@ If the same command produced a *different* outcome, the pair is progress:
228
261
 
229
262
  Results only veto; they never trigger on their own, and calls without a
230
263
  captured result fall back to argument matching alone.
264
+ ```
231
265
 
232
266
  ### Task streams: N tasks of one type ≠ a loop
233
267
 
@@ -262,7 +296,135 @@ Tunables: `detectTaskStreams` (master switch), `taskStreamMinCalls` (batch
262
296
  size needed before recognition), `taskStreamTwinThreshold` (how similar args
263
297
  must be to count as *the same task* — lower it to treat near-duplicate
264
298
  entries as loops again).
265
- ```
299
+
300
+ ### Degenerate repetition: ONE message stuck on a word ≠ a reasoning loop
301
+
302
+ All strategies above need at least two similar messages — they detect a model
303
+ *repeating itself across turns*. There is a different, equally destructive
304
+ failure mode they cannot see: the model's decoder **anchors on a token and
305
+ stops producing new output**, repeating the same word hundreds of times
306
+ *inside a single message or tool call*. Real case (session `2026-09-09T15-43`,
307
+ /home/j — Qwen3.8-27B on llama.cpp): one 46 KB `bash` call whose SSH
308
+ username wordlist repeated `noguerol` **5145 times** (a run of 5140 — 99% of
309
+ the payload). Every cross-message detector stayed silent (nothing to compare
310
+ against — it happened exactly once), and only a manual ESC stopped it.
311
+
312
+ Antiloop v1.6 detects this **degenerate repetition** directly, on the first
313
+ occurrence, with no peer message:
314
+
315
+ - **Run-length** — a payload of ≥ `degenerateMinTokens` (50) normalized words
316
+ containing ≥ `degenerateMaxRun` (16) *consecutive identical* words is a
317
+ meltdown.
318
+ - **Frequency share** — one word appearing ≥ `degenerateMaxFreq` (60) times
319
+ with ≥ `degenerateMaxShare` (40%) of all tokens catches interleaved
320
+ meltdowns (`A B A B A B…`) that have no long run.
321
+ - **Perfect power** — a single giant token with no separators at all
322
+ (`noguerolnoguerol…`) is checked for periodicity.
323
+
324
+ Tokenization uses runs of letters (unicode), so JSON stringification noise
325
+ (escaped `\n` in stored tool args, punctuation, digits, code symbols) never
326
+ fragments the repeated word — and legit payloads full of numbers or code
327
+ symbols don't false-positive. 1-letter tokens are ignored as candidates
328
+ (`{"a":1}` JSON keys can't trigger).
329
+
330
+ Because the signal is conclusive, it acts **before the tools run**: the
331
+ meltdown is caught at `message_end` (pi emits it once the assistant message is
332
+ complete, before tool preflight), escalates immediately (one degenerate turn =
333
+ `degenerateTurnWeight` = 2 consecutive points → warning on first sight, force
334
+ steer on the second consecutive meltdown), and any degenerate `bash` command
335
+ is **blocked in the `tool_call` hook** (`blockDegenerateBash`) — the 46 KB
336
+ brute-force style command never executes. The block reason is returned to the
337
+ model as the tool error, so the next LLM call can still change approach; only
338
+ if it keeps melting does antiloop steer and then hard-stop the run.
339
+
340
+ Legit commands are safe: real scripts never repeat one word 16+ times in a
341
+ row inside a ≥ 50-token payload (verified against the actual sequential bash
342
+ sweeps that motivated the tool-loop threshold). Only `bash` gets the blocking
343
+ gate — writing a repetitive *file* (e.g. a user-requested padding fixture) is
344
+ alerted and escalated, not refused.
345
+
346
+ ### Block repetition: ONE message replaying its own sentences
347
+
348
+ The degenerate detector catches a single *word* repeated hundreds of times.
349
+ There is a second intra-message failure mode, just as obvious to a human and
350
+ invisible to every cross-message detector: **the model replays whole
351
+ sentences/phrases inside one generation**. Real case (a coding session's S0
352
+ scaffold): ONE assistant message cycled ~5 times through
353
+
354
+ > Let me start by checking the environment and the current state of the
355
+ > repository, then set up a plan for the S0 slice and begin building.
356
+ > I'll run several independent checks in parallel.
357
+ > Let me begin the S0 development. First, reconnaissance of the environment
358
+ > and current repo state.
359
+
360
+ …never emitting a tool call. Text/tool/thinking/structural detection all
361
+ compare *across* messages and had nothing to compare against; the degenerate
362
+ scan only watches for one word repeated, and no single word repeated 16×.
363
+
364
+ Antiloop v1.7 detects **block repetition** on the message itself:
365
+
366
+ - the payload is read as a stream of lowercase alphanumeric words;
367
+ - a sliding window of `blockNgram` (default 5) words is built over it;
368
+ - if at least `blockRepeatShare` (default **85%**) of those n-gram positions
369
+ recur somewhere else, the most repeated n-gram appears at least
370
+ `blockMinRepeats` (default 3) times, and the payload has at least
371
+ `blockMinTokens` (default 120) words, the generation is replaying itself.
372
+
373
+ The thresholds are deliberately far apart from normal content: ordinary
374
+ prose — even long, structured documents — scores ≤ 11% coverage, while 3+
375
+ replays of a narration block reach 98–100%. Templated payloads with evolving
376
+ data (log lines, generated code) are not flagged because their n-grams carry
377
+ the varying digits and differ. A single restatement (2 cycles) does not trip
378
+ it — one echo is a summary, three are stuck generation.
379
+
380
+ Because the signal is conclusive it is handled like the degenerate one:
381
+ caught at `message_end` (before the message's tools execute), weighted
382
+ `degenerateTurnWeight` (2) so the first occurrence warns, and a replayed
383
+ `bash` command is refused by the same `blockDegenerateBash` gate. After a
384
+ force-break steer, another replayed block escalates to the hard stop.
385
+
386
+ Tunables: `detectBlockRepeats`, `blockRepeatShare` (lower = earlier),
387
+ `blockMinRepeats`, `blockNgram`, `blockMinTokens`.
388
+
389
+ ### No-progress outcome runs: mutated re-runs, same wall
390
+
391
+ A subtler meltdown than the degenerate one: the model re-runs the SAME
392
+ experiment over and over, mutating a cosmetic label or permutation each time
393
+ so no call ever repeats verbatim — while the outcome stays the SAME FAILURE.
394
+ Real case (same session, rows 95–249): ~90 ssh `exportfs`/`mount` tests,
395
+ labels `test A` … `test QQQ`, targets alternating Javi/Compartido — every one
396
+ ending `access denied` / `rc=32`, with fresh journalctl noise per attempt.
397
+ The tool-loop detector is blind to it BY DESIGN (args mutate every turn:
398
+ mean adjacent trigram similarity 0.93, but the label always changes, so the
399
+ same call never recurs `minToolRepeatCount` times) and results only *veto*
400
+ tool loops today — nothing used "same outcome repeated" as a positive signal.
401
+
402
+ Antiloop v1.6.1's **outcome detector** closes exactly that gap, conservatively:
403
+
404
+ - the LAST turn's single tool call must have a captured result that is a
405
+ FAILURE (fingerprints carry an error signature — `ok|fail|sig|…` — extracted
406
+ around the first failure marker over the FULL output, because a tool run can
407
+ fail with `isError=false`: the ssh pipeline exits 0 while `rc=32` lives
408
+ inside the text);
409
+ - at least `outcomeMinRepeats` (default 8) PRIOR single-call turns (all after
410
+ the last real user message) must share BOTH args ≥ `outcomeArgSimilarity`
411
+ (0.85 — the same experiment reshuffled) AND the same failure signature
412
+ (digit-stripped signatures compared at `outcomeSigThreshold`, 0.7 — mount
413
+ targets that legitimately vary between attempts survive, journalctl noise
414
+ doesn't).
415
+
416
+ Guarantees preserved: converging sweeps change outcome → silent; different
417
+ failures = evolving diagnosis → silent; identical OKs (task-stream batches,
418
+ file writes, idempotent verifications) never count as failures → silent;
419
+ user-steered turns don't count → only the autonomous stretch is judged.
420
+ Because the signal is proven no-progress, one such turn adds
421
+ `degenerateTurnWeight` (2) points → warning on the crossing turn, force-break
422
+ steer on the next, hard stop shortly after if the same wall persists. On the
423
+ real session this fires at `test MM` (warn) → `NN` (steer) → `PP` (abort) —
424
+ ~50 wasted experiment turns cut.
425
+
426
+ Tunables: `detectOutcomeLoops`, `outcomeMinRepeats` (lower = earlier cutoff),
427
+ `outcomeArgSimilarity`, `outcomeSigThreshold`.
266
428
 
267
429
  ### Sliding window
268
430
 
@@ -282,6 +444,22 @@ Persisted as JSON at `~/.pi/agent/antiloop.json`:
282
444
  "toolSimilarityThreshold": 0.95,
283
445
  "minToolRepeatCount": 2,
284
446
  "resultSimilarityThreshold": 0.8,
447
+ "detectDegenerate": true,
448
+ "degenerateMinTokens": 50,
449
+ "degenerateMaxRun": 16,
450
+ "degenerateMaxFreq": 60,
451
+ "degenerateMaxShare": 0.4,
452
+ "degenerateTurnWeight": 2,
453
+ "blockDegenerateBash": true,
454
+ "detectBlockRepeats": true,
455
+ "blockMinTokens": 120,
456
+ "blockNgram": 5,
457
+ "blockMinRepeats": 3,
458
+ "blockRepeatShare": 0.85,
459
+ "detectOutcomeLoops": true,
460
+ "outcomeMinRepeats": 8,
461
+ "outcomeArgSimilarity": 0.85,
462
+ "outcomeSigThreshold": 0.7,
285
463
  "detectTaskStreams": true,
286
464
  "taskStreamMinCalls": 3,
287
465
  "taskStreamTwinThreshold": 0.99,
@@ -306,6 +484,22 @@ Persisted as JSON at `~/.pi/agent/antiloop.json`:
306
484
  | `toolSimilarityThreshold` | `0.95` | How close tool-call arguments must be (0.0–1.0) to count as the *same* call — see [tool loops](#how-it-works) |
307
485
  | `minToolRepeatCount` | `2` | Prior occurrences of a near-identical call set required before a tool loop is flagged (2 = same call seen 3×) |
308
486
  | `resultSimilarityThreshold` | `0.8` | Minimum similarity between captured result tails to still count as the *same outcome*; below this, a repeated command is treated as progress, not a loop |
487
+ | `detectDegenerate` | `true` | Detect intra-message degenerate repetition — one message/call stuck repeating a single word (the `noguerol ×5145` class). Needs no repeated peer message; scanned at `message_end` and on every bash `tool_call` |
488
+ | `degenerateMinTokens` | `50` | Minimum normalized tokens in a payload before it is scanned for degenerate repetition (shorter payloads aren't conclusive) |
489
+ | `degenerateMaxRun` | `16` | Consecutive identical words inside one payload that flag it as degenerate |
490
+ | `degenerateMaxFreq` | `60` | One word's total occurrences (with `degenerateMaxShare` of the payload) that flags interleaved meltdowns |
491
+ | `degenerateMaxShare` | `0.4` | Frequency share (freq/total tokens) required together with `degenerateMaxFreq` |
492
+ | `degenerateTurnWeight` | `2` | Consecutive-detection points added by one degenerate turn (2 = warning on first sight) |
493
+ | `blockDegenerateBash` | `true` | Block a degenerate or block-repetition `bash` command in the `tool_call` hook before it executes; the reason is fed back to the model as the tool error |
494
+ | `detectBlockRepeats` | `true` | Detect intra-message block/narration repetition — ONE message replaying whole sentences/phrases (the `Let me start by checking the environment…` ×5 class). Needs no repeated peer; scanned at `message_end` and on every bash `tool_call` |
495
+ | `blockMinTokens` | `120` | Minimum normalized words in a payload before it is scanned for block repetition (shorter payloads aren't conclusive) |
496
+ | `blockNgram` | `5` | Word window of the n-grams whose recurrence is measured |
497
+ | `blockMinRepeats` | `3` | The most repeated n-gram must occur at least this many times to flag |
498
+ | `blockRepeatShare` | `0.85` | Share of n-gram positions that must recur for the payload to count as a replay |
499
+ | `detectOutcomeLoops` | `true` | No-progress outcome runs: ≥ `outcomeMinRepeats` near-identical attempts (args ≥ `outcomeArgSimilarity`) all ending in the *same failing outcome* — the NFS `test A…QQQ` class |
500
+ | `outcomeMinRepeats` | `8` | Prior same-failure attempts (inside the window, after the last user input) required before the outcome detector fires |
501
+ | `outcomeArgSimilarity` | `0.85` | How similar args must be to count as the *same experiment reshuffled* (mutations of labels/permutations stay under it — distinct tasks don't) |
502
+ | `outcomeSigThreshold` | `0.7` | Minimum similarity between digit-stripped failure signatures to count as the *same failure* |
309
503
  | `detectTaskStreams` | `true` | Recognize homogeneous batch work (same tool called with distinct content — e.g. punched/plan/obsidian extensions) and stay silent; see [task streams](#task-streams-n-tasks-of-one-type--a-loop) |
310
504
  | `taskStreamMinCalls` | `3` | Same-tool calls required inside the window before a task stream is recognized |
311
505
  | `taskStreamTwinThreshold` | `0.99` | Arguments this similar (or identical) count as *the same task* — a twin invalidates the stream and re-enables normal loop detection |
@@ -326,7 +520,10 @@ Persisted as JSON at `~/.pi/agent/antiloop.json`:
326
520
  4. **Per-strategy toggles** — if the model's reasoning legitimately repeats (e.g. it's working through a checklist), disable `thinking` detection and leave text/tool on.
327
521
  5. **Watch the log** — `/antiloop log` shows what's actually triggering. If you see false positives, raise `similarityThreshold` instead of disabling the strategy entirely.
328
522
  6. **Let user input clear state** — each user message decays the consecutive counter by 2, so a fresh prompt naturally resets without `/antiloop reset`.
329
- 7. **`/antiloop test`** — runs the real detection engine (text + tool-call regression cases) to verify calibration after any change.
523
+ 7. **Degenerate detector needs no tuning for most setups** — a run of ≥ 16 identical words (or one word ≥ 40% of a ≥ 50-token payload) inside a single message is conclusive stuck generation; the 46 KB `noguerol ×5145` SSH-wordlist meltdown is caught on first sight (warning), its bash never executes (`blockDegenerateBash`), and a second consecutive meltdown gets the force-break steer. If a model legitimately writes repetitive payloads, raise `degenerateMaxRun` / `degenerateMaxFreq` via `/antiloop config` — don't disable the detector.
524
+ 8. **Outcome detector catches mutated re-run loops** — a model that re-issues the same experiment with cosmetic changes (labels, permutations) while every attempt fails identically gets a warning after `outcomeMinRepeats` (8) same-failure attempts, a steer on the next, and a hard stop shortly after. Converging sweeps, evolving failures and repeated successes stay silent by design. Lower `outcomeMinRepeats` if you want earlier cutoffs.
525
+ 9. **Block detector catches narration loops** — a single message replaying whole sentences (the `Let me start by checking the environment…` ×5 class) warns on first sight and escalates like a meltdown. It only fires when ≥ 85% of the message's 5-word phrases recur, so ordinary prose and templated logs/code stay silent. Raise `blockRepeatShare` (or `blockMinRepeats`) if you ever see a false positive.
526
+ 10. **`/antiloop test`** — runs the real detection engine (text + tool-call + task-stream + degenerate + block + outcome regression cases) to verify calibration after any change.
330
527
 
331
528
  ## Architecture
332
529
 
@@ -351,10 +548,10 @@ Modular extension with zero external dependencies (only pi's bundled `@earendil-
351
548
 
352
549
  - **Levenshtein + trigram Jaccard** hybrid — small texts use edit distance, large texts use n-gram overlap (each is O(N) in text length)
353
550
  - **Sliding window** — only the last `detectionWindow` messages participate, capping memory at O(W × message_size)
354
- - **Early bail** — short messages and empty tool calls skip similarity computation entirely
551
+ - **Early bail** — short messages and empty tool calls skip similarity computation entirely; the degenerate and block scans are single linear passes
355
552
  - **TUI integration** — uses `ctx.ui.select` for the config menu and the log viewer; `ctx.ui.notify` for state notifications; `ctx.ui.setStatus` + a custom `ctx.ui.setFooter` component for the persistent footer indicator, live level info, and the `esc+a` keyboard toggle (`ctx.ui.onTerminalInput`, never consumes input)
356
- - **Hooks** — `message_end` (track messages + tool call ids), `turn_end` (attach result fingerprints, detect, and intervene: steer the force break / abort the run), `input` (decay on real user messages only), `session_start` (load config + install footer + reset), `session_shutdown` (restore built-in footer)
357
- - **Intervention runs on the turn loop, not on user prompts** — escalation is decided at `turn_end`, the break is steered into the running agent before its next LLM call, and the guaranteed hard stop aborts the run (`ctx.abort`, fire-and-forget — never awaited, so the hook can't deadlock). No custom-role messages are injected into the conversation at any level (steering a real user message + aborting are the only levers; custom-role injections were removed because a model can stall on an unexpected injected message)
553
+ - **Hooks** — `message_end` (track messages + tool call ids with a monotonic sequence so the sliding-window trim can never collide turn indices, and pre-handle degenerate/block repetition — the message is complete but its tools haven't executed yet), `tool_call` (block degenerate or replayed `bash` commands before they run), `turn_end` (attach result fingerprints with failure signatures, detect — including no-progress outcome runs — and intervene: steer the force break / abort the run), `input` (decay on real user messages only), `session_start` (load config + install footer + reset), `session_shutdown` (restore built-in footer)
554
+ - **Intervention runs on the turn loop, not on user prompts** — escalation is decided at `turn_end` (and at `message_end` for the self-contained intra-message signals), the break is steered into the running agent before its next LLM call, and the guaranteed hard stop aborts the run (`ctx.abort`, fire-and-forget — never awaited, so the hook can't deadlock). No custom-role messages are injected into the conversation at any level (steering a real user message + aborting are the only levers; custom-role injections were removed because a model can stall on an unexpected injected message)
358
555
 
359
556
  ## License
360
557
 
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "name": "pi-antiloop",
3
- "version": "1.5.1",
4
- "description": "Antiloop: detect reasoning loops and force a break (warn \u2192 force \u2192 abort) across text, tool, thinking, and structural patterns. The force break is delivered mid-run: a real break message is steered into the running agent right before its next LLM call, and if the model ignores it and keeps repeating verbatim, antiloop aborts the run \u2014 an autonomous tool loop always terminates. Tool-loop detection is result-aware: only near-identical repeated calls with the same outcome count, so sequential bash operations and retries that make progress don't false-positive. Task-stream recognition: when an extension (punched, plan, \u2026) makes the model call the SAME tool many times with DIFFERENT content \u2014 N distinct tasks of one type, e.g. appending lines or adding plan tasks \u2014 antiloop stays silent.",
3
+ "version": "1.7.0",
4
+ "description": "Antiloop: detect reasoning loops and force a break (warn → force → abort) across text, tool, thinking, structural, degenerate-repetition, block/narration-repetition, and no-progress outcome patterns. Degenerate (one message stuck repeating a word hundreds of times — noguerol ×5145) is caught at message_end and its bash blocked before executing. Block (ONE message replaying whole sentences/phrases — the 'Let me start by checking the environment…' ×5 narration loop, invisible to every cross-message detector because there is no peer) fires on the first such message with the same strong turn weight and blocks a replayed bash command. No-progress (the NFS test-A…QQQ class: ~90 mutated re-runs of the same experiment, every one failing identically) fires when the same failing outcome repeats ≥ outcomeMinRepeats times with near-identical args. Fixes antiloop going blind mid-session: tracked messages use a monotonic sequence so the trim of the sliding window can never collide turn indices. The force break is delivered mid-run (steer before the next LLM call) and if the model ignores it antiloop aborts the run — an autonomous tool loop always terminates. Result-aware tool-loop detection keeps sequential bash and converging sweeps quiet; task-stream recognition keeps punched/plan batch work quiet.",
5
5
  "keywords": [
6
6
  "pi-package",
7
7
  "antiloop",
package/src/commands.ts CHANGED
@@ -58,9 +58,12 @@ async function showStatus(ctx: ExtensionCommandContext, rt: Runtime): Promise<vo
58
58
  ` similarity: ${(rt.config.similarityThreshold * 100).toFixed(0)}% window: ${rt.config.detectionWindow}`,
59
59
  ` tool sim: ${(rt.config.toolSimilarityThreshold * 100).toFixed(0)}% tool repeat: ${rt.config.minToolRepeatCount}+ prior`,
60
60
  ` result sim: ${(rt.config.resultSimilarityThreshold * 100).toFixed(0)}% (same cmd + diff outcome = no loop)`,
61
+ ` degenerate: run ≥ ${rt.config.degenerateMaxRun} same word · freq ≥ ${rt.config.degenerateMaxFreq} @ ${(rt.config.degenerateMaxShare * 100).toFixed(0)}% (≥ ${rt.config.degenerateMinTokens} tokens) · weight ${rt.config.degenerateTurnWeight} · block bash ${yn(rt.config.blockDegenerateBash)}`,
62
+ ` block repeats: ${yn(rt.config.detectBlockRepeats)} (≥ ${(rt.config.blockRepeatShare * 100).toFixed(0)}% of ≥ ${rt.config.blockMinTokens} tokens replay ${rt.config.blockNgram}-grams ×${rt.config.blockMinRepeats}+)`,
63
+ ` outcome: same failing result ≥ ${rt.config.outcomeMinRepeats} attempts (args ≥ ${(rt.config.outcomeArgSimilarity * 100).toFixed(0)}% sim, sig ≥ ${(rt.config.outcomeSigThreshold * 100).toFixed(0)}%)`,
61
64
  ` task streams: ${yn(rt.config.detectTaskStreams)} (min ${rt.config.taskStreamMinCalls} calls, twins ≥ ${(rt.config.taskStreamTwinThreshold * 100).toFixed(0)}%)`,
62
65
  "",
63
- `detectors: text ${yn(rt.config.detectTextLoops)} · tool ${yn(rt.config.detectToolLoops)} · think ${yn(rt.config.detectThinkingLoops)}`,
66
+ `detectors: text ${yn(rt.config.detectTextLoops)} · tool ${yn(rt.config.detectToolLoops)} · think ${yn(rt.config.detectThinkingLoops)} · degenerate ${yn(rt.config.detectDegenerate)} · block ${yn(rt.config.detectBlockRepeats)} · outcome ${yn(rt.config.detectOutcomeLoops)}`,
64
67
  `footer: interactive ${yn(rt.config.interactiveFooter)} · toggle: ${rt.config.toggleShortcut}`,
65
68
  ];
66
69
  if (rt.state.activeTaskStreams.length) {
@@ -91,6 +94,13 @@ async function showConfigMenu(ctx: ExtensionCommandContext, rt: Runtime): Promis
91
94
  { value: "toolSim" as const, label: `🔧 call similarity: ${(c.toolSimilarityThreshold * 100).toFixed(0)}%`, description: "how identical tool calls must be to count as the same call" },
92
95
  { value: "toolRepeat" as const, label: `🔁 call repeats: ${c.minToolRepeatCount}+`, description: "how many times the same call must repeat before it flags" },
93
96
  { value: "resultSim" as const, label: `🧾 result similarity: ${(c.resultSimilarityThreshold * 100).toFixed(0)}%`, description: "same command + different result = progress, not a loop" },
97
+ // ── 🌀 Degenerate (intra-message meltdown) ───────────────────────
98
+ { value: "degRun" as const, label: `🌀 degenerate run: ≥ ${c.degenerateMaxRun}`, description: "identical words in a row inside ONE message/call before it counts as stuck generation (noguerol ×5145 class)" },
99
+ { value: "blk" as const, label: `🔁 block repetition: ${yn(c.detectBlockRepeats)}`, description: "ONE message replaying whole sentences/phrases (the narration loop: 'Let me start by checking…' ×5)" },
100
+ { value: "blkShare" as const, label: `🔁 block share: ≥ ${(c.blockRepeatShare * 100).toFixed(0)}%`, description: "share of replayed 5-word phrases in one message before it counts as a replay" },
101
+ { value: "degBlock" as const, label: `⛔ block repetitive bash: ${yn(c.blockDegenerateBash)}`, description: "stop a degenerate or replayed command before it executes (default on)" },
102
+ // ── 📉 Outcome (no-progress) ────────────────────────────────
103
+ { value: "outcomeMin" as const, label: `📉 no-progress after: ${c.outcomeMinRepeats}`, description: "same failing outcome repeated this many times (mutated args ≥ 85% similar) before flagging — the NFS test-A…QQQ class" },
94
104
  // ── 📋 Task streams ─────────────────────────────────────────
95
105
  { value: "streams" as const, label: `📋 task streams: ${yn(c.detectTaskStreams)}`, description: "batch work (punched_log / plan_manager / …) is not a loop" },
96
106
  { value: "streamMin" as const, label: `📋 stream min calls: ${c.taskStreamMinCalls}`, description: "calls of the same tool before a batch is recognized" },
@@ -99,6 +109,8 @@ async function showConfigMenu(ctx: ExtensionCommandContext, rt: Runtime): Promis
99
109
  { value: "text" as const, label: `📝 text: ${yn(c.detectTextLoops)}`, description: "detect repeated text messages" },
100
110
  { value: "tool" as const, label: `🔧 tools: ${yn(c.detectToolLoops)}`, description: "detect repeated tool calls" },
101
111
  { value: "think" as const, label: `🧠 thinking: ${yn(c.detectThinkingLoops)}`, description: "detect repeated internal reasoning" },
112
+ { value: "deg" as const, label: `🌀 degenerate: ${yn(c.detectDegenerate)}`, description: "detect ONE message stuck repeating a single word/token (no repeated peer needed)" },
113
+ { value: "outcome" as const, label: `📉 outcome: ${yn(c.detectOutcomeLoops)}`, description: "detect many near-identical attempts all ending in the SAME failing outcome (no progress)" },
102
114
  // ── 🧹 ──────────────────────────────────────────────────────
103
115
  { value: "reset" as const, label: "🧹 reset state", description: "clear counters and history" },
104
116
  ]);
@@ -208,6 +220,41 @@ async function showConfigMenu(ctx: ExtensionCommandContext, rt: Runtime): Promis
208
220
  if (v !== undefined) { c.resultSimilarityThreshold = v; saveConfig(c); ctx.ui.notify(`result similarity: ${(v * 100).toFixed(0)}%`, "info"); }
209
221
  break;
210
222
  }
223
+ case "degRun": {
224
+ const v = await selectFrom(ctx, "🌀 degenerate run (identical words in a row inside one payload)", [
225
+ { value: 8, label: "⚡ 8 (sensitive)" },
226
+ { value: 16, label: "🎯 16 (default)" },
227
+ { value: 24, label: "24" },
228
+ { value: 48, label: "🐢 48 (relaxed)" },
229
+ ]);
230
+ if (v !== undefined) { c.degenerateMaxRun = v; saveConfig(c); ctx.ui.notify(`degenerate run: ≥ ${v}`, "info"); }
231
+ break;
232
+ }
233
+ case "blk":
234
+ c.detectBlockRepeats = !c.detectBlockRepeats; saveConfig(c);
235
+ ctx.ui.notify(`block repetition: ${yn(c.detectBlockRepeats)}`, "info"); break;
236
+ case "blkShare": {
237
+ const v = await selectFrom(ctx, "🔁 block repetition share (replayed 5-gram positions in ONE message)", [
238
+ { value: 0.7, label: "⚡ 70% (sensitive)" },
239
+ { value: 0.85, label: "🎯 85% (default)" },
240
+ { value: 0.95, label: "🐢 95% (only near-total replays)" },
241
+ ]);
242
+ if (v !== undefined) { c.blockRepeatShare = v; saveConfig(c); ctx.ui.notify(`block share: ${(v * 100).toFixed(0)}%`, "info"); }
243
+ break;
244
+ }
245
+ case "degBlock":
246
+ c.blockDegenerateBash = !c.blockDegenerateBash; saveConfig(c);
247
+ ctx.ui.notify(`block repetitive bash: ${yn(c.blockDegenerateBash)}`, "info"); break;
248
+ case "outcomeMin": {
249
+ const v = await selectFrom(ctx, "📉 no-progress threshold (same failing outcome, near-identical args)", [
250
+ { value: 4, label: "⚡ 4 (sensitive — long experiment series get cut early)" },
251
+ { value: 6, label: "6" },
252
+ { value: 8, label: "🎯 8 (default)" },
253
+ { value: 12, label: "🐢 12 (relaxed)" },
254
+ ]);
255
+ if (v !== undefined) { c.outcomeMinRepeats = v; saveConfig(c); ctx.ui.notify(`no-progress after: ${v}`, "info"); }
256
+ break;
257
+ }
211
258
  case "streams":
212
259
  c.detectTaskStreams = !c.detectTaskStreams; saveConfig(c);
213
260
  ctx.ui.notify(`task streams: ${yn(c.detectTaskStreams)}`, "info"); break;
@@ -248,6 +295,12 @@ async function showConfigMenu(ctx: ExtensionCommandContext, rt: Runtime): Promis
248
295
  case "think":
249
296
  c.detectThinkingLoops = !c.detectThinkingLoops; saveConfig(c);
250
297
  ctx.ui.notify(`thinking: ${yn(c.detectThinkingLoops)}`, "info"); break;
298
+ case "deg":
299
+ c.detectDegenerate = !c.detectDegenerate; saveConfig(c);
300
+ ctx.ui.notify(`degenerate: ${yn(c.detectDegenerate)}`, "info"); break;
301
+ case "outcome":
302
+ c.detectOutcomeLoops = !c.detectOutcomeLoops; saveConfig(c);
303
+ ctx.ui.notify(`outcome: ${yn(c.detectOutcomeLoops)}`, "info"); break;
251
304
  case "reset":
252
305
  resetState(rt.state);
253
306
  ctx.ui.notify("🧹 state reset", "info");
@@ -280,6 +333,7 @@ export function resetState(state: AntiloopState): void {
280
333
  state.lastDetectedTurnIndex = -1;
281
334
  state.steerDelivered = false;
282
335
  state.ignoredSteerCount = 0;
336
+ state.turnSeq = 0;
283
337
  }
284
338
 
285
339
  async function runSelfTest(ctx: ExtensionCommandContext): Promise<void> {
package/src/config.ts CHANGED
@@ -17,6 +17,22 @@ export const DEFAULT_CONFIG: AntiloopConfig = {
17
17
  toolSimilarityThreshold: 0.95,
18
18
  minToolRepeatCount: 2,
19
19
  resultSimilarityThreshold: 0.8,
20
+ detectDegenerate: true,
21
+ degenerateMinTokens: 50,
22
+ degenerateMaxRun: 16,
23
+ degenerateMaxFreq: 60,
24
+ degenerateMaxShare: 0.4,
25
+ degenerateTurnWeight: 2,
26
+ blockDegenerateBash: true,
27
+ detectBlockRepeats: true,
28
+ blockMinTokens: 120,
29
+ blockNgram: 5,
30
+ blockMinRepeats: 3,
31
+ blockRepeatShare: 0.85,
32
+ detectOutcomeLoops: true,
33
+ outcomeMinRepeats: 8,
34
+ outcomeArgSimilarity: 0.85,
35
+ outcomeSigThreshold: 0.7,
20
36
  detectTaskStreams: true,
21
37
  taskStreamMinCalls: 3,
22
38
  taskStreamTwinThreshold: 0.99,