pi-antiloop 1.5.1 → 1.7.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +209 -12
- package/package.json +2 -2
- package/src/commands.ts +55 -1
- package/src/config.ts +16 -0
- package/src/detect.ts +617 -8
- package/src/index.ts +152 -5
- package/src/types.ts +71 -1
package/src/detect.ts
CHANGED
|
@@ -34,6 +34,278 @@ function opening(text: string, n = 10): string {
|
|
|
34
34
|
return normalizeText(text.split(/\s+/).slice(0, n).join(" "));
|
|
35
35
|
}
|
|
36
36
|
|
|
37
|
+
// ---------------------------------------------------------------------------
|
|
38
|
+
// Intra-message degenerate repetition (v1.6).
|
|
39
|
+
//
|
|
40
|
+
// The "noguerol ×5145" class (verified against a real session — /home/j
|
|
41
|
+
// 2026-09-09T15-43: ONE 46 KB bash call whose SSH username list repeats a
|
|
42
|
+
// single word 5145 times, a run of 5140 — 99% of the payload). A model whose
|
|
43
|
+
// decoder anchors on a token stops producing NEW output: it repeats the same
|
|
44
|
+
// word hundreds of times INSIDE one message or tool call. The cross-message
|
|
45
|
+
// detectors (text / tool / thinking / structural) all need >= 2 similar
|
|
46
|
+
// messages and cannot see this — the meltdown happened exactly once, inside a
|
|
47
|
+
// single call, and antiloop stayed silent until the user ESC'd.
|
|
48
|
+
//
|
|
49
|
+
// This detector is self-contained: it flags the FIRST such payload, no peer
|
|
50
|
+
// message required, and it is cheap enough to run at message_end (before the
|
|
51
|
+
// tool calls execute) and on every bash tool_call (blocking gate).
|
|
52
|
+
// ---------------------------------------------------------------------------
|
|
53
|
+
|
|
54
|
+
/** Longest run / top frequency of ONE repeated word inside a payload. */
|
|
55
|
+
export interface DegenerateInfo {
|
|
56
|
+
token: string;
|
|
57
|
+
freq: number;
|
|
58
|
+
maxRun: number;
|
|
59
|
+
total: number;
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
export interface DegenerateHit extends DegenerateInfo {
|
|
63
|
+
where: string;
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
/** Ignore 1-letter tokens as candidates (JSON keys like {"a":1} must not flag). */
|
|
67
|
+
const MIN_TOKEN_LEN = 2;
|
|
68
|
+
|
|
69
|
+
/**
|
|
70
|
+
* Detect pathological single-word repetition in a payload (assistant text or a
|
|
71
|
+
* JSON.stringify'd tool-call argument). Returns the repeated word with its
|
|
72
|
+
* count, longest consecutive run and payload size, or undefined when the
|
|
73
|
+
* payload is normal.
|
|
74
|
+
*
|
|
75
|
+
* Tokenization: runs of unicode letters, lowercased. Stored tool args are
|
|
76
|
+
* JSON-escaped (real newlines arrived as the two characters `\n`), so escaped
|
|
77
|
+
* whitespace is normalized back to a separator first — a word list written
|
|
78
|
+
* across lines must still tokenize word by word. Digits/punctuation/code
|
|
79
|
+
* symbols split tokens instead of polluting them.
|
|
80
|
+
*
|
|
81
|
+
* Signals (both are conclusive for generation quality):
|
|
82
|
+
* - a run of >= degenerateMaxRun consecutive identical words, or
|
|
83
|
+
* - one word occurring >= degenerateMaxFreq times with >= degenerateMaxShare
|
|
84
|
+
* of all tokens (catches interleaved "A B A B" meltdowns with no run).
|
|
85
|
+
* A payload must have >= degenerateMinTokens tokens to be scanned.
|
|
86
|
+
*/
|
|
87
|
+
export function findDegenerateRepetition(
|
|
88
|
+
text: string,
|
|
89
|
+
config: AntiloopConfig,
|
|
90
|
+
): DegenerateInfo | undefined {
|
|
91
|
+
let s = String(text).replace(/\\+[nrt]/g, " ").toLowerCase();
|
|
92
|
+
const raw = s.match(/[\p{L}]+/gu);
|
|
93
|
+
if (!raw) return undefined;
|
|
94
|
+
const total = raw.length;
|
|
95
|
+
|
|
96
|
+
let maxRun = 0;
|
|
97
|
+
let runToken = "";
|
|
98
|
+
let prev = "";
|
|
99
|
+
let run = 0;
|
|
100
|
+
const freq = new Map<string, number>();
|
|
101
|
+
for (const t of raw) {
|
|
102
|
+
if (t.length < MIN_TOKEN_LEN) {
|
|
103
|
+
run = 0;
|
|
104
|
+
prev = "";
|
|
105
|
+
continue;
|
|
106
|
+
}
|
|
107
|
+
freq.set(t, (freq.get(t) ?? 0) + 1);
|
|
108
|
+
if (t === prev) run++;
|
|
109
|
+
else {
|
|
110
|
+
run = 1;
|
|
111
|
+
prev = t;
|
|
112
|
+
}
|
|
113
|
+
if (run > maxRun) {
|
|
114
|
+
maxRun = run;
|
|
115
|
+
runToken = t;
|
|
116
|
+
}
|
|
117
|
+
}
|
|
118
|
+
let topToken = "";
|
|
119
|
+
let topFreq = 0;
|
|
120
|
+
for (const [t, c] of freq) {
|
|
121
|
+
if (c > topFreq) {
|
|
122
|
+
topFreq = c;
|
|
123
|
+
topToken = t;
|
|
124
|
+
}
|
|
125
|
+
}
|
|
126
|
+
if (total >= config.degenerateMinTokens) {
|
|
127
|
+
if (maxRun >= config.degenerateMaxRun) {
|
|
128
|
+
return { token: runToken, freq: freq.get(runToken) ?? topFreq, maxRun, total };
|
|
129
|
+
}
|
|
130
|
+
if (topFreq >= config.degenerateMaxFreq && topFreq / total >= config.degenerateMaxShare) {
|
|
131
|
+
return { token: topToken, freq: topFreq, maxRun, total };
|
|
132
|
+
}
|
|
133
|
+
}
|
|
134
|
+
|
|
135
|
+
// No-space meltdown: one giant periodic token ("noguerolnoguerol…" with all
|
|
136
|
+
// separators stripped) is a perfect power of a short motif. Independent of
|
|
137
|
+
// the token-count gate: a single 1200+ char token has no word runs at all.
|
|
138
|
+
if (raw.length <= 2) {
|
|
139
|
+
const single = raw.join("");
|
|
140
|
+
const len = single.length;
|
|
141
|
+
if (len >= 1200) {
|
|
142
|
+
for (let p = 3; p <= 200 && p * 12 <= len; p++) {
|
|
143
|
+
if (len % p) continue;
|
|
144
|
+
const motif = single.slice(0, p);
|
|
145
|
+
let ok = true;
|
|
146
|
+
for (let i = p; i < len; i += p) {
|
|
147
|
+
if (!single.startsWith(motif, i)) {
|
|
148
|
+
ok = false;
|
|
149
|
+
break;
|
|
150
|
+
}
|
|
151
|
+
}
|
|
152
|
+
if (ok) {
|
|
153
|
+
const repeats = len / p;
|
|
154
|
+
if (repeats >= 12) {
|
|
155
|
+
return { token: motif.slice(0, 40), freq: repeats, maxRun: repeats, total: repeats };
|
|
156
|
+
}
|
|
157
|
+
}
|
|
158
|
+
}
|
|
159
|
+
}
|
|
160
|
+
}
|
|
161
|
+
return undefined;
|
|
162
|
+
}
|
|
163
|
+
|
|
164
|
+
/** Scan one assistant message (text + each tool-call argument) for a meltdown. */
|
|
165
|
+
export function scanMessageDegenerate(
|
|
166
|
+
content: string,
|
|
167
|
+
toolCalls: TrackedToolCall[] | undefined,
|
|
168
|
+
config: AntiloopConfig,
|
|
169
|
+
): DegenerateHit | undefined {
|
|
170
|
+
if (!config.detectDegenerate) return undefined;
|
|
171
|
+
if (content && content.length) {
|
|
172
|
+
const d = findDegenerateRepetition(content, config);
|
|
173
|
+
if (d) return { ...d, where: "message text" };
|
|
174
|
+
}
|
|
175
|
+
for (const tc of toolCalls ?? []) {
|
|
176
|
+
if (!tc.args) continue;
|
|
177
|
+
const d = findDegenerateRepetition(tc.args, config);
|
|
178
|
+
if (d) return { ...d, where: tc.name === "bash" ? "bash command" : `args(${tc.name})` };
|
|
179
|
+
}
|
|
180
|
+
return undefined;
|
|
181
|
+
}
|
|
182
|
+
|
|
183
|
+
export function degenerateDescription(hit: DegenerateHit): string {
|
|
184
|
+
const share = hit.total ? Math.round((hit.freq / hit.total) * 100) : 100;
|
|
185
|
+
return `${hit.where}: degenerate repetition — "${hit.token}" ×${hit.freq} (${share}% of ${hit.total} tokens, longest run ${hit.maxRun})`;
|
|
186
|
+
}
|
|
187
|
+
|
|
188
|
+
// ---------------------------------------------------------------------------
|
|
189
|
+
// Intra-message BLOCK repetition (v1.7).
|
|
190
|
+
//
|
|
191
|
+
// The "narration loop" class (verified against a real coding session: ONE
|
|
192
|
+
// assistant message replaying ~5 near-verbatim cycles of "Let me start by
|
|
193
|
+
// checking the environment and the current state of the repository… / I'll run
|
|
194
|
+
// several independent checks in parallel. / Let me begin the S0 development…").
|
|
195
|
+
// Every pre-existing detector stayed silent: text/tool/thinking/structural all
|
|
196
|
+
// compare ACROSS messages (the model produced one message, no peer), and the
|
|
197
|
+
// degenerate detector only watches ONE word repeated hundreds of times, not a
|
|
198
|
+
// whole sentence/paragraph replayed. The signal here is phrase-level: a sliding
|
|
199
|
+
// window of blockNgram-word n-grams over the normalized payload; if almost ALL
|
|
200
|
+
// of those n-grams recur and the most frequent one recurs ≥ blockMinRepeats
|
|
201
|
+
// times, the generation is replaying itself instead of advancing.
|
|
202
|
+
//
|
|
203
|
+
// Deliberately conservative: only a payload where ≥ blockRepeatShare (default
|
|
204
|
+
// 85%) of 5-gram positions repeat qualifies. Ordinary prose — even long,
|
|
205
|
+
// structured docs — sits far below (README/pi.md paragraphs: ≤ 0.11), while
|
|
206
|
+
// 3+ replays of a narration block reach 0.98–1.0. Repeat-linked code/log lines
|
|
207
|
+
// that share a template are NOT flagged because their n-grams carry the varying
|
|
208
|
+
// digits and differ.
|
|
209
|
+
// ---------------------------------------------------------------------------
|
|
210
|
+
|
|
211
|
+
/** Recursive n-gram coverage of one payload. */
|
|
212
|
+
export interface BlockRepeatInfo {
|
|
213
|
+
/** Recurring n-gram positions / total n-gram positions (0..1). */
|
|
214
|
+
ratio: number;
|
|
215
|
+
/** Occurrences of the MOST repeated n-gram. */
|
|
216
|
+
repeats: number;
|
|
217
|
+
/** Total normalized words in the payload. */
|
|
218
|
+
tokens: number;
|
|
219
|
+
/** The n used (words per window). */
|
|
220
|
+
ngram: number;
|
|
221
|
+
/** The most repeated n-gram, for the description. */
|
|
222
|
+
sample: string;
|
|
223
|
+
}
|
|
224
|
+
|
|
225
|
+
export interface BlockRepeatHit extends BlockRepeatInfo {
|
|
226
|
+
where: string;
|
|
227
|
+
}
|
|
228
|
+
|
|
229
|
+
/**
|
|
230
|
+
* Detect phrase/block-level self-repetition inside ONE payload. Returns the
|
|
231
|
+
* coverage ratio with the most repeated n-gram, or undefined for normal text.
|
|
232
|
+
*
|
|
233
|
+
* Reads the payload as a stream of lowercase alphanumeric words (code symbols
|
|
234
|
+
* and punctuation are separators, so stored-JSON escaping cannot glue tokens).
|
|
235
|
+
* Coverage counts an n-gram position as repeated when the SAME n-word sequence
|
|
236
|
+
* (digits included, so templated lines with varying numbers do not count)
|
|
237
|
+
* occurs somewhere else in the payload.
|
|
238
|
+
*/
|
|
239
|
+
export function findRepetitiveBlock(text: string, config: AntiloopConfig): BlockRepeatInfo | undefined {
|
|
240
|
+
const words = String(text).toLowerCase().match(/[\p{L}\p{N}]+/gu);
|
|
241
|
+
if (!words) return undefined;
|
|
242
|
+
const total = words.length;
|
|
243
|
+
if (total < config.blockMinTokens) return undefined;
|
|
244
|
+
const n = Math.max(2, config.blockNgram);
|
|
245
|
+
const minRepeats = Math.max(2, config.blockMinRepeats);
|
|
246
|
+
if (total < n * minRepeats) return undefined;
|
|
247
|
+
|
|
248
|
+
const counts = new Map<string, number>();
|
|
249
|
+
const grams: string[] = [];
|
|
250
|
+
for (let i = 0; i + n <= total; i++) {
|
|
251
|
+
const g = words.slice(i, i + n).join(" ");
|
|
252
|
+
grams.push(g);
|
|
253
|
+
counts.set(g, (counts.get(g) ?? 0) + 1);
|
|
254
|
+
}
|
|
255
|
+
if (!grams.length) return undefined;
|
|
256
|
+
|
|
257
|
+
let covered = 0;
|
|
258
|
+
let topKey = "";
|
|
259
|
+
let top = 0;
|
|
260
|
+
for (const g of grams) {
|
|
261
|
+
const c = counts.get(g)!;
|
|
262
|
+
if (c > top) {
|
|
263
|
+
top = c;
|
|
264
|
+
topKey = g;
|
|
265
|
+
}
|
|
266
|
+
if (c >= 2) covered++;
|
|
267
|
+
}
|
|
268
|
+
// At least one n-word phrase must recur blockMinRepeats times (a single
|
|
269
|
+
// echo is a restatement, not a replay) and the recurrence must dominate.
|
|
270
|
+
if (top < minRepeats) return undefined;
|
|
271
|
+
const ratio = covered / grams.length;
|
|
272
|
+
if (ratio < config.blockRepeatShare) return undefined;
|
|
273
|
+
return { ratio, repeats: top, tokens: total, ngram: n, sample: topKey };
|
|
274
|
+
}
|
|
275
|
+
|
|
276
|
+
/** Scan one assistant message (text + each tool-call argument) for a replay. */
|
|
277
|
+
export function scanMessageBlock(
|
|
278
|
+
content: string,
|
|
279
|
+
toolCalls: TrackedToolCall[] | undefined,
|
|
280
|
+
config: AntiloopConfig,
|
|
281
|
+
): BlockRepeatHit | undefined {
|
|
282
|
+
if (!config.detectBlockRepeats) return undefined;
|
|
283
|
+
if (content && content.length) {
|
|
284
|
+
const b = findRepetitiveBlock(content, config);
|
|
285
|
+
if (b) return { ...b, where: "message text" };
|
|
286
|
+
}
|
|
287
|
+
for (const tc of toolCalls ?? []) {
|
|
288
|
+
if (!tc.args) continue;
|
|
289
|
+
const b = findRepetitiveBlock(tc.args, config);
|
|
290
|
+
if (b) return { ...b, where: tc.name === "bash" ? "bash command" : `args(${tc.name})` };
|
|
291
|
+
}
|
|
292
|
+
return undefined;
|
|
293
|
+
}
|
|
294
|
+
|
|
295
|
+
export function blockRepeatDescription(hit: BlockRepeatHit): string {
|
|
296
|
+
const pct = Math.round(hit.ratio * 100);
|
|
297
|
+
return `${hit.where}: block repetition — ${pct}% of ${hit.tokens} tokens replay repeated ${hit.ngram}-word phrases ("${hit.sample}" ×${hit.repeats})`;
|
|
298
|
+
}
|
|
299
|
+
|
|
300
|
+
/** Consecutive-detection weight of a detection turn. The strong
|
|
301
|
+
* self-contained signals (degenerate meltdown, proven no-progress outcome run)
|
|
302
|
+
* add degenerateTurnWeight (default 2) points so the FIRST one already reaches
|
|
303
|
+
* the warning level and escalation is fast on repeat. */
|
|
304
|
+
export function detectionTurnWeight(detections: LoopDetection[], config: AntiloopConfig): number {
|
|
305
|
+
const strong = detections.some((d) => d.type === "degenerate" || d.type === "block" || d.type === "outcome");
|
|
306
|
+
return strong ? Math.max(1, config.degenerateTurnWeight) : 1;
|
|
307
|
+
}
|
|
308
|
+
|
|
37
309
|
function similarity(a: string, b: string): number {
|
|
38
310
|
if (a.length < MIN_CONTENT_LENGTH || b.length < MIN_CONTENT_LENGTH) return 0;
|
|
39
311
|
if (a === b) return 1;
|
|
@@ -74,7 +346,18 @@ export function resultFingerprint(
|
|
|
74
346
|
}
|
|
75
347
|
const norm = normalizeText(text);
|
|
76
348
|
if (!norm.length) return undefined;
|
|
77
|
-
|
|
349
|
+
// A tool run can FAIL while isError stays false (the NFS mount errors: the
|
|
350
|
+
// ssh pipeline exits 0 after grep/echo, rc=32 lives inside the output). The
|
|
351
|
+
// tail-only fingerprint would cut the failure markers away AND the tail is
|
|
352
|
+
// noisy (journalctl timestamps differ per attempt), so for failures we keep
|
|
353
|
+
// a SHORT ERROR SIGNATURE around the first failure marker — it repeats
|
|
354
|
+
// verbatim across attempts of the same failure and makes same-outcome
|
|
355
|
+
// comparisons robust. Formats: "err|…" / "ok|fail|sig|…|tail" / "ok|…".
|
|
356
|
+
const failed = FAIL_MARKERS.test(norm);
|
|
357
|
+
if (!failed) return `${isError ? "err" : "ok"}|${norm.slice(-400)}`;
|
|
358
|
+
const at = norm.search(FAIL_MARKERS);
|
|
359
|
+
const sig = norm.slice(Math.max(0, at - 60), at + 160);
|
|
360
|
+
return `${isError ? "err" : "ok"}|fail|${sig}|${norm.slice(-400)}`;
|
|
78
361
|
}
|
|
79
362
|
|
|
80
363
|
/** Same outcome = identical fingerprint, or high similarity of the tails. */
|
|
@@ -89,6 +372,44 @@ function sameOutcome(a: string, b: string, threshold: number): boolean {
|
|
|
89
372
|
return s >= threshold && s > 0;
|
|
90
373
|
}
|
|
91
374
|
|
|
375
|
+
// Failure markers on a NORMALIZED fingerprint tail (lowercased, punctuation
|
|
376
|
+
// stripped — "rc=32" arrives as "rc32"). A conservative "this attempt failed"
|
|
377
|
+
// test: rc ≠ 0, denials, missing files, refusals, syntax crashes… Generic
|
|
378
|
+
// words like "error"/"failed" are intentionally NOT markers (legit outputs
|
|
379
|
+
// like "0 failed, 12 passed" must not count as failures).
|
|
380
|
+
const FAIL_MARKERS =
|
|
381
|
+
/\b(denied|no such file|not found|cannot|unable|refused|syntax error|timed out|timeout|exception|traceback|fatal|core dumped|rc\s*[1-9]\d*)\b/i;
|
|
382
|
+
|
|
383
|
+
/** True when a captured result fingerprint represents a FAILED attempt.
|
|
384
|
+
* Used by the no-progress outcome detector so only repeated FAILURES (not
|
|
385
|
+
* repeated identical successes, which are the norm for task-stream batches
|
|
386
|
+
* like "logged" or file writes) count as a stuck loop. */
|
|
387
|
+
export function isFailResult(fp: string | undefined): boolean {
|
|
388
|
+
if (!fp) return false;
|
|
389
|
+
if (fp.startsWith("err|") || fp.startsWith("ok|fail|")) return true;
|
|
390
|
+
return FAIL_MARKERS.test(fp);
|
|
391
|
+
}
|
|
392
|
+
|
|
393
|
+
/** The short error signature embedded in a failure fingerprint
|
|
394
|
+
* ("ok|fail|SIG|tail"), if present. */
|
|
395
|
+
function failSig(fp: string): string | undefined {
|
|
396
|
+
const m = /^(?:err|ok)\|fail\|(.*?)\|/.exec(fp);
|
|
397
|
+
return m ? m[1] : undefined;
|
|
398
|
+
}
|
|
399
|
+
|
|
400
|
+
/** Same-FAILURE comparison for the outcome detector. Prefers the embedded error
|
|
401
|
+
* signatures (they repeat across attempts of the same failure) with digits
|
|
402
|
+
* stripped — timestamps, PIDs and rc values are noise; the target words that
|
|
403
|
+
* legitimately vary between attempts (Javi vs Compartido) survive but the
|
|
404
|
+
* threshold is looser than the veto threshold on purpose. Falls back to the
|
|
405
|
+
* full fingerprints when no signature is present. */
|
|
406
|
+
function sameFailure(a: string, b: string, threshold: number): boolean {
|
|
407
|
+
const sa = failSig(a);
|
|
408
|
+
const sb = failSig(b);
|
|
409
|
+
if (sa && sb) return sameOutcome(sa.replace(/\d+/g, ""), sb.replace(/\d+/g, ""), threshold);
|
|
410
|
+
return sameOutcome(a, b, threshold);
|
|
411
|
+
}
|
|
412
|
+
|
|
92
413
|
function toolCallsSimilar(
|
|
93
414
|
c1: TrackedToolCall[],
|
|
94
415
|
c2: TrackedToolCall[],
|
|
@@ -177,11 +498,45 @@ export function detectTaskStreams(
|
|
|
177
498
|
export function detectLoops(state: AntiloopState, config: AntiloopConfig): LoopDetection[] {
|
|
178
499
|
const out: LoopDetection[] = [];
|
|
179
500
|
const msgs = state.recentMessages;
|
|
180
|
-
if (msgs.length <
|
|
501
|
+
if (msgs.length < 1) return out;
|
|
181
502
|
const start = Math.max(0, msgs.length - config.detectionWindow);
|
|
182
503
|
const win = msgs.slice(start);
|
|
183
504
|
const now = Date.now();
|
|
184
505
|
|
|
506
|
+
// Intra-message repetition fires on a SINGLE pathological message (no peer
|
|
507
|
+
// needed) and is independent of the task-stream batch gate: a meltdown is a
|
|
508
|
+
// meltdown even mid-batch. Two flavors: degenerate (one word ×hundreds) and
|
|
509
|
+
// block (whole sentences/phrases replayed — the narration-loop class).
|
|
510
|
+
if (config.detectDegenerate) {
|
|
511
|
+
const last = win[win.length - 1];
|
|
512
|
+
const hit = scanMessageDegenerate(last.content, last.toolCalls, config);
|
|
513
|
+
if (hit) {
|
|
514
|
+
out.push({
|
|
515
|
+
type: "degenerate",
|
|
516
|
+
similarity: 1,
|
|
517
|
+
messageIndices: [msgs.length - 1],
|
|
518
|
+
description: degenerateDescription(hit),
|
|
519
|
+
timestamp: now,
|
|
520
|
+
});
|
|
521
|
+
}
|
|
522
|
+
}
|
|
523
|
+
|
|
524
|
+
if (config.detectBlockRepeats) {
|
|
525
|
+
const last = win[win.length - 1];
|
|
526
|
+
const hit = scanMessageBlock(last.content, last.toolCalls, config);
|
|
527
|
+
if (hit) {
|
|
528
|
+
out.push({
|
|
529
|
+
type: "block",
|
|
530
|
+
// 0.99 => isVerbatimRepeat(): after a force break, a replayed block
|
|
531
|
+
// is the model ignoring the break and escalates to the hard stop.
|
|
532
|
+
similarity: 0.99,
|
|
533
|
+
messageIndices: [msgs.length - 1],
|
|
534
|
+
description: blockRepeatDescription(hit),
|
|
535
|
+
timestamp: now,
|
|
536
|
+
});
|
|
537
|
+
}
|
|
538
|
+
}
|
|
539
|
+
|
|
185
540
|
// Task-stream gate: if the window is a homogeneous batch (same extension
|
|
186
541
|
// tool called with DISTINCT content ≥ taskStreamMinCalls times), that tool
|
|
187
542
|
// is exempt from tool-loop detection, and text/thinking/structural
|
|
@@ -264,6 +619,68 @@ export function detectLoops(state: AntiloopState, config: AntiloopConfig): LoopD
|
|
|
264
619
|
}
|
|
265
620
|
}
|
|
266
621
|
|
|
622
|
+
// -------------------------------------------------------------------
|
|
623
|
+
// No-progress outcome runs (v1.6.1).
|
|
624
|
+
//
|
|
625
|
+
// The NFS-test session (/home/j 2026-09-09, rows 95–249): ~90 mutated
|
|
626
|
+
// re-runs of the SAME experiment (sshpass+sudo+exportfs+mount, labels
|
|
627
|
+
// "test A"…"test QQQ"), every one failing identically (rc=32 / access
|
|
628
|
+
// denied). The tool-loop detector is blind to it BY DESIGN: args mutate
|
|
629
|
+
// every turn (mean adjacent trigram similarity 0.93, but the label always
|
|
630
|
+
// changes) so the same call never recurs >= minToolRepeatCount times, and
|
|
631
|
+
// identical results only VETO tool loops — nothing uses "same outcome
|
|
632
|
+
// repeated" as a positive signal. A human sees it instantly: many attempts,
|
|
633
|
+
// same wall, zero progress.
|
|
634
|
+
//
|
|
635
|
+
// Signal: the LAST turn's single tool call has a captured result, and at
|
|
636
|
+
// least outcomeMinRepeats PRIOR single-call turns (after the last real user
|
|
637
|
+
// message — an autonomous stretch, not user-steered iteration) share BOTH
|
|
638
|
+
// args >= outcomeArgSimilarity (the same experiment reshuffled) AND the same
|
|
639
|
+
// outcome (>= resultSimilarityThreshold). Legit work is untouched: distinct
|
|
640
|
+
// operations fail with distinct output; converging sweeps change outcome;
|
|
641
|
+
// batch/stream messages are excluded; a success interspersed resets the
|
|
642
|
+
// class. similarity 0.99 => after a force break, further same-outcome turns
|
|
643
|
+
// count as "ignoring the break" (isVerbatimRepeat) and escalate to the hard
|
|
644
|
+
// stop, exactly like verbatim tool loops.
|
|
645
|
+
// -------------------------------------------------------------------
|
|
646
|
+
if (config.detectOutcomeLoops) {
|
|
647
|
+
const last = win[win.length - 1];
|
|
648
|
+
const lastCalls = last.toolCalls;
|
|
649
|
+
const afterUser = state.lastUserMessageTime;
|
|
650
|
+
if (lastCalls && lastCalls.length === 1) {
|
|
651
|
+
const lc = lastCalls[0];
|
|
652
|
+
// Failure gate: only repeated FAILURES prove no progress. Task-stream
|
|
653
|
+
// batches (punched_log appends, obsidian/file writes) legitimately
|
|
654
|
+
// produce the SAME OK outcome every call — they must never count. And
|
|
655
|
+
// batch messages must NOT be skipped here: a "bash ×N stream" with
|
|
656
|
+
// identical failures is exactly the no-progress loop to catch.
|
|
657
|
+
if (lc.result && isFailResult(lc.result)) {
|
|
658
|
+
let matches = 0;
|
|
659
|
+
for (let i = 0; i < win.length - 1; i++) {
|
|
660
|
+
const m = win[i];
|
|
661
|
+
if (m.timestamp <= afterUser) continue; // user-steered turns don't count
|
|
662
|
+
const prev = m.toolCalls;
|
|
663
|
+
if (!prev || prev.length !== 1) continue;
|
|
664
|
+
const pc = prev[0];
|
|
665
|
+
if (pc.name !== lc.name || !pc.result) continue;
|
|
666
|
+
if (!sameFailure(lc.result, pc.result, config.outcomeSigThreshold)) continue;
|
|
667
|
+
if (!argsTwin(lc.args, pc.args, config.outcomeArgSimilarity)) continue;
|
|
668
|
+
matches++;
|
|
669
|
+
}
|
|
670
|
+
if (matches >= config.outcomeMinRepeats) {
|
|
671
|
+
out.push({
|
|
672
|
+
type: "outcome",
|
|
673
|
+
similarity: 0.99,
|
|
674
|
+
messageIndices: [msgs.length - 1],
|
|
675
|
+
description:
|
|
676
|
+
`no progress: ${matches + 1} near-identical ${lc.name} attempts (args ≥ ${(config.outcomeArgSimilarity * 100).toFixed(0)}% similar) with the same failing outcome — “${lc.result.slice(0, 90)}”`,
|
|
677
|
+
timestamp: now,
|
|
678
|
+
});
|
|
679
|
+
}
|
|
680
|
+
}
|
|
681
|
+
}
|
|
682
|
+
}
|
|
683
|
+
|
|
267
684
|
if (config.detectThinkingLoops) {
|
|
268
685
|
const last = win[win.length - 1];
|
|
269
686
|
if (last.thinking && last.thinking.length > 50) {
|
|
@@ -297,11 +714,13 @@ export function nextLevel(consecutiveDetections: number, config: AntiloopConfig)
|
|
|
297
714
|
}
|
|
298
715
|
|
|
299
716
|
/** True when detections prove the model repeated a message/tool call essentially
|
|
300
|
-
* verbatim (≥98% text similarity or an identical tool-loop)
|
|
301
|
-
*
|
|
302
|
-
*
|
|
303
|
-
*
|
|
304
|
-
*
|
|
717
|
+
* verbatim (≥98% text similarity or an identical tool-loop), OR produced a
|
|
718
|
+
* degenerate meltdown, OR kept re-running the same experiment with the same
|
|
719
|
+
* failing outcome (no-progress, sim 0.99). Weaker signals (thinking echoes,
|
|
720
|
+
* structural repeated openings at 90%) do NOT count — a model that only *thinks*
|
|
721
|
+
* in circles but varies its actual output is still making an attempt and must
|
|
722
|
+
* not be hard-stopped. Used post-force-break: only verbatim repeats and proven
|
|
723
|
+
* no-progress repeats show the model ignored the break instruction. */
|
|
305
724
|
export function isVerbatimRepeat(detections: LoopDetection[]): boolean {
|
|
306
725
|
return detections.some((d) => d.type !== "thinking" && d.similarity >= 0.98);
|
|
307
726
|
}
|
|
@@ -383,11 +802,15 @@ export function runSelfTest(): string[] {
|
|
|
383
802
|
detectTextLoops: true, notifyOnDetection: true, maxHistoryEntries: 100,
|
|
384
803
|
detectionWindow: 10, interactiveFooter: true, toggleShortcut: "esc+a",
|
|
385
804
|
detectTaskStreams: true, taskStreamMinCalls: 3, taskStreamTwinThreshold: 0.99,
|
|
805
|
+
detectDegenerate: true, degenerateMinTokens: 50, degenerateMaxRun: 16,
|
|
806
|
+
degenerateMaxFreq: 60, degenerateMaxShare: 0.4, degenerateTurnWeight: 2, blockDegenerateBash: true,
|
|
807
|
+
detectBlockRepeats: true, blockMinTokens: 120, blockNgram: 5, blockMinRepeats: 3, blockRepeatShare: 0.85,
|
|
808
|
+
detectOutcomeLoops: true, outcomeMinRepeats: 8, outcomeArgSimilarity: 0.85, outcomeSigThreshold: 0.7,
|
|
386
809
|
};
|
|
387
810
|
const asState = (recentMessages: TrackedMessage[]): AntiloopState =>
|
|
388
811
|
({ recentMessages, detections: [], activeTaskStreams: [], currentLevel: 0,
|
|
389
812
|
consecutiveDetections: 0, inForcedBreak: false, totalDetections: 0,
|
|
390
|
-
lastUserMessageTime: 0, lastDetectedTurnIndex: -1,
|
|
813
|
+
lastUserMessageTime: 0, lastDetectedTurnIndex: -1, turnSeq: 0,
|
|
391
814
|
steerDelivered: false, ignoredSteerCount: 0 });
|
|
392
815
|
const NARR = "Now I will append the next decision entry to the project memory document so we keep the context.";
|
|
393
816
|
|
|
@@ -443,5 +866,191 @@ export function runSelfTest(): string[] {
|
|
|
443
866
|
out.push(`thinking-only 1.00 → ${isVerbatimRepeat(dl("thinking", 1)) ? "yes" : "no"} (exp no — output varies) ${!isVerbatimRepeat(dl("thinking", 1)) ? "✅" : "❌"}`);
|
|
444
867
|
out.push(`structural 0.90 → ${isVerbatimRepeat(dl("structural", 0.9)) ? "yes" : "no"} (exp no) ${!isVerbatimRepeat(dl("structural", 0.9)) ? "✅" : "❌"}`);
|
|
445
868
|
|
|
869
|
+
// --- v1.6: intra-message degenerate repetition (single-message meltdown) ---
|
|
870
|
+
// Regression: the real session /home/j 2026-09-09T15-43 — ONE 46 KB bash call
|
|
871
|
+
// whose username list repeats "noguerol" 5145 times (run of 5140). Every
|
|
872
|
+
// cross-message detector needs a peer message and stayed silent; the
|
|
873
|
+
// degenerate scan must fire on the FIRST such message, alone in the window.
|
|
874
|
+
const argJson = (cmd: string) => JSON.stringify({ command: cmd }); // stored args form
|
|
875
|
+
const meltdownCmd =
|
|
876
|
+
`echo "=== brute usernames with petete pw ==="; for u in noguerol noguerol@ j javi javi@ root petete ${`noguerol `.repeat(400)}; do :; done`;
|
|
877
|
+
const meltMsg = mk("", [{ name: "bash", args: argJson(meltdownCmd) }]);
|
|
878
|
+
const mDet = detectLoops(asState([meltMsg]), tcfg);
|
|
879
|
+
const mHit = mDet.find((d) => d.type === "degenerate");
|
|
880
|
+
out.push(`degenerate first sight → ${mHit ? `degenerate (${mHit.description})` : "no"} (exp degenerate — was the miss) ${mHit ? "✅" : "❌"}`);
|
|
881
|
+
|
|
882
|
+
// Legit payloads must NOT flag: short commands are under minTokens, real
|
|
883
|
+
// scripts never repeat one word 16× in a row.
|
|
884
|
+
const leg1 = findDegenerateRepetition(sweepRun1, tcfg);
|
|
885
|
+
const leg2 = findDegenerateRepetition("for i in 1 2 3; do echo step $i; done", tcfg);
|
|
886
|
+
const leg3 = findDegenerateRepetition(
|
|
887
|
+
"set -euo pipefail; mkdir -p build tmp dist logs data assets src test docs lib bin etc usr var opt srv && " +
|
|
888
|
+
"cp -r config.yaml README.md LICENSE package.json tsconfig.json src lib test docs assets && " +
|
|
889
|
+
"chmod +x scripts/deploy.sh scripts/backup.sh scripts/monitor.sh && " +
|
|
890
|
+
"./scripts/deploy.sh --env production --region eu-west-1 --tag v1.2.3 --dry-run false > deploy.log 2>&1 || echo deploy failed",
|
|
891
|
+
tcfg,
|
|
892
|
+
);
|
|
893
|
+
out.push(`degenerate legit cmd → ${leg1 || leg2 || leg3 ? "flag" : "ok"} (exp ok) ${!leg1 && !leg2 && !leg3 ? "✅" : "❌"}`);
|
|
894
|
+
|
|
895
|
+
// Interleaved meltdown ("A B A B…") has no long run — caught via freq/share.
|
|
896
|
+
const inter = findDegenerateRepetition("noguerol petete ".repeat(150), tcfg);
|
|
897
|
+
out.push(`degenerate interleaved → ${inter ? `flag (${inter.token} ×${inter.freq})` : "no"} (exp flag — freq/share clause) ${inter ? "✅" : "❌"}`);
|
|
898
|
+
|
|
899
|
+
// Word list written ACROSS lines: separators are the literal "\n" escapes
|
|
900
|
+
// inside the stored JSON args — must still tokenize word by word.
|
|
901
|
+
const acrossLines = argJson("for u in " + "noguerol\n".repeat(120) + "done");
|
|
902
|
+
const linesHit = findDegenerateRepetition(acrossLines, tcfg);
|
|
903
|
+
out.push(`degenerate across \n → ${linesHit ? `flag (${linesHit.token} ×${linesHit.freq})` : "no"} (exp flag — escaped newlines) ${linesHit ? "✅" : "❌"}`);
|
|
904
|
+
|
|
905
|
+
// No-space giant token ("noguerol" glued) — perfect-power clause.
|
|
906
|
+
const glued = findDegenerateRepetition("noguerol".repeat(300), tcfg);
|
|
907
|
+
out.push(`degenerate glued token → ${glued ? `flag (${glued.token} ×${glued.freq})` : "no"} (exp flag — perfect power) ${glued ? "✅" : "❌"}`);
|
|
908
|
+
|
|
909
|
+
// Turn weight: one degenerate turn = 2 consecutive points (warning on first
|
|
910
|
+
// sight), normal turns stay at 1.
|
|
911
|
+
const wDeg = detectionTurnWeight(mDet, tcfg);
|
|
912
|
+
const wTxt = detectionTurnWeight(dl("text", 0.8), tcfg);
|
|
913
|
+
out.push(`degenerate turn weight → degenerate ${wDeg}, text ${wTxt} (exp 2, 1) ${wDeg === 2 && wTxt === 1 ? "✅" : "❌"}`);
|
|
914
|
+
|
|
915
|
+
// --- v1.7: intra-message BLOCK repetition (narration loops) ---
|
|
916
|
+
// Regression: a real coding session where ONE assistant message replayed the
|
|
917
|
+
// same ~5 sentences in a loop ("Let me start by checking the environment…" /
|
|
918
|
+
// "I'll run several independent checks in parallel." / "Let me begin the S0
|
|
919
|
+
// development…"). Cross-message text/tool/thinking detectors need a peer and
|
|
920
|
+
// stayed silent; the degenerate scan only watches a single word. The block
|
|
921
|
+
// detector must fire on the message itself.
|
|
922
|
+
const NARRV = [
|
|
923
|
+
"Let me start by checking the environment and the current state of the repository, then set up a plan for the S0 slice and begin building.",
|
|
924
|
+
"I'll run several independent checks in parallel.",
|
|
925
|
+
"Let me begin the S0 development. First, reconnaissance of the environment and current repo state.",
|
|
926
|
+
"Let me check what's available in the environment (node, package managers, network, postgres) and the current repo state, then set up the S0 plan and start building the monorepo scaffold.",
|
|
927
|
+
"I'll run a batch of independent environment checks first.",
|
|
928
|
+
];
|
|
929
|
+
const cycles = (k: number): string => {
|
|
930
|
+
let s = "";
|
|
931
|
+
for (let i = 0; i < k; i++) for (const v of NARRV) s += v + "\n\n";
|
|
932
|
+
return s;
|
|
933
|
+
};
|
|
934
|
+
const s0Det = detectLoops(asState([mk(cycles(5))]), tcfg);
|
|
935
|
+
const s0Hit = s0Det.find((d) => d.type === "block");
|
|
936
|
+
out.push(`block S0 narration loop → ${s0Hit ? `block (${s0Hit.description})` : "no"} (exp block — was the miss) ${s0Hit ? "✅" : "❌"}`);
|
|
937
|
+
out.push(`block turn weight → ${detectionTurnWeight(s0Det, tcfg)} (exp 2 — warn on first sight) ${detectionTurnWeight(s0Det, tcfg) === 2 ? "✅" : "❌"}`);
|
|
938
|
+
out.push(`block post-steer = ignored→ ${isVerbatimRepeat(s0Det) ? "yes" : "no"} (exp yes — replayed block after the break) ${isVerbatimRepeat(s0Det) ? "✅" : "❌"}`);
|
|
939
|
+
|
|
940
|
+
// 3 replay cycles still flag; a single restatement (2 cycles, top-gram ×2)
|
|
941
|
+
// does not — one echo is a summary, not stuck generation.
|
|
942
|
+
const c3 = findRepetitiveBlock(cycles(3), tcfg);
|
|
943
|
+
const c2 = findRepetitiveBlock(cycles(2), tcfg);
|
|
944
|
+
out.push(`block 3 cycles / 2 cycles→ ${c3 ? `flag (${Math.round(c3.ratio * 100)}%)` : "no"} / ${c2 ? "flag" : "no"} (exp flag / no — needs ≥3 repeats) ${c3 && !c2 ? "✅" : "❌"}`);
|
|
945
|
+
|
|
946
|
+
// Ordinary prose and templated (but evolving) payloads must stay silent:
|
|
947
|
+
// long docs score ≤ 0.11 coverage; templated logs/code carry varying digits
|
|
948
|
+
// so their 5-grams differ.
|
|
949
|
+
const prose =
|
|
950
|
+
"The extension watches every assistant message and tool call to decide whether the model is making progress. " +
|
|
951
|
+
"It stores a short window of recent turns, fingerprints tool results and compares them with earlier attempts. " +
|
|
952
|
+
"When a pattern repeats it escalates from a quiet warning to a forced change of approach and finally aborts. " +
|
|
953
|
+
"Configuration lives in a small json file next to the agent directory and every threshold can be tuned at runtime. " +
|
|
954
|
+
"The default values were chosen against real sessions so ordinary work never triggers a false positive. " +
|
|
955
|
+
"Detectors are independent: disabling one leaves the others active and the footer keeps the user informed. " +
|
|
956
|
+
"A clean turn cools the counter down so a recovered model is given room to finish the task. " +
|
|
957
|
+
"Everything is written in plain typescript with no runtime dependencies beyond the host package. " +
|
|
958
|
+
"New strategies should be measured against the recorded payloads before they are enabled by default. " +
|
|
959
|
+
"The goal is simple: catch a stuck model early and keep the context useful for the real work.";
|
|
960
|
+
const logLines = [...Array(30)].map((_, i) => `2026-09-24T10:${String(i).padStart(2, "0")}:00Z INFO worker ${i} processed job ${1000 + i} in ${i * 3}ms status ok`).join("\n");
|
|
961
|
+
const proseHit = findRepetitiveBlock(prose, tcfg);
|
|
962
|
+
const logHit = findRepetitiveBlock(logLines, tcfg);
|
|
963
|
+
out.push(`block legit prose/logs → ${proseHit || logHit ? "flag" : "ok"} (exp ok) ${!proseHit && !logHit ? "✅" : "❌"}`);
|
|
964
|
+
|
|
965
|
+
// --- v1.6.1: no-progress outcome runs (mutated re-runs, same outcome) ---
|
|
966
|
+
// Regression: the NFS session — ~90 mutated re-runs of the SAME experiment
|
|
967
|
+
// (ssh exportfs/mount, labels test A…test QQQ, targets alternating Javi /
|
|
968
|
+
// Compartido), every one failing rc=32. Args mutate each turn (same call
|
|
969
|
+
// never recurs → tool-loop silent) and journalctl noise varies per attempt,
|
|
970
|
+
// but the FAILURE SIGNATURE repeats: that IS the loop. Fires on the 9th
|
|
971
|
+
// attempt (8 prior same-failure matches ≥ outcomeMinRepeats).
|
|
972
|
+
const nfsCmd = (label: string, target: string) =>
|
|
973
|
+
argJson(
|
|
974
|
+
`sshpass -p X ssh -o ConnectTimeout=10 noguerol@petete 'cd /tmp && echo X | sudo -S bash -c "echo --- test ${label}: rootdir=/volume2, absolute paths, fsid=0 and 1, mount /${target} ---; ` +
|
|
975
|
+
`cat > /etc/exports << EOF\n/volume2/NAS-8TB-Javi *(rw,sync,no_subtree_check,fsid=0)\nEOF\nexportfs -ra\nsystemctl restart nfs-server\n` +
|
|
976
|
+
`mount -t nfs4 -o vers=4.2 127.0.0.1:/${target} /tmp/nfstest 2>&1; echo rc=\$?; journalctl -u nfs-mountd | tail -4"' 2>&1`,
|
|
977
|
+
);
|
|
978
|
+
const nfsFailFor = (target: string, sec: number) =>
|
|
979
|
+
resultFingerprint(
|
|
980
|
+
[
|
|
981
|
+
{
|
|
982
|
+
type: "text",
|
|
983
|
+
text:
|
|
984
|
+
`--- mount /${target} --- | mount.nfs4: access denied by server while mounting 127.0.0.1:/${target} rc=32 | ` +
|
|
985
|
+
`Sep 09 18:${sec} petete systemd[1]: Started nfs-mountd.service (PID ${1000 + sec})`,
|
|
986
|
+
},
|
|
987
|
+
],
|
|
988
|
+
false,
|
|
989
|
+
)!;
|
|
990
|
+
const nfsOk = resultFingerprint([{ type: "text", text: "rc=0 | TARGET SOURCE FSTYPE | /tmp/nfstest 127.0.0.1:/ nfs4 rw,relatime" }], false)!;
|
|
991
|
+
const targets = ["NAS-8TB-Javi", "NAS-8TB-Compartido"];
|
|
992
|
+
const nfsMsgs = [..."ABCDEFGHI"].map((l, idx) =>
|
|
993
|
+
mk("", [{ name: "bash", args: nfsCmd(l, targets[idx % 2]), result: nfsFailFor(targets[idx % 2], 100 + idx) }]),
|
|
994
|
+
);
|
|
995
|
+
const nfsDet = detectLoops(asState(nfsMsgs), tcfg);
|
|
996
|
+
const nfsHit = nfsDet.find((d) => d.type === "outcome");
|
|
997
|
+
out.push(`outcome fires on 9th → ${nfsHit ? `outcome (${nfsHit.description.slice(0, 100)}…)` : nfsDet.map((d) => d.type).join(",") || "no"} (exp outcome — mixed targets) ${nfsHit ? "✅" : "❌"}`);
|
|
998
|
+
out.push(`outcome weight → ${detectionTurnWeight(nfsDet, tcfg)} (exp 2) ${detectionTurnWeight(nfsDet, tcfg) === 2 ? "✅" : "❌"}`);
|
|
999
|
+
out.push(`outcome post-steer = ignored→ ${isVerbatimRepeat(nfsDet) ? "yes" : "no"} (exp yes — same failing outcome after the break) ${isVerbatimRepeat(nfsDet) ? "✅" : "❌"}`);
|
|
1000
|
+
|
|
1001
|
+
// Below the repeat count: 5 identical-failure attempts → still trying, silent.
|
|
1002
|
+
const fewMsgs = [..."ABCDE"].map((l, idx) =>
|
|
1003
|
+
mk("", [{ name: "bash", args: nfsCmd(l, targets[idx % 2]), result: nfsFailFor(targets[idx % 2], 100 + idx) }]),
|
|
1004
|
+
);
|
|
1005
|
+
const fewDet = detectLoops(asState(fewMsgs), tcfg);
|
|
1006
|
+
out.push(`outcome needs 8 prior → ${fewDet.some((d) => d.type === "outcome") ? "outcome" : "silent"} (exp silent at 5 attempts) ${!fewDet.some((d) => d.type === "outcome") ? "✅" : "❌"}`);
|
|
1007
|
+
|
|
1008
|
+
// Converging sweep (v1.1 guarantee): similar args but the outcome CHANGES
|
|
1009
|
+
// (progress!) — must stay silent even with many attempts.
|
|
1010
|
+
const progMsgs = [..."ABCDEFGHIJ"].map((l, idx) =>
|
|
1011
|
+
mk("", [
|
|
1012
|
+
{
|
|
1013
|
+
name: "bash",
|
|
1014
|
+
args: nfsCmd(l, targets[idx % 2]),
|
|
1015
|
+
result:
|
|
1016
|
+
idx === 9
|
|
1017
|
+
? nfsOk
|
|
1018
|
+
: resultFingerprint(
|
|
1019
|
+
[{ type: "text", text: `attempt ${idx}: failed with rc=${idx + 30} reason=${idx % 3}` }],
|
|
1020
|
+
true,
|
|
1021
|
+
)!,
|
|
1022
|
+
},
|
|
1023
|
+
]),
|
|
1024
|
+
);
|
|
1025
|
+
const progDet = detectLoops(asState(progMsgs), tcfg);
|
|
1026
|
+
out.push(`outcome converging sweep → ${progDet.some((d) => d.type === "outcome") ? "outcome" : "silent"} (exp silent — outcomes differ = progress) ${!progDet.some((d) => d.type === "outcome") ? "✅" : "❌"}`);
|
|
1027
|
+
|
|
1028
|
+
// Different FAILURE kinds with similar args (denied vs timeout vs no-such-file)
|
|
1029
|
+
// = evolving diagnosis, not the same wall — silent too.
|
|
1030
|
+
const diffFailMsgs = [..."ABCDEFGHIJ"].map((l, idx) => {
|
|
1031
|
+
const reasons = ["access denied by server", "timed out after 90 seconds", "No such file or directory"];
|
|
1032
|
+
const r = reasons[idx % 3];
|
|
1033
|
+
return mk("", [
|
|
1034
|
+
{ name: "bash", args: nfsCmd(l, targets[idx % 2]), result: resultFingerprint([{ type: "text", text: `mount failed: ${r} rc=32` }], false)! },
|
|
1035
|
+
]);
|
|
1036
|
+
});
|
|
1037
|
+
const diffFailDet = detectLoops(asState(diffFailMsgs), tcfg);
|
|
1038
|
+
out.push(`outcome diff failures → ${diffFailDet.some((d) => d.type === "outcome") ? "outcome" : "silent"} (exp silent — error changed = progress) ${!diffFailDet.some((d) => d.type === "outcome") ? "✅" : "❌"}`);
|
|
1039
|
+
|
|
1040
|
+
// Task-stream coexistence: a punched_log batch with identical tool results
|
|
1041
|
+
// must NOT count toward the outcome run (identical OK = normal batch).
|
|
1042
|
+
const batchFail = resultFingerprint([{ type: "text", text: "logged" }], false)!;
|
|
1043
|
+
const batchOutMsgs = [1, 2, 3, 4, 5, 6, 7, 8, 9].map((n) => mk(NARR, [{ name: "punched_log", args: noteArgs(String(n)), result: batchFail }]));
|
|
1044
|
+
const batchOutDet = detectLoops(asState(batchOutMsgs), tcfg);
|
|
1045
|
+
out.push(`outcome batch excluded → ${batchOutDet.some((d) => d.type === "outcome") ? "outcome" : "silent"} (exp silent — identical OKs are batch norm) ${!batchOutDet.some((d) => d.type === "outcome") ? "✅" : "❌"}`);
|
|
1046
|
+
|
|
1047
|
+
// Failure gate: 9 near-identical attempts that all SUCCEED identically (e.g.
|
|
1048
|
+
// re-verifying a working setup, or a file-write batch) must stay silent —
|
|
1049
|
+
// only repeated FAILURES prove no progress.
|
|
1050
|
+
const okMsgs = [..."ABCDEFGHI"].map((l, idx) => mk("", [{ name: "bash", args: nfsCmd(l, targets[idx % 2]), result: nfsOk }]));
|
|
1051
|
+
const okDet = detectLoops(asState(okMsgs), tcfg);
|
|
1052
|
+
out.push(`outcome identical OKs → ${okDet.some((d) => d.type === "outcome") ? "outcome" : "silent"} (exp silent — success repeats ≠ loop) ${!okDet.some((d) => d.type === "outcome") ? "✅" : "❌"}`);
|
|
1053
|
+
out.push(`isFailResult gate → err| → ${isFailResult("err|boom") ? "fail" : "ok"}, rc32 → ${isFailResult("ok|rc32 denied") ? "fail" : "ok"}, rc0/ok → ${isFailResult("ok|rc 0 12 passed") ? "fail" : "ok"} (exp fail, fail, ok) ${isFailResult("err|boom") && isFailResult("ok|rc32 denied") && !isFailResult("ok|rc 0 12 passed") ? "✅" : "❌"}`);
|
|
1054
|
+
|
|
446
1055
|
return out;
|
|
447
1056
|
}
|