pi-anti-doom-loop 0.0.8 → 0.0.10
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +26 -0
- package/extensions/controller.ts +1 -0
- package/extensions/detector.ts +175 -26
- package/extensions/index.ts +35 -10
- package/package.json +6 -6
- package/scripts/guard-publish.ts +22 -4
package/CHANGELOG.md
CHANGED
|
@@ -2,6 +2,32 @@
|
|
|
2
2
|
|
|
3
3
|
All notable changes to **pi-anti-doom-loop**.
|
|
4
4
|
|
|
5
|
+
## [0.0.10] — 2026-09-23
|
|
6
|
+
|
|
7
|
+
### Fixed
|
|
8
|
+
|
|
9
|
+
- **Counters reset per user prompt, not per run** — the extension reset on `before_agent_start`, which omp emits on _every_ run start: auto-continue turns (`role:"developer"`), steers/resumes, advisor cards, and queued async-result drains all restart the run, wiping counters (including block-escalation state) between the iterations of exactly the cross-run loops this extension exists to catch. Reported identical `ssh`-polling loops ran 7+ times unblocked while intra-run repeats fired correctly — replaying the incident session on detector-visible input reproduces its 9 exact blocks, with the only discrepancies landing right after post-abort run restarts, where the per-run reset had cleared the window. Reset now keys off `message_end` with `role:"user"` (emitted with the run's input messages on omp and pi alike), matching the documented per-user-prompt contract; steers and auto-continues keep counting, so the steer→abort escalation ladder survives run restarts too.
|
|
10
|
+
|
|
11
|
+
### Added
|
|
12
|
+
|
|
13
|
+
- **Near-identical tool-call detection** — when no exact signature repeats, `check()` counts same-tool window entries whose canonical args share ≥ 55% whitespace tokens (token union ≥ 5): sleep-wrapped polls (`gh run list…` → `sleep 150; gh run list…`), flag/grep-pattern/tail drift. Block reason prefix: _"…" was called with near-identical arguments N times…_. Short inputs stay exact-only via the union guard; exact repeats still block first with the exact reason.
|
|
14
|
+
|
|
15
|
+
### Notes
|
|
16
|
+
|
|
17
|
+
- omp strips the harness intent label (`i`) at parse (`extractIntent`), before extension events — this detector never sees it; no handling was added. pi has no intent field at all.
|
|
18
|
+
|
|
19
|
+
## [0.0.9] — 2026-09-07
|
|
20
|
+
|
|
21
|
+
### Added
|
|
22
|
+
|
|
23
|
+
- **Full anti-slop rule set** — synced the vendored Oxlint plugin with upstream: 5 missing generic rules (`no-module-mocking`, `no-reflect-apply`, `no-reflect-get`, `no-unknown-returns`, `require-safety-comment-for-type-assertion`), the opt-in Effect plugin (`no-service-constructor-imports`, enabled — this repo declares `effect`), and the supporting shared helpers. All 15 generic rules plus the Effect rule run at `error`.
|
|
24
|
+
- **Complexity gate** — `eslint/complexity` with `{ max: 10 }`; anything in the 11+ "refactor now" band fails the build.
|
|
25
|
+
|
|
26
|
+
### Fixed
|
|
27
|
+
|
|
28
|
+
- **Within-message false positive on status lists** — `repeatedSegment` no longer splits on `:` (status updates like "worker dispatched: a. worker dispatched: b. worker dispatched: c." stay whole and distinct). Separator-less concatenation (`S:S:S:`) and truncated repeats are still caught by the new `tandemPrefix` check.
|
|
29
|
+
- **Complexity + slop findings in owned code** — extracted `collectBlockReasons` (`check` CC 11 → 6); replaced `as` casts at I/O boundaries with `SAFETY:` invariants, real `isManifest` / `hasStringVersion` guards, a named `PiHandlerResult` contract, and assertion-free tests.
|
|
30
|
+
|
|
5
31
|
## [0.0.8] — 2026-08-24
|
|
6
32
|
|
|
7
33
|
### Added
|
package/extensions/controller.ts
CHANGED
|
@@ -122,6 +122,7 @@ export function createController(opts: LoopOptions = readOptions()): AntiLoopCon
|
|
|
122
122
|
// Within-message duplicate tool-call spam fires first: it aborts (the
|
|
123
123
|
// calls are already emitted, steering cannot retract them), so it must
|
|
124
124
|
// outrank the steer-able text ladder.
|
|
125
|
+
// SAFETY: tool-call arguments arrive as JSON-safe values matching ToolInput.
|
|
125
126
|
const calls = content
|
|
126
127
|
.filter((c) => c.type === "toolCall")
|
|
127
128
|
.map((c) => ({ toolName: c.name ?? "", input: c.arguments as ToolInput }));
|
package/extensions/detector.ts
CHANGED
|
@@ -102,6 +102,12 @@ export function canonical(input: ToolInput): string {
|
|
|
102
102
|
return JSON.stringify(input) ?? "";
|
|
103
103
|
}
|
|
104
104
|
|
|
105
|
+
/**
|
|
106
|
+
* Callers pass detector-visible input: hosts strip harness-internal fields
|
|
107
|
+
* (omp removes the intent label `i` at parse via `extractIntent`) before the
|
|
108
|
+
* `tool_call` event, so identity is the executed args exactly as the detector
|
|
109
|
+
* receives them.
|
|
110
|
+
*/
|
|
105
111
|
export function signature(toolName: string, input: ToolInput): string {
|
|
106
112
|
return `${toolName}:${canonical(input)}`;
|
|
107
113
|
}
|
|
@@ -113,6 +119,35 @@ export interface BlockDecision {
|
|
|
113
119
|
escalate: boolean;
|
|
114
120
|
}
|
|
115
121
|
|
|
122
|
+
/** Block reasons for one tool call: identical repeats, failure streak, failure rate. */
|
|
123
|
+
function collectBlockReasons(
|
|
124
|
+
toolName: string,
|
|
125
|
+
total: number,
|
|
126
|
+
consecutiveFails: number,
|
|
127
|
+
rate: { calls: number; errors: number; rate: number },
|
|
128
|
+
opts: LoopOptions,
|
|
129
|
+
): string[] {
|
|
130
|
+
const reasons: string[] = [];
|
|
131
|
+
if (total >= opts.repeatThreshold) {
|
|
132
|
+
reasons.push(
|
|
133
|
+
`"${toolName}" was called with identical arguments ${total} times in the last ${opts.windowSize} tool calls with no change`,
|
|
134
|
+
);
|
|
135
|
+
}
|
|
136
|
+
if (consecutiveFails >= opts.failThreshold) {
|
|
137
|
+
reasons.push(`"${toolName}" failed ${consecutiveFails} consecutive times`);
|
|
138
|
+
}
|
|
139
|
+
if (
|
|
140
|
+
opts.failRateThreshold > 0 &&
|
|
141
|
+
rate.calls >= opts.failRateMinCalls &&
|
|
142
|
+
rate.rate >= opts.failRateThreshold
|
|
143
|
+
) {
|
|
144
|
+
reasons.push(
|
|
145
|
+
`"${toolName}" failed ${rate.errors} of ${rate.calls} calls in the window (${Math.round(rate.rate * 100)}%)`,
|
|
146
|
+
);
|
|
147
|
+
}
|
|
148
|
+
return reasons;
|
|
149
|
+
}
|
|
150
|
+
|
|
116
151
|
/**
|
|
117
152
|
* Call `check` in `tool_call` (before executing). If it returns a decision,
|
|
118
153
|
* block the call. Call `record` only for calls that were NOT blocked, and
|
|
@@ -120,7 +155,7 @@ export interface BlockDecision {
|
|
|
120
155
|
*/
|
|
121
156
|
export class LoopDetector {
|
|
122
157
|
readonly opts: LoopOptions;
|
|
123
|
-
private recentSigs: { sig: string; ts: number }[] = [];
|
|
158
|
+
private recentSigs: { sig: string; tool: string; args: string; ts: number }[] = [];
|
|
124
159
|
private recentResults: { tool: string; error: boolean; ts: number }[] = [];
|
|
125
160
|
private blockedBySig = new Map<string, number>();
|
|
126
161
|
private recentTexts: { text: string; ts: number }[] = [];
|
|
@@ -138,33 +173,36 @@ export class LoopDetector {
|
|
|
138
173
|
return Result.err(undefined);
|
|
139
174
|
}
|
|
140
175
|
this.evictSigs();
|
|
141
|
-
const
|
|
176
|
+
const args = canonical(input);
|
|
177
|
+
const sig = `${toolName}:${args}`;
|
|
142
178
|
const repeats = this.recentSigs.filter((s) => s.sig === sig).length;
|
|
143
179
|
const total = repeats + 1; // including this call
|
|
144
|
-
const consecutiveFails = this.consecutiveFails(toolName);
|
|
145
|
-
const rate = this.failRate(toolName);
|
|
146
180
|
|
|
147
181
|
// Rough cost accounting (feature B): every redundant repeat of an already
|
|
148
182
|
// present signature burns tokens with no new information.
|
|
149
183
|
if (repeats >= 1) this.wastedTokens += estimateTokens(stringify(input));
|
|
150
184
|
|
|
151
|
-
const reasons
|
|
152
|
-
|
|
153
|
-
|
|
154
|
-
|
|
155
|
-
)
|
|
156
|
-
|
|
157
|
-
|
|
158
|
-
|
|
159
|
-
|
|
160
|
-
|
|
161
|
-
|
|
162
|
-
|
|
163
|
-
|
|
164
|
-
) {
|
|
165
|
-
|
|
166
|
-
|
|
167
|
-
)
|
|
185
|
+
const reasons = collectBlockReasons(
|
|
186
|
+
toolName,
|
|
187
|
+
total,
|
|
188
|
+
this.consecutiveFails(toolName),
|
|
189
|
+
this.failRate(toolName),
|
|
190
|
+
this.opts,
|
|
191
|
+
);
|
|
192
|
+
|
|
193
|
+
// Exact-signature miss: a loop that varies cosmetic fields (the `i` intent
|
|
194
|
+
// label, tail flags, timeouts) never repeats a signature. Production data:
|
|
195
|
+
// 7 of 9 commands repeated ≥3× in a live session had fully distinct args
|
|
196
|
+
// with an identical `command`. Same tool + ≥55% shared arg tokens still
|
|
197
|
+
// counts as a repeat.
|
|
198
|
+
if (reasons.length === 0) {
|
|
199
|
+
const near = this.nearRepeats(toolName, args);
|
|
200
|
+
if (near >= 1) this.wastedTokens += estimateTokens(stringify(input));
|
|
201
|
+
if (near + 1 >= this.opts.repeatThreshold) {
|
|
202
|
+
reasons.push(
|
|
203
|
+
`"${toolName}" was called with near-identical arguments ${near + 1} times in the last ${this.opts.windowSize} tool calls (only minor changes)`,
|
|
204
|
+
);
|
|
205
|
+
}
|
|
168
206
|
}
|
|
169
207
|
|
|
170
208
|
if (reasons.length === 0) return Result.err(undefined);
|
|
@@ -184,10 +222,22 @@ export class LoopDetector {
|
|
|
184
222
|
|
|
185
223
|
record(toolName: string, input: ToolInput): void {
|
|
186
224
|
if (this.opts.toolExclude.has(toolName)) return;
|
|
187
|
-
|
|
225
|
+
const args = canonical(input);
|
|
226
|
+
this.recentSigs.push({ sig: `${toolName}:${args}`, tool: toolName, args, ts: Date.now() });
|
|
188
227
|
this.evictSigs();
|
|
189
228
|
}
|
|
190
229
|
|
|
230
|
+
/** Windowed same-tool entries whose args are ≥55% shared tokens (identical counts). */
|
|
231
|
+
private nearRepeats(toolName: string, args: string): number {
|
|
232
|
+
let n = 0;
|
|
233
|
+
for (const s of this.recentSigs) {
|
|
234
|
+
if (s.tool !== toolName) continue;
|
|
235
|
+
const sim = argSimilarity(args, s.args);
|
|
236
|
+
if (sim !== null && sim >= NEAR_ARGS_SIMILARITY_THRESHOLD) n++;
|
|
237
|
+
}
|
|
238
|
+
return n;
|
|
239
|
+
}
|
|
240
|
+
|
|
191
241
|
recordResult(toolName: string, error: boolean): void {
|
|
192
242
|
this.recentResults.push({ tool: toolName, error, ts: Date.now() });
|
|
193
243
|
this.evictResults();
|
|
@@ -414,6 +464,38 @@ export const MIN_REPEAT_CHUNK = 16;
|
|
|
414
464
|
/** Jaccard similarity threshold for "near-identical" consecutive texts. */
|
|
415
465
|
export const TEXT_SIMILARITY_THRESHOLD = 0.55;
|
|
416
466
|
|
|
467
|
+
/** Shared-whitespace-token threshold for near-identical tool-call arguments. */
|
|
468
|
+
export const NEAR_ARGS_SIMILARITY_THRESHOLD = 0.55;
|
|
469
|
+
|
|
470
|
+
/**
|
|
471
|
+
* Minimum token union before two arg strings are comparable. Short inputs
|
|
472
|
+
* ({"i":"…","op":"view"}) would score 1.0 on one shared token; they rely on
|
|
473
|
+
* exact matching only. Long inputs — shell commands — always clear this.
|
|
474
|
+
*/
|
|
475
|
+
export const NEAR_ARGS_MIN_UNION = 5;
|
|
476
|
+
// ponytail: 0.55 misses ONE shape — short (~11-token) commands that change
|
|
477
|
+
// BOTH ends at once (added wrapper + changed tail glue → ~0.53); single-axis
|
|
478
|
+
// drift and longer commands clear. Upgrade path if that shape shows up:
|
|
479
|
+
// tokenize on punctuation boundaries instead of whitespace.
|
|
480
|
+
|
|
481
|
+
/**
|
|
482
|
+
* Case-insensitive whitespace-token Jaccard over canonical args.
|
|
483
|
+
* Full tokens (no word filtering): a changed field glues into its own token,
|
|
484
|
+
* so only genuinely shared structure counts — same command with a rephrased
|
|
485
|
+
* `i` label scores ~0.8, two different commands score far below the threshold.
|
|
486
|
+
* Returns null when the union is too small to judge.
|
|
487
|
+
*/
|
|
488
|
+
export function argSimilarity(a: string, b: string): number | null {
|
|
489
|
+
const as = new Set(a.toLowerCase().split(/\s+/).filter(Boolean));
|
|
490
|
+
const bs = new Set(b.toLowerCase().split(/\s+/).filter(Boolean));
|
|
491
|
+
if (as.size === 0 || bs.size === 0) return null;
|
|
492
|
+
let inter = 0;
|
|
493
|
+
for (const t of as) if (bs.has(t)) inter++;
|
|
494
|
+
const union = as.size + bs.size - inter;
|
|
495
|
+
if (union < NEAR_ARGS_MIN_UNION) return null;
|
|
496
|
+
return inter / union;
|
|
497
|
+
}
|
|
498
|
+
|
|
417
499
|
/** Stable string form of a tool input (used for token estimation). */
|
|
418
500
|
export function stringify(input: ToolInput): string {
|
|
419
501
|
return JSON.stringify(input) ?? "";
|
|
@@ -456,15 +538,20 @@ export function tokenSimilarity(a: string, b: string): number {
|
|
|
456
538
|
* within a single normalized message, or null.
|
|
457
539
|
*
|
|
458
540
|
* Catches growing doom loops where the model self-concatenates the same
|
|
459
|
-
* sentence ("…X
|
|
541
|
+
* sentence ("…X…X…X") — the pattern that evaded cross-message verbatim
|
|
460
542
|
* detection in production (each message differs, so no streak forms).
|
|
461
543
|
* Short segments (< MIN_REPEAT_CHUNK) are ignored so pasted logs with
|
|
462
544
|
* repeated one-word lines never false-positive.
|
|
545
|
+
*
|
|
546
|
+
* ':' is deliberately NOT a sentence boundary: status lists like
|
|
547
|
+
* "worker dispatched: a. worker dispatched: b. worker dispatched: c."
|
|
548
|
+
* must stay whole so distinct sentences never count as repeats.
|
|
549
|
+
* Separator-less concatenation ("S:S:S:") is caught by tandemPrefix instead.
|
|
463
550
|
*/
|
|
464
551
|
export function repeatedSegment(normalized: string, threshold: number): string | null {
|
|
465
552
|
const segments = normalized
|
|
466
|
-
.split(/(?<=[
|
|
467
|
-
.map((s) => s.trim().replace(/[
|
|
553
|
+
.split(/(?<=[.!?])\s+/)
|
|
554
|
+
.map((s) => s.trim().replace(/[.!?]+$/, ""))
|
|
468
555
|
.filter((s) => s.length >= MIN_REPEAT_CHUNK);
|
|
469
556
|
const counts = new Map<string, number>();
|
|
470
557
|
for (const seg of segments) {
|
|
@@ -472,9 +559,47 @@ export function repeatedSegment(normalized: string, threshold: number): string |
|
|
|
472
559
|
if (n >= threshold) return seg;
|
|
473
560
|
counts.set(seg, n);
|
|
474
561
|
}
|
|
562
|
+
return tandemPrefix(normalized, threshold);
|
|
563
|
+
}
|
|
564
|
+
|
|
565
|
+
/** Messages longer than this are pasted logs, not loop utterances — skip. */
|
|
566
|
+
const TANDEM_MAX_LEN = 2000;
|
|
567
|
+
/** Longest repeated block worth scanning (real loop sentences are < 200 chars). */
|
|
568
|
+
const TANDEM_MAX_CHUNK = 500;
|
|
569
|
+
|
|
570
|
+
/**
|
|
571
|
+
* Whole-message consecutive repetition anchored at the start ("S:S:S:",
|
|
572
|
+
* "S: S: S:", regex fragments "X X X"). Returns the stripped block or null.
|
|
573
|
+
* Skips long inputs (pasted logs) and pure-separator blocks.
|
|
574
|
+
*/
|
|
575
|
+
export function tandemPrefix(normalized: string, threshold: number): string | null {
|
|
576
|
+
if (normalized.length < MIN_REPEAT_CHUNK * threshold) return null;
|
|
577
|
+
if (normalized.length > TANDEM_MAX_LEN) return null;
|
|
578
|
+
const padded = normalized.endsWith(" ") ? normalized : `${normalized} `;
|
|
579
|
+
const maxL = Math.min(TANDEM_MAX_CHUNK, Math.floor((padded.length - 1) / (threshold - 1)));
|
|
580
|
+
for (let len = MIN_REPEAT_CHUNK; len <= maxL; len++) {
|
|
581
|
+
const first = stripBlock(padded.slice(0, len));
|
|
582
|
+
if (first.length < MIN_REPEAT_CHUNK) continue;
|
|
583
|
+
let ok = true;
|
|
584
|
+
for (let k = 1; k < threshold; k++) {
|
|
585
|
+
if (stripBlock(padded.slice(k * len, (k + 1) * len)) !== first) {
|
|
586
|
+
ok = false;
|
|
587
|
+
break;
|
|
588
|
+
}
|
|
589
|
+
}
|
|
590
|
+
if (ok) return first;
|
|
591
|
+
}
|
|
475
592
|
return null;
|
|
476
593
|
}
|
|
477
594
|
|
|
595
|
+
/** Compare blocks ignoring trailing separators so "X:" and "X" unify. */
|
|
596
|
+
function stripBlock(block: string): string {
|
|
597
|
+
return block
|
|
598
|
+
.trim()
|
|
599
|
+
.replace(/[.:!?\s]+$/, "")
|
|
600
|
+
.trim();
|
|
601
|
+
}
|
|
602
|
+
|
|
478
603
|
// --- self-check (runs under `node extensions/detector.ts`, skipped when loaded by pi) ---
|
|
479
604
|
if (import.meta.main) {
|
|
480
605
|
const opts: LoopOptions = {
|
|
@@ -612,7 +737,7 @@ if (import.meta.main) {
|
|
|
612
737
|
const spamHit = d.checkDuplicateCalls(
|
|
613
738
|
Array.from({ length: 3 }, () => ({
|
|
614
739
|
toolName: "bash",
|
|
615
|
-
input: { command: "true" }
|
|
740
|
+
input: { command: "true" } satisfies ToolInput,
|
|
616
741
|
})),
|
|
617
742
|
);
|
|
618
743
|
assert.ok(spamHit.isOk(), "3 identical calls in one message should fire");
|
|
@@ -627,5 +752,29 @@ if (import.meta.main) {
|
|
|
627
752
|
"distinct parallel args are not spam",
|
|
628
753
|
);
|
|
629
754
|
|
|
755
|
+
// 14. command-tail/wrapper drift: same poll wrapped or slightly reworded —
|
|
756
|
+
// near-identical token overlap catches what exact identity cannot.
|
|
757
|
+
d.reset();
|
|
758
|
+
const ghTail = (command: string) => ({ command, cwd: "/w/flash", timeout: 30 });
|
|
759
|
+
assert.ok(d.check("bash", ghTail("gh run list --branch main --limit 3 2>&1 | head -6")).isErr());
|
|
760
|
+
d.record("bash", ghTail("gh run list --branch main --limit 3 2>&1 | head -6"));
|
|
761
|
+
assert.ok(
|
|
762
|
+
d
|
|
763
|
+
.check("bash", ghTail("sleep 150; gh run list --branch main --limit 3 2>&1 | head -6"))
|
|
764
|
+
.isErr(),
|
|
765
|
+
);
|
|
766
|
+
d.record("bash", ghTail("sleep 150; gh run list --branch main --limit 3 2>&1 | head -6"));
|
|
767
|
+
const nearHit = d.check(
|
|
768
|
+
"bash",
|
|
769
|
+
ghTail("timeout 30; gh run list --branch main --limit 3 2>&1 | head -6"),
|
|
770
|
+
);
|
|
771
|
+
assert.ok(nearHit.isOk(), "3rd command-tail variant should block");
|
|
772
|
+
if (nearHit.isOk()) {
|
|
773
|
+
assert.match(nearHit.value.reason, /near-identical arguments 3 times/);
|
|
774
|
+
assert.equal(nearHit.value.escalate, false, "first near block does not escalate");
|
|
775
|
+
}
|
|
776
|
+
// genuinely different command with the same tool stays unblocked
|
|
777
|
+
assert.ok(d.check("bash", { command: "cargo test -p pools 2>&1 | tail -5" }).isErr());
|
|
778
|
+
|
|
630
779
|
console.log("detector self-check: all assertions passed");
|
|
631
780
|
}
|
package/extensions/index.ts
CHANGED
|
@@ -7,6 +7,9 @@
|
|
|
7
7
|
*
|
|
8
8
|
* - identical (tool, args) repeated `PI_ANTI_LOOP_REPEATS` times (default 3)
|
|
9
9
|
* in the last `PI_ANTI_LOOP_WINDOW` calls → block with an instructive reason
|
|
10
|
+
* - near-identical same-tool args (≥ 55% shared whitespace tokens, union ≥ 5)
|
|
11
|
+
* count toward the same threshold — sleep-wrapped polls, flag/pattern/tail
|
|
12
|
+
* drift that exact identity misses by construction
|
|
10
13
|
* - the same tool failing `PI_ANTI_LOOP_FAILS` consecutive times (default 3)
|
|
11
14
|
* → block with a "stop retrying, fix the root cause" reason
|
|
12
15
|
* - the model repeating text: verbatim, near-identical (token similarity),
|
|
@@ -19,8 +22,12 @@
|
|
|
19
22
|
* instructive reason (that is the steer); re-issuing the exact same blocked
|
|
20
23
|
* call aborts the turn.
|
|
21
24
|
*
|
|
22
|
-
* Counters reset on every user prompt
|
|
23
|
-
* in the session is never a false positive.
|
|
25
|
+
* Counters reset on every user prompt (`message_end` with `role:"user"`), so
|
|
26
|
+
* a task legitimately repeated later in the session is never a false positive.
|
|
27
|
+
* Run restarts (auto-continue, steers/resumes, async-result drains) do NOT
|
|
28
|
+
* reset — hosts emit `before_agent_start` on every run start, and resetting
|
|
29
|
+
* there wiped counters between the iterations of cross-run loops. Disable with
|
|
30
|
+
* PI_ANTI_LOOP_DISABLE=1.
|
|
24
31
|
*
|
|
25
32
|
* All logic lives in `controller.ts` (pure, pi-free, unit-tested); this file
|
|
26
33
|
* is a thin adapter wiring it to pi's event loop. The pi API is consumed
|
|
@@ -38,9 +45,18 @@ import {
|
|
|
38
45
|
} from "./controller.ts";
|
|
39
46
|
import { readOptions, type ToolInput } from "./detector.ts";
|
|
40
47
|
|
|
48
|
+
/** What an event handler hands back to pi: a tool-call block decision, or nothing.
|
|
49
|
+
* Mirrors pi's own `ExtensionHandler<E, R>` contract (`Promise<R | void> | R | void`)
|
|
50
|
+
* narrowed to this extension's result; handlers stay synchronous here.
|
|
51
|
+
*/
|
|
52
|
+
export type PiHandlerResult = { block: true; reason: string } | undefined;
|
|
53
|
+
|
|
41
54
|
/** The subset of pi's ExtensionAPI this extension uses (structural). */
|
|
42
55
|
export interface PiLike {
|
|
43
|
-
on<E = unknown, C = unknown>(
|
|
56
|
+
on<E = unknown, C = unknown>(
|
|
57
|
+
event: string,
|
|
58
|
+
handler: (event: E, ctx: C) => PiHandlerResult | void,
|
|
59
|
+
): void;
|
|
44
60
|
registerCommand(
|
|
45
61
|
name: string,
|
|
46
62
|
opts: {
|
|
@@ -73,13 +89,9 @@ export default function (pi: PiLike): void {
|
|
|
73
89
|
|
|
74
90
|
pi.on("session_start", () => reset());
|
|
75
91
|
|
|
76
|
-
// Fresh counters per user prompt: only the loop happening *right now* counts.
|
|
77
|
-
// Internal reset keeps session-scoped steers/aborts/resume budget.
|
|
78
|
-
pi.on("before_agent_start", () => controller.reset());
|
|
79
|
-
|
|
80
92
|
pi.on("tool_call", (event: ToolCallEventLite, ctx: CtxLite) => {
|
|
81
|
-
//
|
|
82
|
-
// ToolInput domain type at this I/O boundary before the controller sees them.
|
|
93
|
+
// SAFETY: pi delivers JSON-safe tool arguments, which is exactly the ToolInput domain.
|
|
94
|
+
// Decode them into the ToolInput domain type at this I/O boundary before the controller sees them.
|
|
83
95
|
const outcome = controller.onToolCall(
|
|
84
96
|
event.toolName,
|
|
85
97
|
event.input as ToolInput,
|
|
@@ -99,8 +111,21 @@ export default function (pi: PiLike): void {
|
|
|
99
111
|
|
|
100
112
|
// Text-only doom loops (model re-emits/rephrases the same thing with no
|
|
101
113
|
// tool calls) never reach tool_call. Steer first, abort as escalation,
|
|
102
|
-
// then a bounded auto-resume so
|
|
114
|
+
// then a bounded auto-resume so work continues.
|
|
103
115
|
pi.on("message_end", (event: MessageEndEventLite, ctx: CtxLite) => {
|
|
116
|
+
// Fresh counters per genuine user prompt: only the loop happening *right
|
|
117
|
+
// now* counts (session-scoped steers/aborts/resume budget survive).
|
|
118
|
+
// Deliberately NOT on `before_agent_start`: hosts emit that on every run
|
|
119
|
+
// start — auto-continue turns, steers/resumes, advisor cards, and queued
|
|
120
|
+
// async-result drains — which wiped counters between the iterations of
|
|
121
|
+
// cross-run doom loops, so they never reached the repeat threshold.
|
|
122
|
+
// `role:"user"` input messages are emitted with the run's input messages
|
|
123
|
+
// (agent-core `emitInputMessages`) on omp and pi alike.
|
|
124
|
+
if (event.message.role === "user") {
|
|
125
|
+
controller.reset();
|
|
126
|
+
return;
|
|
127
|
+
}
|
|
128
|
+
// SAFETY: pi delivers message content as JSON-safe blocks matching MessageContent.
|
|
104
129
|
// Decode the untyped message content into MessageContent at this boundary.
|
|
105
130
|
const outcome = controller.onMessageEnd(
|
|
106
131
|
event.message.role,
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "pi-anti-doom-loop",
|
|
3
|
-
"version": "0.0.
|
|
3
|
+
"version": "0.0.10",
|
|
4
4
|
"description": "Detect and break agent doom loops in pi: blocks identical repeated tool calls and blind retries before they burn tokens.",
|
|
5
5
|
"keywords": [
|
|
6
6
|
"anti-doom-loop",
|
|
@@ -37,19 +37,19 @@
|
|
|
37
37
|
"lint": "oxlint --deny-warnings",
|
|
38
38
|
"format": "oxfmt .",
|
|
39
39
|
"format:check": "oxfmt --check .",
|
|
40
|
-
"prepare": "effect-tsgo patch --no-typescript --oxlint"
|
|
40
|
+
"prepare": "effect-tsgo patch --no-typescript --oxlint --skip-missing"
|
|
41
41
|
},
|
|
42
42
|
"dependencies": {
|
|
43
|
+
"@effect/tsgo": "^0.36.4",
|
|
43
44
|
"better-result": "^3.0.0",
|
|
44
|
-
"effect": "^4.0.0-rc.108"
|
|
45
|
+
"effect": "^4.0.0-rc.108",
|
|
46
|
+
"oxlint": "1.77.0"
|
|
45
47
|
},
|
|
46
48
|
"devDependencies": {
|
|
47
49
|
"@earendil-works/pi-coding-agent": "*",
|
|
48
|
-
"@
|
|
49
|
-
"@oxlint/plugins": "^1.77.0",
|
|
50
|
+
"@oxlint/plugins": "1.77.0",
|
|
50
51
|
"@types/node": "^22.0.0",
|
|
51
52
|
"oxfmt": "^0.62.0",
|
|
52
|
-
"oxlint": "^1.77.0",
|
|
53
53
|
"oxlint-tsgolint": "^7.0.2001",
|
|
54
54
|
"typescript": "^5.6.0"
|
|
55
55
|
},
|
package/scripts/guard-publish.ts
CHANGED
|
@@ -35,14 +35,31 @@ const readManifest: Effect.Effect<string, GuardError> = Effect.tryPromise({
|
|
|
35
35
|
catch: () => ({ message: "GUARD FAIL: could not read package.json" }),
|
|
36
36
|
});
|
|
37
37
|
|
|
38
|
+
/** True when an unknown JSON value has the manifest shape this guard needs. */
|
|
39
|
+
const isManifest = (value: unknown): value is Manifest => {
|
|
40
|
+
if (typeof value !== "object" || value === null) return false;
|
|
41
|
+
if (!("name" in value) || !("version" in value)) return false;
|
|
42
|
+
return typeof value.name === "string" && typeof value.version === "string";
|
|
43
|
+
};
|
|
44
|
+
|
|
45
|
+
/** True when an unknown JSON value carries a string version field. */
|
|
46
|
+
const hasStringVersion = (value: unknown): value is { version: string } => {
|
|
47
|
+
if (typeof value !== "object" || value === null) return false;
|
|
48
|
+
if (!("version" in value)) return false;
|
|
49
|
+
return typeof value.version === "string";
|
|
50
|
+
};
|
|
51
|
+
|
|
38
52
|
const parseManifest = (raw: string): Effect.Effect<Manifest, GuardError> => {
|
|
39
|
-
let
|
|
53
|
+
let parsed: unknown;
|
|
40
54
|
try {
|
|
41
|
-
|
|
55
|
+
parsed = JSON.parse(raw);
|
|
42
56
|
} catch {
|
|
43
57
|
return Effect.fail({ message: "GUARD FAIL: package.json is not valid JSON" });
|
|
44
58
|
}
|
|
45
|
-
|
|
59
|
+
if (!isManifest(parsed)) {
|
|
60
|
+
return Effect.fail({ message: "GUARD FAIL: package.json has no string name and version" });
|
|
61
|
+
}
|
|
62
|
+
return Effect.succeed(parsed);
|
|
46
63
|
};
|
|
47
64
|
const semverLike = (value: string): boolean => /^\d+\.\d+\.\d+$/.test(value);
|
|
48
65
|
|
|
@@ -77,7 +94,8 @@ const fetchPublished = (name: string): Effect.Effect<string | null> =>
|
|
|
77
94
|
headers: { accept: "application/json" },
|
|
78
95
|
});
|
|
79
96
|
if (!res.ok) return null;
|
|
80
|
-
|
|
97
|
+
const body: unknown = await res.json();
|
|
98
|
+
return hasStringVersion(body) ? body.version : null;
|
|
81
99
|
},
|
|
82
100
|
catch: () => {
|
|
83
101
|
console.warn(
|