billion-context-pi 0.1.63 → 0.1.64-pr.357.148

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/config.d.ts CHANGED
@@ -1,5 +1,6 @@
1
1
  import { type Config, type Prompts } from "acp-kernel";
2
2
  import type { CompressReasoningConfig } from "./reasoning-drop.js";
3
+ import type { DegenerationGuardConfig } from "./degeneration.js";
3
4
  import type { ThrottleRetryConfig } from "./throttle-retry.js";
4
5
  /** Per-role delegate defaults. Lets long-lived automation pin a cheaper or
5
6
  * more capable model and a thinking level per delegate role, so the main
@@ -222,6 +223,15 @@ export interface AdapterConfig {
222
223
  * warn=3, abort=5. Stops greedy small models looping on byte-identical
223
224
  * tool calls (issue #308). */
224
225
  repetitionGuard?: boolean | RepetitionGuardConfig;
226
+ /** Character-level degenerate-repeat guard (see DegenerationGuardConfig).
227
+ * Collapses long single-codepoint runs (e.g. 4655×「【」) in assistant
228
+ * text/thinking of the outgoing view and injects a one-shot recovery notice
229
+ * after a degenerated turn — breaking the abort loop where pi replays the
230
+ * degenerated thinking back to the provider on every request (issue #351).
231
+ * Distinct from `repetitionGuard`, which is tool-call level. Accepts a
232
+ * boolean shorthand (`false` disables) or an object. Default: enabled,
233
+ * minRun=200. */
234
+ degenerationGuard?: boolean | DegenerationGuardConfig;
225
235
  /** Legacy flat alias for `delegate.displayUsage`. Kept for backward
226
236
  * compatibility with existing acp.json files. Prefer `delegate.displayUsage`. */
227
237
  displayUsage?: "merged" | "separate";
@@ -0,0 +1,98 @@
1
+ import type { SessionMessageEntry } from "@earendil-works/pi-coding-agent";
2
+ type AgentMessage = SessionMessageEntry["message"];
3
+ /** [#351] Character-level degenerate-repeat guard. Models occasionally
4
+ * degenerate into long single-codepoint runs (observed in the wild: a
5
+ * 5968-char thinking block ending in 4655 consecutive 「【」, escalating over
6
+ * turns until the turn aborts). This is distinct from `repetitionGuard`
7
+ * (issue #308), which breaks byte-identical TOOL-CALL loops; here the
8
+ * attractor is inside the generated text/thinking itself.
9
+ *
10
+ * Why the adapter must act: pi replays prior assistant thinking back to the
11
+ * provider on every subsequent request (openai-completions sends it as
12
+ * `reasoning_content`, or as plain text when `requiresThinkingAsText`), and
13
+ * an aborted turn's partial message persists in the session log. A
14
+ * degenerated tail therefore rides along every later prompt, where the model
15
+ * sees its own previous output ending in thousands of repeated characters —
16
+ * a continuation bias that re-triggers the same degeneration, aborting the
17
+ * next turn too. The session dies in an abort loop with no recovery path.
18
+ *
19
+ * The pass below collapses runs >= minRun in assistant text/thinking of the
20
+ * OUTGOING view (persisted history is never modified) and appends a one-shot
21
+ * recovery notice while the degenerated message is the most recent assistant
22
+ * turn. Position-based self-limiting: a fresh model turn makes the old
23
+ * message non-last, so the notice stops appearing without any persistent
24
+ * state (and cannot accumulate — #223 lesson). */
25
+ export interface DegenerationGuardConfig {
26
+ /** Master switch. Default: true. `false` disables the pass entirely
27
+ * (kill-switch). */
28
+ enabled?: boolean;
29
+ /** Minimum length of a single-codepoint run (counted in codepoints) before
30
+ * it is treated as degeneration. Legitimate single-char runs in coding
31
+ * sessions (markdown hrules, dotted leaders, comment banners) stay well
32
+ * under this; observed pre-degeneration drift maxed at ~60 before the
33
+ * catastrophic 4655×「【」 run. Default: 200. Values below 8 are raised to
34
+ * 8: the collapse marker must stay shorter than any detectable run, and a
35
+ * sub-8 threshold would start matching ordinary dotted leaders/hrules. */
36
+ minRun?: number;
37
+ }
38
+ export declare const DEFAULT_DEGENERATION_GUARD: Required<DegenerationGuardConfig>;
39
+ /** Resolve the guard config, handling the boolean shorthand (`false`
40
+ * disables). Invalid minRun values (< 2 or non-numeric) fall back to the
41
+ * default with a logged warning — they never fail the session. Valid values
42
+ * below MIN_VALID_MIN_RUN are raised to it. */
43
+ export declare function resolveDegenerationGuard(cfg?: boolean | DegenerationGuardConfig): Required<DegenerationGuardConfig>;
44
+ /** One maximal run of a repeated codepoint. */
45
+ export interface DegenerateRun {
46
+ /** The repeated codepoint (string of length 1 or 2 — surrogate pairs kept whole). */
47
+ char: string;
48
+ /** Run length in codepoints. */
49
+ count: number;
50
+ /** UTF-16 index of the run start in the original string. */
51
+ index: number;
52
+ }
53
+ /** Find maximal runs of a single repeated codepoint with length >= minRun.
54
+ * Codepoint-safe: surrogate pairs count as one unit, so a run of astral
55
+ * characters is detected like any other. Returns [] for empty input or
56
+ * minRun < 2 (a "run" shorter than 2 carries no signal). */
57
+ export declare function findDegenerateRuns(text: string, minRun: number): DegenerateRun[];
58
+ /** Collapse every degenerate run in `text` into a short marker that names the
59
+ * character and its original count. Pure, idempotent, fail-safe. Returns the
60
+ * input unchanged when there is nothing to collapse. */
61
+ export declare function collapseDegenerateRuns(text: string, minRun: number): string;
62
+ /** Evidence of one rewritten message: which blocks changed and what was collapsed. */
63
+ export interface CollapseEvidence {
64
+ /** Index of the rewritten message in the input array. */
65
+ msgIndex: number;
66
+ /** Block kinds rewritten ("content" for string content, else "text"/"thinking"). */
67
+ blocks: string[];
68
+ /** Runs collapsed across those blocks. */
69
+ runs: DegenerateRun[];
70
+ }
71
+ /** Request-time pass: collapse degenerate single-codepoint runs in ASSISTANT
72
+ * messages' text/thinking blocks (string content included). Every position is
73
+ * scanned — including the current turn's partial assistant message, which is
74
+ * exactly the one pi will replay back to the provider on the next request.
75
+ * Tool-call arguments are never touched (rewriting them would desync the
76
+ * model's view from the call that actually executed). Pure: returns the same
77
+ * array reference when nothing changed; idempotent; fail-safe (any error
78
+ * returns the input unchanged). Persisted history is never modified. */
79
+ export declare function collapseAssistantDegeneration(messages: AgentMessage[], cfg?: boolean | DegenerationGuardConfig): {
80
+ messages: AgentMessage[];
81
+ evidence: CollapseEvidence[];
82
+ };
83
+ /** Scan backward for the LAST assistant message and collect its degenerate
84
+ * runs. Call it with the PERSISTED originals in session order (not the
85
+ * outgoing view): thinking-only aborted turns never reach the outgoing view
86
+ * (projectMessage drops them), yet they are still "the previous turn".
87
+ * Returns null when absent or clean. Gates the recovery notice: because the
88
+ * notice fires only while the degenerated message is the most recent
89
+ * assistant turn, it self-limits — once the model produces a new turn the
90
+ * old message is no longer last and the notice disappears on the next
91
+ * rebuild (no persistent state needed). */
92
+ export declare function lastAssistantRuns(messages: AgentMessage[], minRun: number): DegenerateRun[] | null;
93
+ /** One-shot recovery notice appended while a degenerated assistant message is
94
+ * the most recent turn: tells the model the repeated segment carries no
95
+ * information and was truncated, and to resume from the last valid step
96
+ * instead of continuing the attractor. */
97
+ export declare function degenerationNotice(runs: DegenerateRun[]): AgentMessage;
98
+ export {};
package/dist/index.js CHANGED
@@ -10,7 +10,7 @@ import { readFileSync as readFileSync2 } from "fs";
10
10
  import { homedir as homedir8 } from "os";
11
11
  import { join as join12 } from "path";
12
12
 
13
- // node_modules/acp-kernel/dist/chunk-JGNVAFFE.js
13
+ // node_modules/acp-kernel/dist/chunk-6TAK7DSI.js
14
14
  import { createRequire } from "module";
15
15
  var require2 = createRequire(import.meta.url);
16
16
  function defaultCountTokens(text) {
@@ -19,6 +19,12 @@ function defaultCountTokens(text) {
19
19
  const cjkCount = cjk?.length ?? 0;
20
20
  return cjkCount + Math.ceil((text.length - cjkCount) / 4);
21
21
  }
22
+ function thinkingTokenValue(thinking) {
23
+ return typeof thinking === "number" && Number.isFinite(thinking) && thinking > 0 ? thinking : 0;
24
+ }
25
+ function countMessageTokens(message, countTokens = defaultCountTokens) {
26
+ return countTokens(message.text ?? "") + thinkingTokenValue(message.thinkingTokens);
27
+ }
22
28
  function estimateTokensFast(text) {
23
29
  if (!text) return 0;
24
30
  return Math.ceil(text.length / 4);
@@ -1564,7 +1570,8 @@ function renderMessage(message, map, countTokens, strategy, snapshot = null) {
1564
1570
  "^" + escapeRegex(TAG_OPEN) + "[^>]*" + GT + escapeRegex(ref) + escapeRegex(TAG_CLOSE) + "\\n?"
1565
1571
  );
1566
1572
  const cleanText = (message.text || "").replace(ownTagRe, "");
1567
- const tokens = snapshot ? snapshot[ref] ?? (snapshot[ref] = countTokens(cleanText)) : countTokens(cleanText);
1573
+ const textTokens = snapshot ? snapshot[ref] ?? (snapshot[ref] = countTokens(cleanText)) : countTokens(cleanText);
1574
+ const tokens = textTokens + thinkingTokenValue(message.thinkingTokens);
1568
1575
  const type = classifyType(message);
1569
1576
  const prefix = acpTag(ref, tokens, type) + "\n";
1570
1577
  if (!cleanText) return { ...message, text: prefix };
@@ -1683,7 +1690,7 @@ function computeProtectedRefs(messages, state, config, countTokens = estimateTex
1683
1690
  if (isNeverPreserveRecent(msg)) continue;
1684
1691
  const ref = state.messageRefs.byRaw[msg.id];
1685
1692
  if (!ref || ref === "BLOCKED") continue;
1686
- visible.push({ ref, tokens: countTokens(msg.text ?? "") });
1693
+ visible.push({ ref, tokens: countMessageTokens(msg, countTokens) });
1687
1694
  }
1688
1695
  if (preserveN > 0) {
1689
1696
  for (const m2 of visible.slice(-preserveN)) {
@@ -1726,7 +1733,7 @@ function buildCompressibleRanges(messages, state, config, protectedZoneRefs, cou
1726
1733
  protectedMsgs.push({
1727
1734
  ref,
1728
1735
  gapBefore: skipSinceProtected,
1729
- tokens: countTokens(msg.text ?? ""),
1736
+ tokens: countMessageTokens(msg, countTokens),
1730
1737
  tools: msg.toolName ? [msg.toolName] : []
1731
1738
  });
1732
1739
  skipSinceProtected = false;
@@ -1741,7 +1748,7 @@ function buildCompressibleRanges(messages, state, config, protectedZoneRefs, cou
1741
1748
  compressibleMsgs.push({
1742
1749
  ref,
1743
1750
  gapBefore: skipSinceCompressible,
1744
- tokens: countTokens(msg.text ?? ""),
1751
+ tokens: countMessageTokens(msg, countTokens),
1745
1752
  chars: (msg.text ?? "").length,
1746
1753
  isTool: isToolMessage(msg),
1747
1754
  isUser: msg.role === "user"
@@ -2387,7 +2394,7 @@ function applySingleRange(input) {
2387
2394
  let compressedTokens = 0;
2388
2395
  for (const id of filteredIds) {
2389
2396
  const message = input.messages.find((entry) => entry.id === id);
2390
- compressedTokens += input.countTokens(message?.text ?? "");
2397
+ compressedTokens += message ? countMessageTokens(message, input.countTokens) : 0;
2391
2398
  }
2392
2399
  for (const consumedId of consumedBlockIds) {
2393
2400
  const consumed = blockById(input.state, consumedId);
@@ -2594,6 +2601,9 @@ function decideNudge(input) {
2594
2601
  const growthReady = firstSightMassReady || growthSinceReference >= growthFloor;
2595
2602
  const t2Count = tiers[2]?.targetBlocks.length ?? 0;
2596
2603
  const t3Count = tiers[3]?.targetBlocks.length ?? 0;
2604
+ const tierCountUsageFloor = config.nudge.minContextLimitPct;
2605
+ const t2CountReady = t2Count >= config.tiers.tier2Trigger && usage >= tierCountUsageFloor;
2606
+ const t3CountReady = t3Count >= config.tiers.tier3Trigger && usage >= tierCountUsageFloor;
2597
2607
  if (pressure) {
2598
2608
  const candidates = [1];
2599
2609
  if (config.tiers.enabled) {
@@ -2616,19 +2626,19 @@ function decideNudge(input) {
2616
2626
  if (t1Eff >= nudgeGrowthTokens) {
2617
2627
  injectedTier = 1;
2618
2628
  injectedReason = `T1 effective ${t1Eff} >= ${nudgeGrowthTokens}, growth ${growthSinceReference}, usage ${Math.round(usage * 100)}%`;
2619
- } else if (config.tiers.enabled && (t2Count >= config.tiers.tier2Trigger || t2Pen >= tier2Threshold && t2Pen > t1Eff)) {
2629
+ } else if (config.tiers.enabled && (t2CountReady || t2Pen >= tier2Threshold && t2Pen > t1Eff)) {
2620
2630
  const lastShown = state.nudge.lastShownByTier[2] ?? 0;
2621
2631
  const cadenceMet = lastShown === 0 || tokenCount - lastShown >= growthFloor;
2622
2632
  if (cadenceMet) {
2623
2633
  injectedTier = 2;
2624
- injectedReason = t2Count >= config.tiers.tier2Trigger ? `T2 distill ready: ${t2Count} tier-1 blocks >= tier2Trigger ${config.tiers.tier2Trigger} (${t2Pen} tokens), usage ${Math.round(usage * 100)}%` : `T2 distill ready: ${tiers[2].targetBlocks.length} tier-1 blocks (${t2Pen} tokens) >= ${tier2Threshold} (1.5x) and > T1 effective ${t1Eff}, usage ${Math.round(usage * 100)}%`;
2634
+ injectedReason = t2CountReady ? `T2 distill ready: ${t2Count} tier-1 blocks >= tier2Trigger ${config.tiers.tier2Trigger} (${t2Pen} tokens), usage ${Math.round(usage * 100)}%` : `T2 distill ready: ${tiers[2].targetBlocks.length} tier-1 blocks (${t2Pen} tokens) >= ${tier2Threshold} (1.5x) and > T1 effective ${t1Eff}, usage ${Math.round(usage * 100)}%`;
2625
2635
  }
2626
- } else if (config.tiers.enabled && (t3Count >= config.tiers.tier3Trigger || t3Pen >= tier2Threshold && t3Pen > t2Pen && t3Pen > t1Eff)) {
2636
+ } else if (config.tiers.enabled && (t3CountReady || t3Pen >= tier2Threshold && t3Pen > t2Pen && t3Pen > t1Eff)) {
2627
2637
  const lastShown = state.nudge.lastShownByTier[3] ?? 0;
2628
2638
  const cadenceMet = lastShown === 0 || tokenCount - lastShown >= growthFloor;
2629
2639
  if (cadenceMet) {
2630
2640
  injectedTier = 3;
2631
- injectedReason = t3Count >= config.tiers.tier3Trigger ? `T3 condense ready: ${t3Count} tier-2 blocks >= tier3Trigger ${config.tiers.tier3Trigger} (${t3Pen} tokens), usage ${Math.round(usage * 100)}%` : `T3 condense ready: ${tiers[3].targetBlocks.length} tier-2 blocks (${t3Pen} tokens) >= ${tier2Threshold} (1.5x) and > T2 ${t2Pen} and > T1 effective ${t1Eff}, usage ${Math.round(usage * 100)}%`;
2641
+ injectedReason = t3CountReady ? `T3 condense ready: ${t3Count} tier-2 blocks >= tier3Trigger ${config.tiers.tier3Trigger} (${t3Pen} tokens), usage ${Math.round(usage * 100)}%` : `T3 condense ready: ${tiers[3].targetBlocks.length} tier-2 blocks (${t3Pen} tokens) >= ${tier2Threshold} (1.5x) and > T2 ${t2Pen} and > T1 effective ${t1Eff}, usage ${Math.round(usage * 100)}%`;
2632
2642
  }
2633
2643
  }
2634
2644
  }
@@ -2645,9 +2655,14 @@ function decideNudge(input) {
2645
2655
  } else {
2646
2656
  const tiersList = [1, 2, 3];
2647
2657
  const eligible = tiersList.filter((t) => config.tiers.enabled || t === 1);
2648
- const countReady = (t) => t === 2 ? t2Count >= config.tiers.tier2Trigger : t === 3 ? t3Count >= config.tiers.tier3Trigger : false;
2658
+ const countReadyUngated = (t) => t === 2 ? t2Count >= config.tiers.tier2Trigger : t === 3 ? t3Count >= config.tiers.tier3Trigger : false;
2659
+ const countReady = (t) => countReadyUngated(t) && usage >= tierCountUsageFloor;
2649
2660
  const ready = eligible.filter((t) => (tiers[t]?.pending ?? 0) >= nudgeGrowthTokens).map((t) => `T${t} ${tiers[t].pending}`);
2650
- const readyCount = eligible.filter((t) => (tiers[t]?.pending ?? 0) < nudgeGrowthTokens && countReady(t)).map((t) => `T${t} ${t === 2 ? t2Count : t3Count} blocks (count)`);
2661
+ const readyCount = eligible.filter(
2662
+ (t) => (tiers[t]?.pending ?? 0) < nudgeGrowthTokens && countReadyUngated(t)
2663
+ ).map(
2664
+ (t) => `T${t} ${t === 2 ? t2Count : t3Count} blocks (count${usage >= tierCountUsageFloor ? "" : ", usage-gated"})`
2665
+ );
2651
2666
  const readyAll = [...ready, ...readyCount];
2652
2667
  const readyHint = readyAll.length > 0 ? `, ready: ${readyAll.join(", ")}` : "";
2653
2668
  const blocked = eligible.filter(
@@ -2709,7 +2724,7 @@ function computeContextBreakdown(messages, total, growth, countTokens) {
2709
2724
  const count = countTokens ?? ((t) => Math.ceil(t.length / 4));
2710
2725
  let system = 0, tool = 0, summaries = 0, code = 0, text = 0;
2711
2726
  for (const msg of messages) {
2712
- const tokens = count(msg.text ?? "");
2727
+ const tokens = countMessageTokens(msg, count);
2713
2728
  if (msg.text?.startsWith("[Compressed conversation section]")) {
2714
2729
  summaries += tokens;
2715
2730
  } else if (msg.contentType === "tool-call" || msg.contentType === "tool-result") {
@@ -2871,7 +2886,7 @@ function collectVisible(messages, state, countTokens) {
2871
2886
  if (coveredIds.has(message.id)) return;
2872
2887
  const ref = refForRaw(state.messageRefs, message.id);
2873
2888
  if (!ref) return;
2874
- const tokens = countTokens(message.text ?? "");
2889
+ const tokens = countMessageTokens(message, countTokens);
2875
2890
  const tool = isToolMessage(message) ? message.toolName ?? (message.toolCallId ? toolCallNames.get(message.toolCallId) : void 0) ?? "tool" : "text";
2876
2891
  if (tokens > 0) visible.push({ ref, tokens, tool, index });
2877
2892
  });
@@ -4088,6 +4103,8 @@ function projectMessage(message, id) {
4088
4103
  }];
4089
4104
  }
4090
4105
  if (role === "assistant") {
4106
+ const thinking = thinkingTokenCount(msg.content);
4107
+ const thinkingField = thinking > 0 ? { thinkingTokens: thinking } : {};
4091
4108
  const calls = allToolCalls(msg.content);
4092
4109
  if (calls.length > 0) {
4093
4110
  const textParts = extractText(msg.content);
@@ -4096,9 +4113,9 @@ function projectMessage(message, id) {
4096
4113
  const argStr = stringifyArgs(call.arguments);
4097
4114
  const text2 = argStr && textParts ? `${textParts}
4098
4115
  ${argStr}` : argStr || textParts;
4099
- return [{ id, role: "assistant", contentType: "tool-call", toolName: call.name, toolCallId: call.id, text: text2 }];
4116
+ return [{ id, role: "assistant", contentType: "tool-call", toolName: call.name, toolCallId: call.id, text: text2, ...thinkingField }];
4100
4117
  }
4101
- return calls.map((call) => {
4118
+ return calls.map((call, i) => {
4102
4119
  const argStr = stringifyArgs(call.arguments);
4103
4120
  return {
4104
4121
  id: `${id}#${call.id}`,
@@ -4106,13 +4123,14 @@ ${argStr}` : argStr || textParts;
4106
4123
  contentType: "tool-call",
4107
4124
  toolName: call.name,
4108
4125
  toolCallId: call.id,
4109
- text: argStr || textParts
4126
+ text: argStr || textParts,
4127
+ ...i === 0 ? thinkingField : {}
4110
4128
  };
4111
4129
  });
4112
4130
  }
4113
4131
  const text = extractText(msg.content);
4114
4132
  if (!text.trim()) return [];
4115
- return [{ id, role: "assistant", contentType: "text", text }];
4133
+ return [{ id, role: "assistant", contentType: "text", text, ...thinkingField }];
4116
4134
  }
4117
4135
  const customText = extractText(msg.content) || fallbackText(msg);
4118
4136
  return customText.length > 0 ? [{ id, role: "user", contentType: "text", text: customText }] : [];
@@ -4140,6 +4158,15 @@ function extractText(content) {
4140
4158
  }
4141
4159
  return parts.join("\n");
4142
4160
  }
4161
+ function thinkingTokenCount(content) {
4162
+ if (!Array.isArray(content)) return 0;
4163
+ const parts = [];
4164
+ for (const block of content) {
4165
+ const b2 = block;
4166
+ if (b2.type === "thinking" && typeof b2.thinking === "string") parts.push(b2.thinking);
4167
+ }
4168
+ return parts.length > 0 ? defaultCountTokens(parts.join("\n")) : 0;
4169
+ }
4143
4170
  function stripRefTag(text) {
4144
4171
  return text.replace(REF_TAG, "").replace(TRAILING_REF_TAG, "");
4145
4172
  }
@@ -4739,6 +4766,8 @@ var KNOWN = /* @__PURE__ */ new Set([
4739
4766
  "displayUsage",
4740
4767
  "throttleRetry",
4741
4768
  "outputHeadroomMaxPct",
4769
+ "repetitionGuard",
4770
+ "degenerationGuard",
4742
4771
  "prompts",
4743
4772
  "acknowledgePromptsRisk"
4744
4773
  ]);
@@ -9767,6 +9796,7 @@ function estimateTokens(messages, coveredIds, imageTokensById) {
9767
9796
  if (m2.toolName === "compress") continue;
9768
9797
  if (coveredIds?.has(m2.id)) continue;
9769
9798
  tokens += defaultCountTokens(m2.text ?? "");
9799
+ tokens += m2.thinkingTokens ?? 0;
9770
9800
  const img = imageTokensById?.get(m2.id);
9771
9801
  if (img) tokens += img;
9772
9802
  }
@@ -16831,12 +16861,13 @@ async function statusReport(runtime, ctx) {
16831
16861
  const systemPromptTokens = systemPromptText ? defaultCountTokens(systemPromptText) : 0;
16832
16862
  const imageTokens = collectImageTokens(entries, modelSupportsImages(ctx.model));
16833
16863
  const imageTokensTotal = [...imageTokens.values()].reduce((a, b2) => a + b2, 0);
16834
- const sessionTokens = !anchorStale && realUsage?.tokens && realUsage.tokens > 0 ? realUsage.tokens : defaultCountTokens(coreMessages.map((m2) => m2.text ?? "").join("\n")) + imageTokensTotal;
16864
+ const thinkingTokensTotal = coreMessages.reduce((sum, m2) => sum + (m2.thinkingTokens ?? 0), 0);
16865
+ const sessionTokens = !anchorStale && realUsage?.tokens && realUsage.tokens > 0 ? realUsage.tokens : defaultCountTokens(coreMessages.map((m2) => m2.text ?? "").join("\n")) + imageTokensTotal + thinkingTokensTotal;
16835
16866
  const coveredIds = collectCoveredMessageIds(state);
16836
16867
  const sentTokens = estimateTokens(coreMessages, coveredIds, imageTokens) + systemPromptTokens;
16837
16868
  const viewSentTokens = adjustedTokenCount(runtime.core, coreMessages, state, config, sentTokens, imageTokens, systemPromptTokens);
16838
16869
  const turn = runtime.core.processTurn({ messages: coreMessages, state, config, tokenCount: anchorStale ? viewSentTokens : Math.max(viewSentTokens, realUsage?.tokens ?? 0) });
16839
- const versionStr = "0.1.63" ? `billion-context-pi@${"0.1.63"}` : void 0;
16870
+ const versionStr = "0.1.64" ? `billion-context-pi@${"0.1.64"}` : void 0;
16840
16871
  let text = buildStatusPanel({
16841
16872
  version: versionStr,
16842
16873
  tokenCount: sessionTokens,
@@ -16844,7 +16875,7 @@ async function statusReport(runtime, ctx) {
16844
16875
  state: turn.state,
16845
16876
  nudge: turn.nudge,
16846
16877
  modelContextLimit: config.modelContextLimit,
16847
- unprunedTokens: coreMessages.reduce((sum, m2) => sum + defaultCountTokens(m2.text ?? "") + (imageTokens.get(m2.id) ?? 0), 0),
16878
+ unprunedTokens: coreMessages.reduce((sum, m2) => sum + defaultCountTokens(m2.text ?? "") + (m2.thinkingTokens ?? 0) + (imageTokens.get(m2.id) ?? 0), 0),
16848
16879
  cacheUsages: cacheUsageSamples(entries ?? [])
16849
16880
  });
16850
16881
  const delegateUsage = getDelegateUsage();
@@ -16857,6 +16888,170 @@ async function statusReport(runtime, ctx) {
16857
16888
  return text;
16858
16889
  }
16859
16890
 
16891
+ // src/degeneration.ts
16892
+ var DEFAULT_DEGENERATION_GUARD = { enabled: true, minRun: 200 };
16893
+ var MIN_VALID_MIN_RUN = 8;
16894
+ function resolveDegenerationGuard(cfg) {
16895
+ if (cfg === false) return { enabled: false, minRun: DEFAULT_DEGENERATION_GUARD.minRun };
16896
+ const c = typeof cfg === "object" && cfg !== null ? cfg : {};
16897
+ let minRun = DEFAULT_DEGENERATION_GUARD.minRun;
16898
+ if (c.minRun !== void 0) {
16899
+ const n = c.minRun;
16900
+ if (typeof n === "number" && Number.isFinite(n) && n >= 2) {
16901
+ minRun = Math.max(MIN_VALID_MIN_RUN, Math.floor(n));
16902
+ } else {
16903
+ logWarn("config", { event: "degeneration-guard-invalid", field: "minRun", value: n, fallback: DEFAULT_DEGENERATION_GUARD.minRun });
16904
+ }
16905
+ }
16906
+ return { enabled: c.enabled !== false, minRun };
16907
+ }
16908
+ var preScreenMinRun = Number.NaN;
16909
+ var preScreenRe = null;
16910
+ function hasLongRun(text, minRun) {
16911
+ const n = Math.floor(minRun);
16912
+ if (preScreenRe === null || preScreenMinRun !== n) {
16913
+ preScreenMinRun = n;
16914
+ preScreenRe = new RegExp(`(.)\\1{${Math.max(n - 1, 0)},}`, "su");
16915
+ }
16916
+ return preScreenRe.test(text);
16917
+ }
16918
+ function findDegenerateRuns(text, minRun) {
16919
+ const runs2 = [];
16920
+ if (!text || !Number.isFinite(minRun) || minRun < 2) return runs2;
16921
+ if (!hasLongRun(text, minRun)) return runs2;
16922
+ const chars = Array.from(text);
16923
+ let utf16 = 0;
16924
+ let k = 0;
16925
+ while (k < chars.length) {
16926
+ const ch = chars[k];
16927
+ let m2 = k + 1;
16928
+ while (m2 < chars.length && chars[m2] === ch) m2++;
16929
+ const count = m2 - k;
16930
+ if (count >= minRun) runs2.push({ char: ch, count, index: utf16 });
16931
+ utf16 += count * ch.length;
16932
+ k = m2;
16933
+ }
16934
+ return runs2;
16935
+ }
16936
+ function describeChar(ch) {
16937
+ if (ch === " ") return "space";
16938
+ if (ch === "\n") return "newline";
16939
+ if (ch === " ") return "tab";
16940
+ if (ch === "\r") return "carriage-return";
16941
+ const cp = ch.codePointAt(0) ?? 0;
16942
+ if (cp < 32 || cp === 127) return `U+${cp.toString(16).toUpperCase().padStart(4, "0")}`;
16943
+ return JSON.stringify(ch);
16944
+ }
16945
+ function applyRuns(text, runs2, minRun) {
16946
+ const keep = Math.max(1, Math.min(3, minRun - 1));
16947
+ let out = "";
16948
+ let last = 0;
16949
+ for (const r of runs2) {
16950
+ out += text.slice(last, r.index);
16951
+ out += `${r.char.repeat(keep)}\u2026 [${r.count}\xD7 identical chars cut \u2014 degenerate repeat]`;
16952
+ last = r.index + r.count * r.char.length;
16953
+ }
16954
+ out += text.slice(last);
16955
+ return out;
16956
+ }
16957
+ function collapseAssistantDegeneration(messages, cfg) {
16958
+ const { enabled, minRun } = resolveDegenerationGuard(cfg);
16959
+ if (!enabled || messages.length === 0) return { messages, evidence: [] };
16960
+ try {
16961
+ const evidence = [];
16962
+ let changed = false;
16963
+ const out = messages.slice();
16964
+ for (let i = 0; i < messages.length; i++) {
16965
+ const msg = messages[i];
16966
+ if (msg.role !== "assistant") continue;
16967
+ const c = msg.content;
16968
+ if (typeof c === "string") {
16969
+ const runs3 = findDegenerateRuns(c, minRun);
16970
+ if (runs3.length > 0) {
16971
+ out[i] = { ...msg, content: applyRuns(c, runs3, minRun) };
16972
+ evidence.push({ msgIndex: i, blocks: ["content"], runs: runs3 });
16973
+ changed = true;
16974
+ }
16975
+ continue;
16976
+ }
16977
+ if (!Array.isArray(c)) continue;
16978
+ const blocks = [];
16979
+ const runs2 = [];
16980
+ let blockChanged = false;
16981
+ const nc = c.map((p) => {
16982
+ const b2 = p;
16983
+ if (b2?.type === "text" && typeof b2.text === "string") {
16984
+ const r = findDegenerateRuns(b2.text, minRun);
16985
+ if (r.length > 0) {
16986
+ blockChanged = true;
16987
+ blocks.push("text");
16988
+ runs2.push(...r);
16989
+ return { ...b2, text: applyRuns(b2.text, r, minRun) };
16990
+ }
16991
+ return p;
16992
+ }
16993
+ if (b2?.type === "thinking" && typeof b2.thinking === "string") {
16994
+ const r = findDegenerateRuns(b2.thinking, minRun);
16995
+ if (r.length > 0) {
16996
+ blockChanged = true;
16997
+ blocks.push("thinking");
16998
+ runs2.push(...r);
16999
+ return { ...b2, thinking: applyRuns(b2.thinking, r, minRun) };
17000
+ }
17001
+ return p;
17002
+ }
17003
+ return p;
17004
+ });
17005
+ if (blockChanged) {
17006
+ out[i] = { ...msg, content: nc };
17007
+ evidence.push({ msgIndex: i, blocks, runs: runs2 });
17008
+ changed = true;
17009
+ }
17010
+ }
17011
+ return { messages: changed ? out : messages, evidence };
17012
+ } catch {
17013
+ return { messages, evidence: [] };
17014
+ }
17015
+ }
17016
+ function lastAssistantRuns(messages, minRun) {
17017
+ try {
17018
+ for (let i = messages.length - 1; i >= 0; i--) {
17019
+ const msg = messages[i];
17020
+ if (msg.role !== "assistant") continue;
17021
+ const runs2 = [];
17022
+ const c = msg.content;
17023
+ if (typeof c === "string") {
17024
+ runs2.push(...findDegenerateRuns(c, minRun));
17025
+ } else if (Array.isArray(c)) {
17026
+ for (const p of c) {
17027
+ const b2 = p;
17028
+ if (b2?.type === "text" && typeof b2.text === "string") runs2.push(...findDegenerateRuns(b2.text, minRun));
17029
+ else if (b2?.type === "thinking" && typeof b2.thinking === "string") runs2.push(...findDegenerateRuns(b2.thinking, minRun));
17030
+ }
17031
+ }
17032
+ return runs2.length > 0 ? runs2 : null;
17033
+ }
17034
+ return null;
17035
+ } catch {
17036
+ return null;
17037
+ }
17038
+ }
17039
+ function degenerationNotice(runs2) {
17040
+ const sorted = [...runs2].sort((a, b2) => b2.count - a.count);
17041
+ const top = sorted[0];
17042
+ const extra = sorted.length > 1 ? ` and ${sorted.length - 1} other repeated segment(s)` : "";
17043
+ return {
17044
+ role: "user",
17045
+ content: [
17046
+ {
17047
+ type: "text",
17048
+ text: `[ACP recovery notice] Your previous turn ended in degenerate generation: its output contained ${top.count} consecutive repetitions of ${describeChar(top.char)}${extra}. That repeated segment carries no information and has been truncated in the context above. Do not reproduce it or continue the pattern. Resume your task from your last valid step.`
17049
+ }
17050
+ ],
17051
+ timestamp: Date.now()
17052
+ };
17053
+ }
17054
+
16860
17055
  // src/system-prompt.ts
16861
17056
  function buildAcpSystemPrompt(prompts) {
16862
17057
  return `
@@ -17370,7 +17565,7 @@ async function autoInstallLatest(latest, extDirOverride) {
17370
17565
  "--no-fund",
17371
17566
  "--no-save"
17372
17567
  ];
17373
- const prevVersion = (await readPackageJson(join11(extDir, "package.json")))?.version ?? "0.1.63";
17568
+ const prevVersion = (await readPackageJson(join11(extDir, "package.json")))?.version ?? "0.1.64";
17374
17569
  const { code, stderr } = await runNpmImpl(installArgs(latest), { cwd: npmDir, timeout: 6e4 });
17375
17570
  if (code !== 0) {
17376
17571
  logWarn("update", {
@@ -17388,7 +17583,7 @@ async function autoInstallLatest(latest, extDirOverride) {
17388
17583
  }
17389
17584
  const verify = await verifyInstall(npmDir, latest);
17390
17585
  if (!verify.ok) {
17391
- const rollbackTo = SEMVER_RE.test(prevVersion) ? prevVersion : "0.1.63";
17586
+ const rollbackTo = SEMVER_RE.test(prevVersion) ? prevVersion : "0.1.64";
17392
17587
  logWarn("update", { event: "auto-install-verify-failed", latest, reason: verify.reason, rollbackTo });
17393
17588
  const rb = await runNpmImpl(installArgs(rollbackTo), { cwd: npmDir, timeout: 6e4 });
17394
17589
  logInfo("update", { event: "rollback", from: latest, to: rollbackTo, ok: rb.code === 0 });
@@ -17463,7 +17658,7 @@ async function checkForUpdate(autoUpdate, notify) {
17463
17658
  }
17464
17659
  const latest = await fetchLatestVersion(tag);
17465
17660
  if (!latest) return;
17466
- const current = runtimeVersion ?? "0.1.63";
17661
+ const current = runtimeVersion ?? "0.1.64";
17467
17662
  const hasUpdate = isVersionNewer(latest, current);
17468
17663
  debug.event("update-check", {
17469
17664
  current,
@@ -17641,7 +17836,7 @@ function wireSessionLifecycle(pi, runtime, standDownIfProxied) {
17641
17836
  setDelegatePolicy(DEFAULT_DELEGATE_POLICY);
17642
17837
  const sid = ctx.sessionManager.getSessionId();
17643
17838
  const modelInfo = ctx.model;
17644
- logInfo("session", { event: "start", sid, cwd: ctx.cwd, debug: runtime.adapter.debug ?? null, version: true ? "0.1.63" : null, model: modelInfo?.id ?? null, modelApi: modelInfo?.api ?? null, contextWindow: modelInfo?.contextWindow ?? null });
17839
+ logInfo("session", { event: "start", sid, cwd: ctx.cwd, debug: runtime.adapter.debug ?? null, version: true ? "0.1.64" : null, model: modelInfo?.id ?? null, modelApi: modelInfo?.api ?? null, contextWindow: modelInfo?.contextWindow ?? null });
17645
17840
  try {
17646
17841
  await runtime.reloadConfig(ctx.cwd);
17647
17842
  const delegateCfg = resolveDelegate(runtime.adapter);
@@ -17684,6 +17879,7 @@ function wireSessionLifecycle(pi, runtime, standDownIfProxied) {
17684
17879
  closeLogStream();
17685
17880
  });
17686
17881
  }
17882
+ var lastDegNoticeKey = null;
17687
17883
  function wireContextTransform(pi, runtime, standDownIfProxied) {
17688
17884
  pi.on("context", async (event, ctx) => {
17689
17885
  if (runtime.refused) return;
@@ -17796,6 +17992,24 @@ function wireContextTransform(pi, runtime, standDownIfProxied) {
17796
17992
  debug.event("reasoning-drop", { sid, droppedChars, drop: reasoningDrop.drop, threshold: reasoningDrop.threshold });
17797
17993
  }
17798
17994
  rebuilt = droppedThinking;
17995
+ const degCfg = resolveDegenerationGuard(runtime.adapter.degenerationGuard);
17996
+ const tailRuns = degCfg.enabled ? lastAssistantRuns([...originalById.values()], degCfg.minRun) : null;
17997
+ const deg = collapseAssistantDegeneration(rebuilt, degCfg);
17998
+ if (deg.messages !== rebuilt) {
17999
+ rebuilt = deg.messages;
18000
+ const maxRun = deg.evidence.reduce((m2, e) => e.runs.reduce((x2, r) => Math.max(x2, r.count), m2), 0);
18001
+ logWarn("degeneration", { sid, event: "runs-collapsed", msgs: deg.evidence.length, maxRun, minRun: degCfg.minRun });
18002
+ debug.event("degeneration-collapsed", { sid, msgs: deg.evidence.length, maxRun });
18003
+ }
18004
+ if (tailRuns) {
18005
+ rebuilt.push(degenerationNotice(tailRuns));
18006
+ const top = [...tailRuns].sort((a, b2) => b2.count - a.count)[0];
18007
+ logInfo("degeneration", { sid, event: "recovery-notice", maxRun: top.count });
18008
+ if (ctx.hasUI && lastDegNoticeKey !== `${sid}:${top.count}:${top.char.codePointAt(0)}`) {
18009
+ lastDegNoticeKey = `${sid}:${top.count}:${top.char.codePointAt(0)}`;
18010
+ ctx.ui.notify(`[ACP] previous turn ended in degenerate generation (${top.count}\xD7 repeat) \u2014 the repeated segment was truncated above and a recovery notice injected.`);
18011
+ }
18012
+ }
17799
18013
  const debugOn2 = debug.enabled;
17800
18014
  const turnKey = lastUserMessageId(entries) ?? sid;
17801
18015
  const compressOutcomes = collectCompressOutcomes(entries, turnStartIndex(entries));