@prestyj/cli 5.7.0 → 5.8.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (193) hide show
  1. package/README.md +2 -2
  2. package/dist/app-sidecar.js +243 -54
  3. package/dist/app-sidecar.js.map +1 -1
  4. package/dist/cli/auth.d.ts.map +1 -1
  5. package/dist/cli/auth.js +5 -2
  6. package/dist/cli/auth.js.map +1 -1
  7. package/dist/cli/shared.d.ts.map +1 -1
  8. package/dist/cli/shared.js +2 -0
  9. package/dist/cli/shared.js.map +1 -1
  10. package/dist/cli.d.ts.map +1 -1
  11. package/dist/cli.js +18 -3
  12. package/dist/cli.js.map +1 -1
  13. package/dist/config.d.ts.map +1 -1
  14. package/dist/config.js +1 -0
  15. package/dist/config.js.map +1 -1
  16. package/dist/config.test.js +7 -0
  17. package/dist/config.test.js.map +1 -1
  18. package/dist/core/agent-session-compaction.test.js +521 -10
  19. package/dist/core/agent-session-compaction.test.js.map +1 -1
  20. package/dist/core/agent-session-memory-tail.test.js +1 -1
  21. package/dist/core/agent-session-memory-tail.test.js.map +1 -1
  22. package/dist/core/agent-session-queue.test.js +1 -1
  23. package/dist/core/agent-session-queue.test.js.map +1 -1
  24. package/dist/core/agent-session-review-coverage.test.js +26 -0
  25. package/dist/core/agent-session-review-coverage.test.js.map +1 -1
  26. package/dist/core/agent-session-tool-result-policy.test.d.ts +2 -0
  27. package/dist/core/agent-session-tool-result-policy.test.d.ts.map +1 -0
  28. package/dist/core/agent-session-tool-result-policy.test.js +25 -0
  29. package/dist/core/agent-session-tool-result-policy.test.js.map +1 -0
  30. package/dist/core/agent-session.d.ts +40 -8
  31. package/dist/core/agent-session.d.ts.map +1 -1
  32. package/dist/core/agent-session.js +285 -87
  33. package/dist/core/agent-session.js.map +1 -1
  34. package/dist/core/auth-providers.d.ts.map +1 -1
  35. package/dist/core/auth-providers.js +15 -8
  36. package/dist/core/auth-providers.js.map +1 -1
  37. package/dist/core/compaction/active-context.d.ts +17 -0
  38. package/dist/core/compaction/active-context.d.ts.map +1 -0
  39. package/dist/core/compaction/active-context.js +20 -0
  40. package/dist/core/compaction/active-context.js.map +1 -0
  41. package/dist/core/compaction/active-context.test.d.ts +2 -0
  42. package/dist/core/compaction/active-context.test.d.ts.map +1 -0
  43. package/dist/core/compaction/active-context.test.js +36 -0
  44. package/dist/core/compaction/active-context.test.js.map +1 -0
  45. package/dist/core/compaction/compactor.d.ts +19 -18
  46. package/dist/core/compaction/compactor.d.ts.map +1 -1
  47. package/dist/core/compaction/compactor.js +64 -42
  48. package/dist/core/compaction/compactor.js.map +1 -1
  49. package/dist/core/compaction/compactor.test.js +122 -65
  50. package/dist/core/compaction/compactor.test.js.map +1 -1
  51. package/dist/core/compaction/token-estimator.d.ts +25 -1
  52. package/dist/core/compaction/token-estimator.d.ts.map +1 -1
  53. package/dist/core/compaction/token-estimator.js +86 -1
  54. package/dist/core/compaction/token-estimator.js.map +1 -1
  55. package/dist/core/compaction/token-estimator.test.js +147 -1
  56. package/dist/core/compaction/token-estimator.test.js.map +1 -1
  57. package/dist/core/compaction/tool-result-pruner.d.ts +37 -0
  58. package/dist/core/compaction/tool-result-pruner.d.ts.map +1 -0
  59. package/dist/core/compaction/tool-result-pruner.js +103 -0
  60. package/dist/core/compaction/tool-result-pruner.js.map +1 -0
  61. package/dist/core/compaction/tool-result-pruner.test.d.ts +2 -0
  62. package/dist/core/compaction/tool-result-pruner.test.d.ts.map +1 -0
  63. package/dist/core/compaction/tool-result-pruner.test.js +148 -0
  64. package/dist/core/compaction/tool-result-pruner.test.js.map +1 -0
  65. package/dist/core/ideal-review.d.ts +6 -0
  66. package/dist/core/ideal-review.d.ts.map +1 -1
  67. package/dist/core/ideal-review.js +14 -0
  68. package/dist/core/ideal-review.js.map +1 -1
  69. package/dist/core/ideal-review.test.js +10 -1
  70. package/dist/core/ideal-review.test.js.map +1 -1
  71. package/dist/core/index.d.ts +1 -1
  72. package/dist/core/index.d.ts.map +1 -1
  73. package/dist/core/index.js +1 -1
  74. package/dist/core/index.js.map +1 -1
  75. package/dist/core/project-discovery.d.ts +6 -9
  76. package/dist/core/project-discovery.d.ts.map +1 -1
  77. package/dist/core/project-discovery.js +35 -28
  78. package/dist/core/project-discovery.js.map +1 -1
  79. package/dist/core/project-discovery.test.js +148 -0
  80. package/dist/core/project-discovery.test.js.map +1 -1
  81. package/dist/core/resolve-start.test.js +1 -0
  82. package/dist/core/resolve-start.test.js.map +1 -1
  83. package/dist/core/session-compaction.d.ts +3 -0
  84. package/dist/core/session-compaction.d.ts.map +1 -1
  85. package/dist/core/session-compaction.js +14 -1
  86. package/dist/core/session-compaction.js.map +1 -1
  87. package/dist/core/session-compaction.test.js +5 -0
  88. package/dist/core/session-compaction.test.js.map +1 -1
  89. package/dist/core/session-manager.d.ts +8 -1
  90. package/dist/core/session-manager.d.ts.map +1 -1
  91. package/dist/core/session-manager.js +7 -1
  92. package/dist/core/session-manager.js.map +1 -1
  93. package/dist/core/session-manager.test.js +17 -0
  94. package/dist/core/session-manager.test.js.map +1 -1
  95. package/dist/core/session-preview.d.ts +11 -0
  96. package/dist/core/session-preview.d.ts.map +1 -0
  97. package/dist/core/session-preview.js +50 -0
  98. package/dist/core/session-preview.js.map +1 -0
  99. package/dist/core/settings-manager.d.ts +1 -0
  100. package/dist/core/settings-manager.d.ts.map +1 -1
  101. package/dist/core/settings-manager.js +3 -2
  102. package/dist/core/settings-manager.js.map +1 -1
  103. package/dist/core/subagent-manager.d.ts +2 -4
  104. package/dist/core/subagent-manager.d.ts.map +1 -1
  105. package/dist/core/subagent-manager.js +4 -0
  106. package/dist/core/subagent-manager.js.map +1 -1
  107. package/dist/core/subagent-manager.test.js +13 -3
  108. package/dist/core/subagent-manager.test.js.map +1 -1
  109. package/dist/tools/bash.d.ts +9 -0
  110. package/dist/tools/bash.d.ts.map +1 -1
  111. package/dist/tools/bash.js +28 -18
  112. package/dist/tools/bash.js.map +1 -1
  113. package/dist/tools/bash.test.d.ts +2 -0
  114. package/dist/tools/bash.test.d.ts.map +1 -0
  115. package/dist/tools/bash.test.js +56 -0
  116. package/dist/tools/bash.test.js.map +1 -0
  117. package/dist/tools/overflow.d.ts +12 -2
  118. package/dist/tools/overflow.d.ts.map +1 -1
  119. package/dist/tools/overflow.js +49 -4
  120. package/dist/tools/overflow.js.map +1 -1
  121. package/dist/tools/overflow.test.d.ts +2 -0
  122. package/dist/tools/overflow.test.d.ts.map +1 -0
  123. package/dist/tools/overflow.test.js +62 -0
  124. package/dist/tools/overflow.test.js.map +1 -0
  125. package/dist/tools/subagent-shared.d.ts +8 -0
  126. package/dist/tools/subagent-shared.d.ts.map +1 -1
  127. package/dist/tools/subagent-shared.js.map +1 -1
  128. package/dist/tools/subagent.d.ts +3 -8
  129. package/dist/tools/subagent.d.ts.map +1 -1
  130. package/dist/tools/subagent.js +10 -8
  131. package/dist/tools/subagent.js.map +1 -1
  132. package/dist/tools/subagent.test.js +17 -1
  133. package/dist/tools/subagent.test.js.map +1 -1
  134. package/dist/ui/App.d.ts +0 -2
  135. package/dist/ui/App.d.ts.map +1 -1
  136. package/dist/ui/App.js +6 -82
  137. package/dist/ui/App.js.map +1 -1
  138. package/dist/ui/components/ModelSelector.d.ts.map +1 -1
  139. package/dist/ui/components/ModelSelector.js +1 -0
  140. package/dist/ui/components/ModelSelector.js.map +1 -1
  141. package/dist/ui/components/SubAgentPanel.d.ts +2 -0
  142. package/dist/ui/components/SubAgentPanel.d.ts.map +1 -1
  143. package/dist/ui/components/SubAgentPanel.js +4 -2
  144. package/dist/ui/components/SubAgentPanel.js.map +1 -1
  145. package/dist/ui/hooks/useAgentLoop.d.ts +2 -4
  146. package/dist/ui/hooks/useAgentLoop.d.ts.map +1 -1
  147. package/dist/ui/hooks/useAgentLoop.js +7 -3
  148. package/dist/ui/hooks/useAgentLoop.js.map +1 -1
  149. package/dist/ui/hooks/useAgentLoop.test.js +31 -0
  150. package/dist/ui/hooks/useAgentLoop.test.js.map +1 -1
  151. package/dist/ui/hooks/useContextCompaction.d.ts +5 -8
  152. package/dist/ui/hooks/useContextCompaction.d.ts.map +1 -1
  153. package/dist/ui/hooks/useContextCompaction.js +77 -16
  154. package/dist/ui/hooks/useContextCompaction.js.map +1 -1
  155. package/dist/ui/hooks/useContextCompaction.test.d.ts +2 -0
  156. package/dist/ui/hooks/useContextCompaction.test.d.ts.map +1 -0
  157. package/dist/ui/hooks/useContextCompaction.test.js +135 -0
  158. package/dist/ui/hooks/useContextCompaction.test.js.map +1 -0
  159. package/dist/ui/hooks/useSessionPersistence.d.ts.map +1 -1
  160. package/dist/ui/hooks/useSessionPersistence.js +3 -0
  161. package/dist/ui/hooks/useSessionPersistence.js.map +1 -1
  162. package/dist/ui/hooks/useTerminalTitle.d.ts +3 -3
  163. package/dist/ui/hooks/useTerminalTitle.d.ts.map +1 -1
  164. package/dist/ui/hooks/useTerminalTitle.js +5 -9
  165. package/dist/ui/hooks/useTerminalTitle.js.map +1 -1
  166. package/dist/ui/login.d.ts.map +1 -1
  167. package/dist/ui/login.js +7 -2
  168. package/dist/ui/login.js.map +1 -1
  169. package/dist/ui/render.d.ts +0 -2
  170. package/dist/ui/render.d.ts.map +1 -1
  171. package/dist/ui/render.js +0 -4
  172. package/dist/ui/render.js.map +1 -1
  173. package/dist/ui/render.test.js +10 -1
  174. package/dist/ui/render.test.js.map +1 -1
  175. package/dist/ui/terminal-history.js +6 -2
  176. package/dist/ui/terminal-history.js.map +1 -1
  177. package/dist/ui/terminal-history.test.js +2 -2
  178. package/dist/ui/terminal-history.test.js.map +1 -1
  179. package/dist/ui/tui-history-parity.test.js +8 -1
  180. package/dist/ui/tui-history-parity.test.js.map +1 -1
  181. package/dist/utils/git.d.ts +2 -0
  182. package/dist/utils/git.d.ts.map +1 -1
  183. package/dist/utils/git.js +12 -0
  184. package/dist/utils/git.js.map +1 -1
  185. package/dist/utils/git.test.d.ts +2 -0
  186. package/dist/utils/git.test.d.ts.map +1 -0
  187. package/dist/utils/git.test.js +36 -0
  188. package/dist/utils/git.test.js.map +1 -0
  189. package/package.json +4 -4
  190. package/dist/utils/session-title.d.ts +0 -19
  191. package/dist/utils/session-title.d.ts.map +0 -1
  192. package/dist/utils/session-title.js +0 -82
  193. package/dist/utils/session-title.js.map +0 -1
@@ -1,4 +1,4 @@
1
- import { agentLoop, isAbortError, } from "@prestyj/agent";
1
+ import { agentLoop, isAbortError, isUsageLimitError, } from "@prestyj/agent";
2
2
  import { ProviderError, } from "@prestyj/ai";
3
3
  import { EventBus } from "./event-bus.js";
4
4
  import { SlashCommandRegistry, createBuiltinCommands, } from "./slash-commands.js";
@@ -6,12 +6,13 @@ import { PROMPT_COMMANDS, getPromptCommand } from "./prompt-commands.js";
6
6
  import { loadCustomCommands } from "./custom-commands.js";
7
7
  import { SettingsManager } from "./settings-manager.js";
8
8
  import { AuthStorage } from "./auth-storage.js";
9
+ import { MOONSHOT_OAUTH_KEY } from "@prestyj/core";
9
10
  import { getClaudeCliUserAgent } from "./claude-code-version.js";
10
11
  import { kimiCodingHeaders, isKimiCodingEndpoint } from "./oauth/kimi.js";
11
12
  import { SessionManager, NOLAN_TURN_CUSTOM_KIND, AUTOPILOT_MARKER_CUSTOM_KIND, APP_MARKER_CUSTOM_KIND, } from "./session-manager.js";
12
13
  import { ExtensionLoader } from "./extensions/loader.js";
13
- import { shouldCompact, compact, getCompactionReserveTokens } from "./compaction/compactor.js";
14
- import { getAuthStorageKeys, getContextWindow, getModel, MODELS } from "./model-registry.js";
14
+ import { shouldCompact, compact } from "./compaction/compactor.js";
15
+ import { getAuthStorageKeys, getContextWindow, getModel, getToolResultCharLimit, MODELS, } from "./model-registry.js";
15
16
  import { discoverSkills } from "./skills.js";
16
17
  import { ensureAppDirs } from "../config.js";
17
18
  import { buildSystemPrompt } from "../system-prompt.js";
@@ -22,18 +23,38 @@ import { MCPClientManager, getAllMcpServers } from "./mcp/index.js";
22
23
  import { DeferredToolCatalog } from "./mcp/deferred-catalog.js";
23
24
  import { createToolSearchTool } from "../tools/tool-search.js";
24
25
  import { log } from "./logger.js";
25
- import { setEstimatorModel } from "./compaction/token-estimator.js";
26
+ import { setEstimatorModel, calibrateEstimatorFromUsage } from "./compaction/token-estimator.js";
27
+ import { calculateActiveContextTokens } from "./compaction/active-context.js";
28
+ import { pruneStaleToolResults } from "./compaction/tool-result-pruner.js";
26
29
  import { discoverAgents } from "./agents.js";
27
- import { generateSessionTitle } from "../utils/session-title.js";
28
30
  import { enhancePrompt } from "../utils/prompt-enhancer.js";
29
31
  import { detectProjectStack } from "./language-detector.js";
30
- import { evaluateIdealReview, buildIdealReviewMessage, buildReviewCoverageMessage, detectTestDrift, ReviewCoverageTracker, } from "./ideal-review.js";
32
+ import { evaluateIdealReview, buildIdealReviewMessage, buildReviewCoverageMessage, withReviewCoverageRequirements, detectTestDrift, ReviewCoverageTracker, } from "./ideal-review.js";
31
33
  import { evaluateLoopBreak, buildLoopBreakMessage, ToolCallProgressTracker, detectTextRepetition, } from "./loop-breaker.js";
32
34
  import { buildRegroundingMessage } from "./regrounding.js";
33
35
  import { wrapSteeringText, STEERING_PREFIX } from "./steering.js";
36
+ import { findUserSessionPrompt, getUserSessionPrompt } from "./session-preview.js";
34
37
  import crypto from "node:crypto";
35
38
  import fs from "node:fs/promises";
36
39
  import path from "node:path";
40
+ // ── Tool-result policy ─────────────────────────────────────
41
+ /** Resolve the per-result cap passed to the agent loop for the active transport. */
42
+ export function resolveSessionToolResultCharLimit(model, provider, accountId) {
43
+ return (getToolResultCharLimit(model, { provider, accountId }) ??
44
+ Math.floor(getContextWindow(model, { provider, accountId }) * 3.5 * 0.3));
45
+ }
46
+ /**
47
+ * Aggregate budget for ALL tool results produced in one assistant turn.
48
+ * Individual results are already capped, but wide parallel fan-outs (GPT-5.6's
49
+ * signature behavior) were observed injecting 100k+ uncached tokens in a single
50
+ * turn. ~15% of the context window in chars (1 token ≈ 3.5 chars), floored at
51
+ * 100KB so small windows still fit two full-size reads, ceilinged at 240KB so
52
+ * 1M-context models don't waive the budget entirely.
53
+ */
54
+ export function resolveSessionTurnToolResultCharLimit(model, provider, accountId) {
55
+ const contextChars = getContextWindow(model, { provider, accountId }) * 3.5;
56
+ return Math.max(100_000, Math.min(Math.floor(contextChars * 0.15), 240_000));
57
+ }
37
58
  // ── Agent Session ──────────────────────────────────────────
38
59
  export class AgentSession {
39
60
  eventBus = new EventBus();
@@ -87,10 +108,16 @@ export class AgentSession {
87
108
  hookFileEditCounts = new Map();
88
109
  hookToolCalls = new Map();
89
110
  idealReviewPhase = "idle";
111
+ /** Runtime-only suppression while Nolan owns verification in autopilot mode. */
112
+ idealReviewSuppressed = false;
90
113
  reviewCoverage;
91
114
  loopBreakInjected = false;
92
115
  regroundingInjected = false;
93
116
  compactionOccurred = false;
117
+ lastCompactionCompacted = false;
118
+ compactionRetryAfter = 0;
119
+ /** Latest provider count, anchored to the assistant response it measured. */
120
+ providerContext = null;
94
121
  originalRequest = "";
95
122
  // Messages queued by the user while a run is in flight. Drained at the
96
123
  // mid-loop steering boundary (user steering wins over the hooks), mirroring
@@ -128,6 +155,10 @@ export class AgentSession {
128
155
  * model emits step-completion markers the UI's plan-progress widget reads. */
129
156
  approvedPlanPath;
130
157
  sessionId = "";
158
+ /** Stable identity shared by compaction and approved-plan checkpoint files. */
159
+ conversationId = "";
160
+ /** Original user-authored prompt, retained when internal messages replace history. */
161
+ sessionPreview = "";
131
162
  /** Runtime conversation identity for provider transport headers. Transient
132
163
  * children need one even though they intentionally have no persisted session. */
133
164
  transportSessionId = crypto.randomUUID();
@@ -633,6 +664,13 @@ export class AgentSession {
633
664
  }
634
665
  case "turn_end":
635
666
  this.hookStats.turns = event.turn;
667
+ for (let index = this.messages.length - 1; index >= 0; index--) {
668
+ const anchor = this.messages[index];
669
+ if (anchor?.role === "assistant") {
670
+ this.providerContext = { usage: { ...event.usage }, anchor };
671
+ break;
672
+ }
673
+ }
636
674
  await this.persistTurnMetric(event);
637
675
  break;
638
676
  }
@@ -700,7 +738,7 @@ export class AgentSession {
700
738
  const childCompletionFollowUp = buildSubAgentCompletionFollowUp(this.subAgentManager);
701
739
  if (childCompletionFollowUp)
702
740
  return childCompletionFollowUp;
703
- if (this.opts.selfCorrectionHooks === false)
741
+ if (this.opts.selfCorrectionHooks === false || this.idealReviewSuppressed)
704
742
  return null;
705
743
  if (this.idealReviewPhase === "reviewing") {
706
744
  const coverage = this.reviewCoverage.evidence();
@@ -745,7 +783,7 @@ export class AgentSession {
745
783
  lspMissing: lspEvidence.missing,
746
784
  });
747
785
  return [
748
- this.withReviewLspEvidence(buildIdealReviewMessage(decision.reasons, driftedFiles), lspEvidence),
786
+ this.withReviewLspEvidence(withReviewCoverageRequirements(buildIdealReviewMessage(decision.reasons, driftedFiles), coverage.missing), lspEvidence),
749
787
  ];
750
788
  }
751
789
  reviewLspEvidence(files) {
@@ -805,24 +843,45 @@ export class AgentSession {
805
843
  this.lastAccountId = creds.accountId;
806
844
  // Auto-compact if needed. This must happen after credential resolution so
807
845
  // OpenAI OAuth/Codex sessions use the Codex product context window instead
808
- // of the public API model window.
809
- if (this.settingsManager.get("autoCompact")) {
846
+ // of the public API model window. Failed/no-op attempts cool down across
847
+ // prompts; provider overflow recovery still bypasses this path entirely.
848
+ if (this.settingsManager.get("autoCompact") && Date.now() >= this.compactionRetryAfter) {
810
849
  const contextWindow = getContextWindow(this.model, {
811
850
  provider: this.provider,
812
851
  accountId: creds.accountId,
813
852
  });
814
853
  const threshold = this.settingsManager.get("compactThreshold");
815
- // Reserve headroom for this model's real output budget (e.g. GPT-5.5 over
816
- // Codex OAuth: 272K window but up to 128K max output) — without this the
817
- // default 16K reserve lets compaction skip until input alone is near the
818
- // window, then `input + max_tokens` exceeds it and the provider rejects
819
- // the turn outright with "exceeds the context window". Mirrors the TUI's
820
- // useContextCompaction hook.
821
- const reserveTokens = getCompactionReserveTokens(this.maxTokens);
822
- if (shouldCompact(this.messages, contextWindow, threshold, undefined, reserveTokens)) {
823
- await this.compact(creds);
824
- // Re-grounding hook keys off this — the context was just summarized.
825
- this.compactionOccurred = true;
854
+ let activeTokens;
855
+ if (this.providerContext) {
856
+ const anchorIndex = this.messages.lastIndexOf(this.providerContext.anchor);
857
+ if (anchorIndex >= 0) {
858
+ activeTokens = calculateActiveContextTokens(this.messages, {
859
+ usage: this.providerContext.usage,
860
+ pendingMessages: this.messages.slice(anchorIndex + 1),
861
+ });
862
+ }
863
+ else {
864
+ this.providerContext = null;
865
+ }
866
+ }
867
+ if (shouldCompact(this.messages, contextWindow, threshold, activeTokens)) {
868
+ try {
869
+ await this.compact(creds);
870
+ if (this.lastCompactionCompacted) {
871
+ // Re-grounding hook keys off this — the context was just summarized.
872
+ this.compactionOccurred = true;
873
+ this.compactionRetryAfter = 0;
874
+ }
875
+ else {
876
+ this.compactionRetryAfter = Date.now() + 30_000;
877
+ }
878
+ }
879
+ catch (error) {
880
+ this.compactionRetryAfter = Date.now() + 30_000;
881
+ if (isAbortError(error) || this.opts.signal?.aborted)
882
+ throw error;
883
+ log("WARN", "compaction", `Pre-run compaction failed; cooling down for 30s: ${error instanceof Error ? error.message : String(error)}`);
884
+ }
826
885
  }
827
886
  }
828
887
  const userAgent = this.provider === "anthropic" ? await getClaudeCliUserAgent() : undefined;
@@ -856,29 +915,99 @@ export class AgentSession {
856
915
  supportsImages: modelInfo?.supportsImages,
857
916
  supportsVideo: modelInfo?.supportsVideo,
858
917
  userAgent,
859
- // clearToolUses disabled — causes model to output unsolicited context summaries
860
- // Single tool result shouldn't exceed 30% of context window (in chars)
861
- maxToolResultChars: Math.floor(getContextWindow(this.model, { provider: this.provider, accountId }) * 3.5 * 0.3),
918
+ // Codex caps each tool output at 10K tokens. Other transports retain the
919
+ // generic 30%-of-context allowance used before this provider policy.
920
+ maxToolResultChars: resolveSessionToolResultCharLimit(this.model, this.provider, accountId),
921
+ // Aggregate per-turn budget across parallel tool results (fan-out guard).
922
+ maxTurnToolResultChars: resolveSessionTurnToolResultCharLimit(this.model, this.provider, accountId),
862
923
  // Self-correction hooks (same as the TUI): loop-break + re-grounding are
863
924
  // polled mid-loop; the ideal review is polled when the agent would stop.
864
925
  getSteeringMessages: () => this.getHookSteeringMessages(),
865
926
  getFollowUpMessages: () => this.getHookFollowUpMessages(),
866
- // Overflow recovery: the loop calls this with { force: true } when the
867
- // provider rejects a turn as too large (request_too_large / context
868
- // overflow). Force-compact the in-flight history and hand it back so the
869
- // loop retries with a smaller request, instead of surfacing the error.
870
- // Without this the desktop app (which drives the loop through
871
- // AgentSession, not the TUI's useContextCompaction hook) had NO auto
872
- // recovery on 413 — the error went straight to the user. The non-force
873
- // pre-call invocations pass through untouched: pre-turn compaction is
874
- // already handled above, so we only act on the overflow force path.
875
- // `loopMessages === this.messages` (prepareDynamicContext returns it by
876
- // reference) and the post-loop `this.messages = loopMessages` re-sync
877
- // keeps persistence correct after compact() swaps the array.
927
+ // Check authoritative provider usage before every model/tool step.
928
+ // Forced overflow recovery bypasses settings and cooldown; proactive
929
+ // checks honor both and estimate only messages unseen by the provider.
878
930
  transformContext: async (messages, transformOpts) => {
879
- if (!transformOpts?.force)
931
+ if (transformOpts.usage) {
932
+ const anchorIndex = messages.length - transformOpts.pendingMessages.length - 1;
933
+ const anchor = messages[anchorIndex];
934
+ if (anchor?.role === "assistant") {
935
+ this.providerContext = { usage: { ...transformOpts.usage }, anchor };
936
+ // Feed the authoritative usage back into the token estimator so
937
+ // char-based estimates track this session's real tokenizer.
938
+ calibrateEstimatorFromUsage(messages.slice(0, anchorIndex), transformOpts.usage);
939
+ }
940
+ }
941
+ const force = transformOpts.force === true;
942
+ if (!force) {
943
+ if (!this.settingsManager.get("autoCompact"))
944
+ return messages;
945
+ // Cheap stale-tool-output pruning before the expensive LLM
946
+ // compaction check. In-place mutation preserves anchors; drop the
947
+ // retained usage afterwards since it counted the pruned content.
948
+ const pruneResult = pruneStaleToolResults(messages);
949
+ if (pruneResult.pruned) {
950
+ this.providerContext = null;
951
+ log("INFO", "compaction", "Pruned stale tool outputs", {
952
+ prunedResults: String(pruneResult.prunedResults),
953
+ freedTokens: String(pruneResult.freedTokens),
954
+ });
955
+ }
956
+ if (Date.now() < this.compactionRetryAfter)
957
+ return messages;
958
+ // The turn's own usage also counted the pruned content — after a
959
+ // prune, fall back to estimating the (now smaller) history so the
960
+ // freed tokens actually defer the LLM compaction.
961
+ let usage = pruneResult.pruned ? undefined : transformOpts.usage;
962
+ let pendingMessages = transformOpts.pendingMessages;
963
+ if (!usage && this.providerContext) {
964
+ const anchorIndex = messages.lastIndexOf(this.providerContext.anchor);
965
+ if (anchorIndex >= 0) {
966
+ usage = this.providerContext.usage;
967
+ pendingMessages = messages.slice(anchorIndex + 1);
968
+ }
969
+ else {
970
+ this.providerContext = null;
971
+ }
972
+ }
973
+ const contextWindow = getContextWindow(this.model, {
974
+ provider: this.provider,
975
+ accountId,
976
+ });
977
+ const threshold = this.settingsManager.get("compactThreshold");
978
+ const activeTokens = calculateActiveContextTokens(messages, {
979
+ usage,
980
+ pendingMessages,
981
+ });
982
+ if (!shouldCompact(messages, contextWindow, threshold, activeTokens))
983
+ return messages;
984
+ }
985
+ // compact() operates on this.messages, while an earlier transform may
986
+ // have replaced the loop's in-flight array. Rebind before every attempt
987
+ // so the current tool results are included in the summary.
988
+ this.messages = messages;
989
+ try {
990
+ await this.compact({
991
+ accessToken: apiKey,
992
+ accountId,
993
+ projectId,
994
+ baseUrl: effectiveBaseUrl,
995
+ });
996
+ }
997
+ catch (error) {
998
+ this.messages = messages;
999
+ this.compactionRetryAfter = Date.now() + 30_000;
1000
+ if (force || isAbortError(error) || this.opts.signal?.aborted)
1001
+ throw error;
1002
+ log("WARN", "compaction", `In-flight compaction failed; cooling down for 30s: ${error instanceof Error ? error.message : String(error)}`);
880
1003
  return messages;
881
- await this.compact();
1004
+ }
1005
+ if (!this.lastCompactionCompacted) {
1006
+ this.messages = messages;
1007
+ this.compactionRetryAfter = Date.now() + 30_000;
1008
+ return messages;
1009
+ }
1010
+ this.compactionRetryAfter = 0;
882
1011
  this.compactionOccurred = true;
883
1012
  return this.messages;
884
1013
  },
@@ -888,6 +1017,19 @@ export class AgentSession {
888
1017
  this.eventBus.forwardAgentEvent(event);
889
1018
  }
890
1019
  };
1020
+ const clearInvalidStaticApiKey = async (error) => {
1021
+ if (!(error instanceof ProviderError) || error.statusCode !== 401)
1022
+ return false;
1023
+ if (!(await this.authStorage.isStaticApiKey(this.provider)))
1024
+ return false;
1025
+ // Clear whichever key actually resolved (the request may have used a
1026
+ // fallback key, not the model's first preference).
1027
+ const badKey = (await this.authStorage.pickStorageKey(this.currentAuthStorageKeys())) ??
1028
+ this.currentAuthStorageKeys()[0];
1029
+ log("WARN", "auth", `Got 401 for ${this.provider} (${badKey}) — API key is invalid or revoked`);
1030
+ await this.authStorage.clearCredentials(badKey);
1031
+ return true;
1032
+ };
891
1033
  try {
892
1034
  await runAgentLoop(creds.accessToken, creds.accountId, creds.projectId);
893
1035
  }
@@ -896,21 +1038,48 @@ export class AgentSession {
896
1038
  if (isAbortError(err) || this.opts.signal?.aborted) {
897
1039
  return;
898
1040
  }
899
- if (err instanceof ProviderError && err.statusCode === 401) {
1041
+ // Kimi OAuth plan ran out of usage (hard usage-limit stop, or an HTTP 402
1042
+ // billing stop). If the user ALSO configured a Moonshot API key, mark
1043
+ // the OAuth credential usage-exhausted (honoring the provider-stated
1044
+ // reset time when present) and retry this turn on the API key — OAuth
1045
+ // stays the preferred credential and resumes automatically once the mark
1046
+ // lapses. A generic 429 is deliberately excluded: it may be a transient
1047
+ // rate limit and must not silently switch the user to a billed API key.
1048
+ // Guarded on the Kimi managed endpoint actually being in use: if the API
1049
+ // key was already active, the same error means BOTH are out and must surface.
1050
+ if (this.provider === "moonshot" &&
1051
+ !this.baseUrl &&
1052
+ isKimiCodingEndpoint(creds.baseUrl) &&
1053
+ (isUsageLimitError(err) || (err instanceof ProviderError && err.statusCode === 402)) &&
1054
+ (await this.authStorage.hasCredentials("moonshot"))) {
1055
+ const resetsAt = err instanceof ProviderError ? err.resetsAt : undefined;
1056
+ await this.authStorage.markUsageExhausted(MOONSHOT_OAUTH_KEY, resetsAt);
1057
+ log("WARN", "auth", "Kimi OAuth usage limit reached — retrying this turn on the Moonshot API key", { resetsAt: resetsAt !== undefined ? String(resetsAt) : "unknown" });
1058
+ creds = await this.authStorage.resolveCredentials(this.provider, {
1059
+ storageKeys: this.currentAuthStorageKeys(),
1060
+ });
1061
+ this.lastAccountId = creds.accountId;
1062
+ // The runAgentLoop closure re-reads `creds`, so the retry picks up the
1063
+ // API key's baseUrl (api.moonshot.ai) and drops the Kimi coding headers.
1064
+ try {
1065
+ await runAgentLoop(creds.accessToken, creds.accountId, creds.projectId);
1066
+ }
1067
+ catch (fallbackErr) {
1068
+ // The fallback is inside this catch branch, so its errors do not pass
1069
+ // through the outer 401 handler. Clear a rejected API key explicitly
1070
+ // before surfacing the error and prompting the user to log in again.
1071
+ await clearInvalidStaticApiKey(fallbackErr);
1072
+ throw fallbackErr;
1073
+ }
1074
+ }
1075
+ else if (err instanceof ProviderError && err.statusCode === 401) {
900
1076
  // Static API-key providers (GLM, Moonshot API key, etc.) have no refresh
901
1077
  // mechanism — retrying with the same key is pointless. Clear the
902
1078
  // credential and surface the error so the user re-logins. Kimi OAuth
903
1079
  // (active for `moonshot` when present) is refreshable, so it falls
904
1080
  // through to the force-refresh path below.
905
- if (await this.authStorage.isStaticApiKey(this.provider)) {
906
- // Clear whichever key actually resolved (the request may have used
907
- // a fallback key, not the model's first preference).
908
- const badKey = (await this.authStorage.pickStorageKey(this.currentAuthStorageKeys())) ??
909
- this.currentAuthStorageKeys()[0];
910
- log("WARN", "auth", `Got 401 for ${this.provider} (${badKey}) — API key is invalid or revoked`);
911
- await this.authStorage.clearCredentials(badKey);
1081
+ if (await clearInvalidStaticApiKey(err))
912
1082
  throw err;
913
- }
914
1083
  log("INFO", "auth", "Got 401, force-refreshing token and retrying");
915
1084
  creds = await this.authStorage.resolveCredentials(this.provider, {
916
1085
  forceRefresh: true,
@@ -935,6 +1104,7 @@ export class AgentSession {
935
1104
  if (provider)
936
1105
  this.provider = provider;
937
1106
  this.model = model;
1107
+ this.providerContext = null;
938
1108
  // Keep host-provided option closures (notably chat delegation) aligned with
939
1109
  // the live selection after an in-session model switch.
940
1110
  this.opts.provider = this.provider;
@@ -1012,6 +1182,7 @@ export class AgentSession {
1012
1182
  }
1013
1183
  }
1014
1184
  async compact(existingCredentials) {
1185
+ this.lastCompactionCompacted = false;
1015
1186
  const creds = existingCredentials ??
1016
1187
  (await this.authStorage.resolveCredentials(this.provider, {
1017
1188
  storageKeys: this.currentAuthStorageKeys(),
@@ -1032,6 +1203,15 @@ export class AgentSession {
1032
1203
  signal: this.opts.signal,
1033
1204
  });
1034
1205
  this.messages = result.messages;
1206
+ this.lastCompactionCompacted = result.result.compacted;
1207
+ if (!result.result.compacted) {
1208
+ this.eventBus.emit("compaction_end", {
1209
+ originalCount: result.result.originalCount,
1210
+ newCount: result.result.newCount,
1211
+ });
1212
+ return;
1213
+ }
1214
+ this.providerContext = null;
1035
1215
  // Transient sessions (Nolan chat/autopilot, subagent spawns) must NEVER touch
1036
1216
  // the session store: without this guard, the first auto-compaction called
1037
1217
  // sessionManager.create() and assigned a real sessionPath, silently turning
@@ -1044,8 +1224,12 @@ export class AgentSession {
1044
1224
  else {
1045
1225
  // Persist compacted messages to a new session file so `ezcoder continue`
1046
1226
  // picks up the compacted state instead of the full original history.
1047
- const session = await this.sessionManager.create(this.cwd, this.provider, this.model);
1227
+ const session = await this.sessionManager.create(this.cwd, this.provider, this.model, {
1228
+ conversationId: this.conversationId || undefined,
1229
+ preview: this.sessionPreview || undefined,
1230
+ });
1048
1231
  this.sessionId = session.id;
1232
+ this.conversationId = session.header.conversationId ?? session.id;
1049
1233
  this.sessionPath = session.path;
1050
1234
  await this.subAgentManager?.rebindParentSession(this.sessionId);
1051
1235
  // Write compacted messages (skip system — it's rebuilt on load)
@@ -1072,7 +1256,13 @@ export class AgentSession {
1072
1256
  newCount: result.result.newCount,
1073
1257
  });
1074
1258
  }
1075
- async newSession() {
1259
+ async newSession(preserveConversation = false) {
1260
+ // Approved-plan execution is a clean checkpoint of the same conversation;
1261
+ // explicit new sessions reset the conversation identity.
1262
+ if (!preserveConversation) {
1263
+ this.conversationId = "";
1264
+ this.sessionPreview = "";
1265
+ }
1076
1266
  // A fresh session drops any in-flight plan state so its prompt is clean.
1077
1267
  this.planModeRef.current = false;
1078
1268
  this.approvedPlanPath = undefined;
@@ -1098,6 +1288,8 @@ export class AgentSession {
1098
1288
  // polluted the project's session list.
1099
1289
  if (this.opts.transient) {
1100
1290
  this.sessionId = "";
1291
+ this.conversationId = "";
1292
+ this.sessionPreview = "";
1101
1293
  this.sessionPath = "";
1102
1294
  this.lastPersistedIndex = this.messages.length;
1103
1295
  }
@@ -1170,6 +1362,18 @@ export class AgentSession {
1170
1362
  getPlanMode() {
1171
1363
  return this.planModeRef.current;
1172
1364
  }
1365
+ /**
1366
+ * Suppress only the pre-final Ideal self-review for this live session.
1367
+ * Autopilot uses this while Nolan independently owns verification; loop-break
1368
+ * and post-compaction re-grounding remain active.
1369
+ */
1370
+ setIdealReviewSuppressed(suppressed) {
1371
+ this.idealReviewSuppressed = suppressed;
1372
+ if (suppressed) {
1373
+ this.idealReviewPhase = "idle";
1374
+ this.reviewCoverage.reset();
1375
+ }
1376
+ }
1173
1377
  /** Queue a user message (optionally with attachments) to be injected mid-run
1174
1378
  * as steering. Returns the new queue length. No-op semantics are the caller's
1175
1379
  * concern. */
@@ -1467,46 +1671,12 @@ export class AgentSession {
1467
1671
  await this.sessionManager.appendEntry(this.sessionPath, entry);
1468
1672
  }
1469
1673
  }
1470
- /**
1471
- * Generate a short LLM session title from the conversation so far (first user
1472
- * message + first assistant reply). Best-effort; returns null on failure or
1473
- * when there's no user message yet. Uses the cheapest model for the provider.
1474
- */
1475
- async generateTitle() {
1476
- const extractText = (content) => typeof content === "string"
1477
- ? content
1478
- : content
1479
- .map((c) => c.type === "text" && "text" in c && typeof c.text === "string" ? c.text : "")
1480
- .join(" ");
1481
- const userMsg = this.messages.find((m) => m.role === "user");
1482
- const assistantMsg = this.messages.find((m) => m.role === "assistant");
1483
- const userText = userMsg ? extractText(userMsg.content) : "";
1484
- if (!userText.trim())
1485
- return null;
1486
- try {
1487
- const creds = await this.authStorage.resolveCredentials(this.provider, {
1488
- storageKeys: this.currentAuthStorageKeys(),
1489
- });
1490
- const title = await generateSessionTitle({
1491
- provider: this.provider,
1492
- userMessage: userText,
1493
- assistantPreview: assistantMsg ? extractText(assistantMsg.content).slice(0, 200) : "",
1494
- apiKey: creds.accessToken,
1495
- baseUrl: this.baseUrl ?? creds.baseUrl,
1496
- accountId: creds.accountId,
1497
- });
1498
- return title || null;
1499
- }
1500
- catch {
1501
- return null;
1502
- }
1503
- }
1504
1674
  /**
1505
1675
  * Rewrite a draft prompt into a tighter, terminology-correct version using
1506
1676
  * the ACTIVE provider/model. A stateless one-off LLM call (no agent loop, no
1507
1677
  * tools, no session mutation) — safe to run even mid-run. Returns the plain
1508
1678
  * enhanced text plus typed segments marking each corrected term. Errors throw
1509
- * so the caller can surface them (unlike best-effort title generation).
1679
+ * so the caller can surface them.
1510
1680
  */
1511
1681
  async enhancePrompt(text) {
1512
1682
  if (!text.trim())
@@ -1604,8 +1774,12 @@ export class AgentSession {
1604
1774
  }
1605
1775
  // ── Private ────────────────────────────────────────────
1606
1776
  async createNewSession() {
1607
- const session = await this.sessionManager.create(this.cwd, this.provider, this.model);
1777
+ const session = await this.sessionManager.create(this.cwd, this.provider, this.model, {
1778
+ conversationId: this.conversationId || undefined,
1779
+ preview: this.sessionPreview || undefined,
1780
+ });
1608
1781
  this.sessionId = session.id;
1782
+ this.conversationId = session.header.conversationId ?? session.id;
1609
1783
  this.sessionPath = session.path;
1610
1784
  this.lastPersistedIndex = this.messages.length;
1611
1785
  }
@@ -1613,6 +1787,15 @@ export class AgentSession {
1613
1787
  const loaded = await this.sessionManager.load(sessionPath);
1614
1788
  // Use the leaf from the header to walk the correct branch
1615
1789
  const loadedMessages = this.sessionManager.getMessages(loaded.entries, loaded.header.leafId);
1790
+ this.conversationId = loaded.header.conversationId ?? loaded.header.id;
1791
+ const legacyLabel = [...loaded.entries]
1792
+ .reverse()
1793
+ .find((entry) => entry.type === "label")
1794
+ ?.label.replace(/\s+/g, " ")
1795
+ .trim()
1796
+ .slice(0, 80);
1797
+ this.sessionPreview =
1798
+ legacyLabel || loaded.header.preview || findUserSessionPrompt(loadedMessages);
1616
1799
  // Restore Nolan's advisory turns (custom entries, not on the message branch) so
1617
1800
  // they reappear in the transcript and survive into the continuation file.
1618
1801
  this.nolanTurns = this.sessionManager.getNolanTurns(loaded.entries);
@@ -1639,7 +1822,15 @@ export class AgentSession {
1639
1822
  provider: this.provider,
1640
1823
  accountId: creds.accountId,
1641
1824
  });
1642
- if (shouldCompact(this.messages, contextWindow, 0.8, undefined, getCompactionReserveTokens(this.maxTokens))) {
1825
+ const needsLoadCompaction = this.settingsManager.get("autoCompact") &&
1826
+ shouldCompact(this.messages, contextWindow, this.settingsManager.get("compactThreshold"));
1827
+ if (needsLoadCompaction && this.opts.deferLoadCompaction) {
1828
+ // Host readiness is gated on initialize() — don't block it on a summary
1829
+ // LLM call (up to 30s). runLoop()'s pre-run auto-compaction picks this
1830
+ // up on the first prompt and emits compaction_start/_end for the UI.
1831
+ log("INFO", "session", "Restored session exceeds context — deferring compaction to first prompt");
1832
+ }
1833
+ else if (needsLoadCompaction) {
1643
1834
  await this.subAgentManager?.hydrate(loaded.header.id);
1644
1835
  log("INFO", "session", `Restored session exceeds context — auto-compacting`);
1645
1836
  const compacted = await compact(this.messages, {
@@ -1661,8 +1852,12 @@ export class AgentSession {
1661
1852
  // what's in memory — fork a fresh session file for the compacted state
1662
1853
  // (mirrors compact()'s own persistence) so `ezcoder continue` picks up
1663
1854
  // the summary instead of the full original transcript.
1664
- const session = await this.sessionManager.create(this.cwd, this.provider, this.model);
1855
+ const session = await this.sessionManager.create(this.cwd, this.provider, this.model, {
1856
+ conversationId: this.conversationId || undefined,
1857
+ preview: this.sessionPreview || undefined,
1858
+ });
1665
1859
  this.sessionId = session.id;
1860
+ this.conversationId = session.header.conversationId ?? session.id;
1666
1861
  this.sessionPath = session.path;
1667
1862
  await this.subAgentManager?.rebindParentSession(this.sessionId);
1668
1863
  this.currentLeafId = null;
@@ -1698,6 +1893,9 @@ export class AgentSession {
1698
1893
  return this.messages;
1699
1894
  }
1700
1895
  async persistMessage(message) {
1896
+ if (!this.sessionPreview && message.role === "user") {
1897
+ this.sessionPreview = getUserSessionPrompt(message.content) ?? "";
1898
+ }
1701
1899
  // Transient sessions (subagent spawns) have no session file — skip.
1702
1900
  if (!this.sessionPath)
1703
1901
  return;