@prestyj/cli 5.28.1 → 5.29.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/assets/motion/bin/contact-sheet.mjs +9 -3
- package/assets/motion/bin/cues.mjs +337 -0
- package/assets/motion/bin/library.mjs +53 -10
- package/assets/motion/bin/motion-blur.mjs +943 -0
- package/assets/motion/bin/motion-check.mjs +354 -3
- package/assets/motion/bin/music-fit.mjs +436 -0
- package/assets/motion/bin/pdf-extract.mjs +22 -10
- package/assets/motion/bin/reference-study.mjs +346 -0
- package/assets/motion/bin/score-synth.mjs +1052 -93
- package/assets/motion/library/README.md +50 -14
- package/assets/motion/library/kit/moves.js +1981 -0
- package/assets/motion/library/library.json +232 -0
- package/assets/motion/library/pieces/camera-rig/meta.json +13 -0
- package/assets/motion/library/pieces/camera-rig/piece.html +153 -0
- package/assets/motion/library/pieces/camera-rig/preview.jpg +0 -0
- package/assets/motion/library/pieces/chain-knock/meta.json +13 -0
- package/assets/motion/library/pieces/chain-knock/piece.html +195 -0
- package/assets/motion/library/pieces/chain-knock/preview.jpg +0 -0
- package/assets/motion/library/pieces/gather-to-logo/meta.json +13 -0
- package/assets/motion/library/pieces/gather-to-logo/piece.html +159 -0
- package/assets/motion/library/pieces/gather-to-logo/preview.jpg +0 -0
- package/assets/motion/library/pieces/morph-carry/meta.json +13 -0
- package/assets/motion/library/pieces/morph-carry/piece.html +173 -0
- package/assets/motion/library/pieces/morph-carry/preview.jpg +0 -0
- package/assets/motion/library/pieces/one-shape-journey/meta.json +13 -0
- package/assets/motion/library/pieces/one-shape-journey/piece.html +195 -0
- package/assets/motion/library/pieces/one-shape-journey/preview.jpg +0 -0
- package/assets/motion/library/pieces/open-from-subject/meta.json +13 -0
- package/assets/motion/library/pieces/open-from-subject/piece.html +168 -0
- package/assets/motion/library/pieces/open-from-subject/preview.jpg +0 -0
- package/assets/motion/library/pieces/request-to-result/meta.json +13 -0
- package/assets/motion/library/pieces/request-to-result/piece.html +212 -0
- package/assets/motion/library/pieces/request-to-result/preview.jpg +0 -0
- package/assets/motion/library/pieces/scale-dive/meta.json +13 -0
- package/assets/motion/library/pieces/scale-dive/piece.html +321 -0
- package/assets/motion/library/pieces/scale-dive/preview.jpg +0 -0
- package/assets/motion/library/pieces/screen-replica-steps/meta.json +13 -0
- package/assets/motion/library/pieces/screen-replica-steps/piece.html +366 -0
- package/assets/motion/library/pieces/screen-replica-steps/preview.jpg +0 -0
- package/assets/motion/library/pieces/zoom-into-card/meta.json +13 -0
- package/assets/motion/library/pieces/zoom-into-card/piece.html +179 -0
- package/assets/motion/library/pieces/zoom-into-card/preview.jpg +0 -0
- package/assets/motion/library/sheets/diagram.jpg +0 -0
- package/assets/motion/library/sheets/frame.jpg +0 -0
- package/assets/motion/library/sheets/transition.jpg +0 -0
- package/assets/motion/library/sheets/ui.jpg +0 -0
- package/assets/motion/references/build-sheet.md +206 -0
- package/assets/motion/references/runtime/determinism-rules.md +1 -1
- package/assets/motion/references/runtime/gsap-easing-and-stagger.md +29 -29
- package/assets/motion/references/runtime/inputs-and-assets.md +7 -12
- package/assets/motion/references/runtime/lint-validate-inspect.md +3 -3
- package/assets/motion/references/runtime/minimal-composition.md +1 -1
- package/assets/motion/references/runtime/preview-render.md +3 -3
- package/assets/motion/skills/app-walkthrough/SKILL.md +66 -0
- package/assets/motion/skills/before-after/SKILL.md +53 -0
- package/assets/motion/skills/brand-kit/SKILL.md +3 -3
- package/assets/motion/skills/dev-tool-video/SKILL.md +57 -0
- package/assets/motion/skills/launch-video/SKILL.md +62 -0
- package/assets/motion/skills/match-reference/SKILL.md +58 -0
- package/assets/motion/skills/motion/SKILL.md +78 -85
- package/assets/motion/skills/source-ingest/SKILL.md +17 -6
- package/assets/motion/skills/website-video/SKILL.md +59 -0
- package/assets/skills/bulletproof/SKILL.md +36 -11
- package/assets/skills/bulletproof/references/agent-surface.md +19 -9
- package/assets/skills/bulletproof/references/audit-protocol.md +20 -5
- package/assets/skills/bulletproof/references/platform-playbooks.md +5 -4
- package/assets/skills/bulletproof/references/provenance.md +26 -1
- package/assets/skills/bulletproof/references/secure-defaults.md +6 -5
- package/assets/skills/bulletproof/references/supply-chain.md +21 -17
- package/assets/skills/bulletproof/references/threat-landscape.md +28 -26
- package/assets/skills/bulletproof/references/verification.md +2 -0
- package/assets/skills/clarify/SKILL.md +25 -16
- package/assets/skills/code-review/SKILL.md +71 -13
- package/assets/skills/code-review/references/agent-diffs.md +27 -0
- package/assets/skills/code-review/references/tests.md +19 -0
- package/assets/skills/compliance-guard/SKILL.md +20 -5
- package/assets/skills/compliance-guard/references/artifacts.md +1 -1
- package/assets/skills/compliance-guard/references/eu-uk.md +16 -16
- package/assets/skills/compliance-guard/references/lawsuit-vectors.md +5 -5
- package/assets/skills/compliance-guard/references/provenance.md +41 -2
- package/assets/skills/compliance-guard/references/sector-gates.md +3 -3
- package/assets/skills/compliance-guard/references/security-baseline.md +2 -2
- package/assets/skills/compliance-guard/references/trigger-map.md +5 -5
- package/assets/skills/compliance-guard/references/us.md +27 -21
- package/assets/skills/durable/SKILL.md +87 -79
- package/assets/skills/durable/references/agent-db-safety.md +69 -0
- package/assets/skills/durable/references/backups-and-runtime.md +19 -12
- package/assets/skills/durable/references/migrations-and-schema.md +13 -6
- package/assets/skills/evidence-led-ui/SKILL.md +69 -127
- package/assets/skills/evidence-led-ui/references/anti-defaults.md +107 -208
- package/assets/skills/evidence-led-ui/references/direction.md +124 -0
- package/assets/skills/evidence-led-ui/references/production-contract.md +8 -0
- package/assets/skills/evidence-led-ui/references/provenance.md +24 -1
- package/assets/skills/lean/SKILL.md +90 -71
- package/assets/skills/lean/references/memory-and-processes.md +3 -2
- package/assets/skills/lean/references/playbooks.md +37 -12
- package/assets/skills/refactoring/SKILL.md +24 -3
- package/assets/skills/refactoring/references/agent-pitfalls.md +4 -1
- package/assets/skills/refactoring/references/legacy.md +21 -0
- package/assets/skills/root-cause/SKILL.md +20 -10
- package/assets/skills/shared-language/SKILL.md +16 -14
- package/assets/skills/tdd/SKILL.md +27 -15
- package/dist/app-sidecar.js +203 -47
- package/dist/app-sidecar.js.map +1 -1
- package/dist/cli.js +17 -26
- package/dist/cli.js.map +1 -1
- package/dist/core/acceptance-checks.d.ts +48 -0
- package/dist/core/acceptance-checks.js +144 -0
- package/dist/core/acceptance-checks.js.map +1 -0
- package/dist/core/agent-session.d.ts +106 -72
- package/dist/core/agent-session.js +539 -399
- package/dist/core/agent-session.js.map +1 -1
- package/dist/core/agents.d.ts +6 -5
- package/dist/core/agents.js.map +1 -1
- package/dist/core/ask-user.d.ts +90 -8
- package/dist/core/ask-user.js +124 -13
- package/dist/core/ask-user.js.map +1 -1
- package/dist/core/bundled-agents.js +1 -3
- package/dist/core/bundled-agents.js.map +1 -1
- package/dist/core/cache-diagnostics.d.ts +68 -0
- package/dist/core/cache-diagnostics.js +196 -0
- package/dist/core/cache-diagnostics.js.map +1 -0
- package/dist/core/cache-expiry.d.ts +87 -0
- package/dist/core/cache-expiry.js +111 -0
- package/dist/core/cache-expiry.js.map +1 -0
- package/dist/core/compaction/compactor.js +78 -48
- package/dist/core/compaction/compactor.js.map +1 -1
- package/dist/core/compaction/plan-step-policy.d.ts +46 -0
- package/dist/core/compaction/plan-step-policy.js +57 -0
- package/dist/core/compaction/plan-step-policy.js.map +1 -0
- package/dist/core/destructive-git-guard.d.ts +90 -0
- package/dist/core/destructive-git-guard.js +871 -0
- package/dist/core/destructive-git-guard.js.map +1 -0
- package/dist/core/event-bus.d.ts +3 -0
- package/dist/core/event-bus.js +5 -0
- package/dist/core/event-bus.js.map +1 -1
- package/dist/core/injection-detect.d.ts +38 -0
- package/dist/core/injection-detect.js +232 -0
- package/dist/core/injection-detect.js.map +1 -0
- package/dist/core/keep-awake.d.ts +88 -0
- package/dist/core/keep-awake.js +251 -0
- package/dist/core/keep-awake.js.map +1 -0
- package/dist/core/mcp/client.d.ts +72 -0
- package/dist/core/mcp/client.js +264 -41
- package/dist/core/mcp/client.js.map +1 -1
- package/dist/core/mcp/content.js +6 -2
- package/dist/core/mcp/content.js.map +1 -1
- package/dist/core/mcp/store.d.ts +6 -1
- package/dist/core/mcp/store.js +12 -1
- package/dist/core/mcp/store.js.map +1 -1
- package/dist/core/mcp/types.d.ts +18 -0
- package/dist/core/model-unavailable.d.ts +14 -0
- package/dist/core/model-unavailable.js +23 -0
- package/dist/core/model-unavailable.js.map +1 -0
- package/dist/core/node-debugger.d.ts +148 -0
- package/dist/core/node-debugger.js +642 -0
- package/dist/core/node-debugger.js.map +1 -0
- package/dist/core/package-threats.d.ts +18 -0
- package/dist/core/package-threats.js +168 -0
- package/dist/core/package-threats.js.map +1 -0
- package/dist/core/persistent-shell.d.ts +58 -6
- package/dist/core/persistent-shell.js +331 -49
- package/dist/core/persistent-shell.js.map +1 -1
- package/dist/core/process-manager.d.ts +14 -0
- package/dist/core/process-manager.js +61 -0
- package/dist/core/process-manager.js.map +1 -1
- package/dist/core/progress/git-xp.js +8 -14
- package/dist/core/progress/git-xp.js.map +1 -1
- package/dist/core/session-history.d.ts +12 -0
- package/dist/core/session-history.js +27 -0
- package/dist/core/session-history.js.map +1 -1
- package/dist/core/session-manager.d.ts +13 -1
- package/dist/core/session-manager.js +38 -18
- package/dist/core/session-manager.js.map +1 -1
- package/dist/core/session-summary-index.d.ts +37 -0
- package/dist/core/session-summary-index.js +172 -0
- package/dist/core/session-summary-index.js.map +1 -0
- package/dist/core/settings-manager.d.ts +2 -0
- package/dist/core/settings-manager.js +10 -0
- package/dist/core/settings-manager.js.map +1 -1
- package/dist/core/shell-threats-popular-packages.d.ts +11 -0
- package/dist/core/shell-threats-popular-packages.js +675 -0
- package/dist/core/shell-threats-popular-packages.js.map +1 -0
- package/dist/core/shell-threats.d.ts +8 -0
- package/dist/core/shell-threats.js +186 -0
- package/dist/core/shell-threats.js.map +1 -0
- package/dist/core/skills.js +3 -1
- package/dist/core/skills.js.map +1 -1
- package/dist/core/stream-rules.d.ts +30 -0
- package/dist/core/stream-rules.js +151 -0
- package/dist/core/stream-rules.js.map +1 -0
- package/dist/core/subagent-manager.d.ts +20 -5
- package/dist/core/subagent-manager.js +22 -8
- package/dist/core/subagent-manager.js.map +1 -1
- package/dist/core/subagent-receipt.d.ts +54 -0
- package/dist/core/subagent-receipt.js +276 -0
- package/dist/core/subagent-receipt.js.map +1 -0
- package/dist/core/subagent-turn-record.d.ts +2 -0
- package/dist/core/subagent-turn-record.js.map +1 -1
- package/dist/core/test-impact.d.ts +73 -0
- package/dist/core/test-impact.js +467 -0
- package/dist/core/test-impact.js.map +1 -0
- package/dist/core/thinking-level.d.ts +1 -1
- package/dist/core/thinking-level.js +1 -1
- package/dist/core/thinking-level.js.map +1 -1
- package/dist/core/verification-gate.d.ts +2 -0
- package/dist/core/verification-gate.js +4 -0
- package/dist/core/verification-gate.js.map +1 -1
- package/dist/core/verification-snapshot.js +3 -5
- package/dist/core/verification-snapshot.js.map +1 -1
- package/dist/core/workspace-guard.d.ts +18 -7
- package/dist/core/workspace-guard.js +227 -60
- package/dist/core/workspace-guard.js.map +1 -1
- package/dist/interactive.js +2 -1
- package/dist/interactive.js.map +1 -1
- package/dist/modes/subagent-worker-mode.js +34 -6
- package/dist/modes/subagent-worker-mode.js.map +1 -1
- package/dist/motion-agent/motion-agent.d.ts +6 -2
- package/dist/motion-agent/motion-agent.js +5 -7
- package/dist/motion-agent/motion-agent.js.map +1 -1
- package/dist/motion-agent/motion-prompt.d.ts +1 -1
- package/dist/motion-agent/motion-prompt.js +14 -19
- package/dist/motion-agent/motion-prompt.js.map +1 -1
- package/dist/motion-agent/motion-review.d.ts +10 -3
- package/dist/motion-agent/motion-review.js +15 -7
- package/dist/motion-agent/motion-review.js.map +1 -1
- package/dist/motion-agent/motion-studio-context.js +1 -1
- package/dist/motion-agent/motion-studio-context.js.map +1 -1
- package/dist/system-prompt.js +3 -1
- package/dist/system-prompt.js.map +1 -1
- package/dist/test-support/keep-alive.d.ts +14 -0
- package/dist/test-support/keep-alive.js +17 -0
- package/dist/test-support/keep-alive.js.map +1 -0
- package/dist/tools/ask-user.js +3 -3
- package/dist/tools/ask-user.js.map +1 -1
- package/dist/tools/bash-read-evidence.d.ts +10 -0
- package/dist/tools/bash-read-evidence.js +133 -0
- package/dist/tools/bash-read-evidence.js.map +1 -0
- package/dist/tools/bash.d.ts +10 -1
- package/dist/tools/bash.js +115 -7
- package/dist/tools/bash.js.map +1 -1
- package/dist/tools/debug.d.ts +54 -0
- package/dist/tools/debug.js +233 -0
- package/dist/tools/debug.js.map +1 -0
- package/dist/tools/edit.js +12 -4
- package/dist/tools/edit.js.map +1 -1
- package/dist/tools/goals.d.ts +1 -1
- package/dist/tools/index.d.ts +19 -2
- package/dist/tools/index.js +47 -7
- package/dist/tools/index.js.map +1 -1
- package/dist/tools/prompt-hints.js +2 -0
- package/dist/tools/prompt-hints.js.map +1 -1
- package/dist/tools/read-tracker.d.ts +5 -0
- package/dist/tools/read-tracker.js +19 -9
- package/dist/tools/read-tracker.js.map +1 -1
- package/dist/tools/read.js +3 -2
- package/dist/tools/read.js.map +1 -1
- package/dist/tools/skill.js +5 -0
- package/dist/tools/skill.js.map +1 -1
- package/dist/tools/subagent-control.js +44 -8
- package/dist/tools/subagent-control.js.map +1 -1
- package/dist/tools/subagent-shared.d.ts +29 -8
- package/dist/tools/subagent-shared.js +45 -14
- package/dist/tools/subagent-shared.js.map +1 -1
- package/dist/tools/subagent.d.ts +8 -2
- package/dist/tools/subagent.js +28 -10
- package/dist/tools/subagent.js.map +1 -1
- package/dist/tools/task-output.js +3 -2
- package/dist/tools/task-output.js.map +1 -1
- package/dist/tools/task-send.d.ts +1 -1
- package/dist/tools/task-send.js +15 -1
- package/dist/tools/task-send.js.map +1 -1
- package/dist/tools/tool-tiers.d.ts +2 -2
- package/dist/tools/tool-tiers.js +3 -2
- package/dist/tools/tool-tiers.js.map +1 -1
- package/dist/tools/truncate.d.ts +21 -0
- package/dist/tools/truncate.js +187 -0
- package/dist/tools/truncate.js.map +1 -1
- package/dist/tools/ui-adopt.js +2 -0
- package/dist/tools/ui-adopt.js.map +1 -1
- package/dist/ui/App.d.ts +0 -4
- package/dist/ui/App.js +5 -28
- package/dist/ui/App.js.map +1 -1
- package/dist/ui/components/ActivityIndicator.js +1 -0
- package/dist/ui/components/ActivityIndicator.js.map +1 -1
- package/dist/ui/hooks/useAgentLoop.d.ts +1 -8
- package/dist/ui/hooks/useAgentLoop.js +1 -119
- package/dist/ui/hooks/useAgentLoop.js.map +1 -1
- package/dist/ui/render.d.ts +0 -4
- package/dist/ui/render.js +0 -2
- package/dist/ui/render.js.map +1 -1
- package/dist/utils/git.d.ts +77 -0
- package/dist/utils/git.js +285 -21
- package/dist/utils/git.js.map +1 -1
- package/dist/utils/github-ci.js +2 -1
- package/dist/utils/github-ci.js.map +1 -1
- package/dist/utils/github.js +11 -9
- package/dist/utils/github.js.map +1 -1
- package/dist/utils/image.d.ts +14 -0
- package/dist/utils/image.js +16 -0
- package/dist/utils/image.js.map +1 -1
- package/dist/utils/process.d.ts +20 -0
- package/dist/utils/process.js +98 -0
- package/dist/utils/process.js.map +1 -1
- package/package.json +5 -5
- package/assets/motion/references/motion-language.md +0 -128
- package/assets/motion/skills/video-qa/SKILL.md +0 -89
- package/dist/core/ideal-review-subagent.d.ts +0 -56
- package/dist/core/ideal-review-subagent.js +0 -112
- package/dist/core/ideal-review-subagent.js.map +0 -1
- package/dist/core/ideal-review.d.ts +0 -82
- package/dist/core/ideal-review.js +0 -242
- package/dist/core/ideal-review.js.map +0 -1
- package/dist/motion-agent/motion-check-tool.d.ts +0 -35
- package/dist/motion-agent/motion-check-tool.js +0 -514
- package/dist/motion-agent/motion-check-tool.js.map +0 -1
|
@@ -1,6 +1,7 @@
|
|
|
1
|
-
import { agentLoop, isAbortError, isUsageLimitError, } from "@prestyj/agent";
|
|
1
|
+
import { agentLoop, isAbortError, isUsageLimitError, repairToolPairingAdjacent, } from "@prestyj/agent";
|
|
2
2
|
import { ProviderError, stream, } from "@prestyj/ai";
|
|
3
3
|
import { EventBus } from "./event-bus.js";
|
|
4
|
+
import { flagUntrustedToolResult } from "./injection-detect.js";
|
|
4
5
|
import { COMPLETION_REVIEW_STATE_KIND, } from "./completion-review.js";
|
|
5
6
|
import { SlashCommandRegistry, createBuiltinCommands, } from "./slash-commands.js";
|
|
6
7
|
import { PROMPT_COMMANDS, getPromptCommand } from "./prompt-commands.js";
|
|
@@ -23,8 +24,10 @@ import { ensureAppDirs } from "../config.js";
|
|
|
23
24
|
import { buildSubAgentSystemPrompt, buildSystemPrompt, } from "../system-prompt.js";
|
|
24
25
|
import { createTools, createWebSearchTool, } from "../tools/index.js";
|
|
25
26
|
import { partitionToolsByTier } from "../tools/tool-tiers.js";
|
|
27
|
+
import { formatImpactForVerification } from "./test-impact.js";
|
|
28
|
+
import { autoBackgroundedId } from "../tools/bash.js";
|
|
26
29
|
import { buildProcessCompletionFollowUp } from "./process-gate.js";
|
|
27
|
-
import { buildSubAgentCompletionFollowUp
|
|
30
|
+
import { buildSubAgentCompletionFollowUp } from "./subagent-manager.js";
|
|
28
31
|
import { applyAsyncSubagentPolicy } from "./subagent-policy.js";
|
|
29
32
|
import { z } from "zod";
|
|
30
33
|
import { MCPClientManager, getAllMcpServers } from "./mcp/index.js";
|
|
@@ -36,27 +39,31 @@ import { createToolSearchTool } from "../tools/tool-search.js";
|
|
|
36
39
|
import { createSessionStatsTool } from "../tools/session-stats.js";
|
|
37
40
|
import { createDiagnoseCommand, isInternalDiagnosticsEnabled, SessionDiagnosticsRecorder, } from "./internal-diagnostics.js";
|
|
38
41
|
import { log } from "./logger.js";
|
|
39
|
-
import {
|
|
42
|
+
import { CacheDiagnostics } from "./cache-diagnostics.js";
|
|
43
|
+
import { assessCacheExpiry, resolveCacheTtl } from "./cache-expiry.js";
|
|
44
|
+
import { setEstimatorModel, calibrateEstimatorFromUsage, estimateConversationTokens, } from "./compaction/token-estimator.js";
|
|
40
45
|
import { calculateActiveContextTokens } from "./compaction/active-context.js";
|
|
41
46
|
import { resolveCompactionPolicy } from "./compaction/policy.js";
|
|
47
|
+
import { decidePlanStepCompaction, DEFAULT_CACHE_WRITE_READ_RATIO, PLAN_STEP_KEEP_TOKENS, } from "./compaction/plan-step-policy.js";
|
|
48
|
+
import { extractPlanSteps, findCompletedMarkers } from "../utils/plan-steps.js";
|
|
49
|
+
import { readFileSync } from "node:fs";
|
|
42
50
|
import { clampThinkingForPlanMode } from "./thinking-level.js";
|
|
43
51
|
import { pruneStaleToolResults } from "./compaction/tool-result-pruner.js";
|
|
44
52
|
import { discoverAgents } from "./agents.js";
|
|
45
53
|
import { enhancePrompt } from "../utils/prompt-enhancer.js";
|
|
46
54
|
import { detectLanguages, detectProjectStack } from "./language-detector.js";
|
|
47
|
-
import { evaluateIdealReview, buildIdealReviewMessage, buildReviewCoverageEscalationMessage, buildReviewCoverageMessage, MAX_REVIEW_COVERAGE_INJECTIONS, withReviewCoverageRequirements, detectTestDrift, ReviewCoverageTracker, } from "./ideal-review.js";
|
|
48
55
|
import { evaluateLoopBreak, buildLoopBreakMessage, CycleDetector, ToolCallProgressTracker, detectTextRepetition, } from "./loop-breaker.js";
|
|
49
56
|
import { buildRegroundingMessage, requestTextForRegrounding } from "./regrounding.js";
|
|
50
57
|
import { buildSemanticLoopJudgePrompt, buildSemanticLoopMessage, MAX_SEMANTIC_LOOP_CALLS, parseSemanticLoopVerdict, shouldRunSemanticLoopCheck, SEMANTIC_LOOP_JUDGE_TIMEOUT_MS, withJudgeTimeout, } from "./semantic-loop-check.js";
|
|
51
|
-
import { buildIndependentReviewMessage, buildReviewerTask, INDEPENDENT_REVIEW_SCORE_THRESHOLD, parseReviewerFindings, REVIEWER_TOOLS, REVIEWER_TURN_TIMEOUT_MS, REVIEWER_WAIT_MS, } from "./ideal-review-subagent.js";
|
|
52
58
|
import { buildEnvDeltaMessage } from "./env-delta.js";
|
|
53
59
|
import { wrapSteeringText, buildNotificationSteeringText, STEERING_PREFIX } from "./steering.js";
|
|
54
60
|
import { AgentNotificationQueue } from "./agent-notifications.js";
|
|
55
|
-
import { VerificationGate,
|
|
61
|
+
import { VerificationGate, isCheckOwnFile, extractAddedLines, isCodeFilePath, VERIFICATION_STATE_KIND, isVerificationCommand, } from "./verification-gate.js";
|
|
56
62
|
import { classifyVerificationCommand } from "./verification-evidence.js";
|
|
57
63
|
import { captureVerificationSnapshot } from "./verification-snapshot.js";
|
|
58
64
|
import { findUserSessionPrompt, getUserSessionPrompt } from "./session-preview.js";
|
|
59
65
|
import { normalizeMessageImages } from "./message-images.js";
|
|
66
|
+
import { loadStreamRules } from "./stream-rules.js";
|
|
60
67
|
import crypto from "node:crypto";
|
|
61
68
|
import fs from "node:fs/promises";
|
|
62
69
|
import os from "node:os";
|
|
@@ -66,14 +73,6 @@ import path from "node:path";
|
|
|
66
73
|
* progressing — refuse to extend its turn budget.
|
|
67
74
|
*/
|
|
68
75
|
const TURN_EXTENSION_MAX_FAILURE_RATIO = 0.5;
|
|
69
|
-
/** Terminal subagent states — mirrors SubAgentManager's private isTerminal. */
|
|
70
|
-
function isTerminalSubAgentState(state) {
|
|
71
|
-
return (state === "completed" ||
|
|
72
|
-
state === "failed" ||
|
|
73
|
-
state === "interrupted" ||
|
|
74
|
-
state === "closed" ||
|
|
75
|
-
state === "reaped");
|
|
76
|
-
}
|
|
77
76
|
// ── Tool-result policy ─────────────────────────────────────
|
|
78
77
|
/** Resolve the per-result cap passed to the agent loop for the active transport. */
|
|
79
78
|
export function resolveSessionToolResultCharLimit(model, provider, accountId) {
|
|
@@ -130,6 +129,7 @@ export class AgentSession {
|
|
|
130
129
|
// transcript rows the live run showed.
|
|
131
130
|
appMarkers = [];
|
|
132
131
|
turnMetrics = [];
|
|
132
|
+
cacheDiagnostics = new CacheDiagnostics();
|
|
133
133
|
/** Internal-only (EZ_INTERNAL): live per-session cost/reliability recorder.
|
|
134
134
|
* Absent entirely in public builds — see core/internal-diagnostics.ts. */
|
|
135
135
|
diagnosticsRecorder;
|
|
@@ -141,20 +141,13 @@ export class AgentSession {
|
|
|
141
141
|
/** Forgets every file read; called whenever the conversation is replaced or
|
|
142
142
|
* rewound, so the model must re-read a file before changing it. */
|
|
143
143
|
clearReadTracker;
|
|
144
|
+
recordBashReads;
|
|
144
145
|
skills = [];
|
|
145
146
|
cacheKeyLogged = false;
|
|
146
147
|
// ── Self-correction hook state (mirrors the TUI's useAgentLoop refs) ──
|
|
147
148
|
// Reset at the start of every run; observed from the event stream; read by
|
|
148
|
-
// the loop-break (mid-loop)
|
|
149
|
-
hookStats = {
|
|
150
|
-
changedLines: 0,
|
|
151
|
-
toolCalls: 0,
|
|
152
|
-
toolFailures: 0,
|
|
153
|
-
turns: 0,
|
|
154
|
-
writeCalls: 0,
|
|
155
|
-
editCalls: 0,
|
|
156
|
-
bashCalls: 0,
|
|
157
|
-
};
|
|
149
|
+
// the loop-break (mid-loop) callback.
|
|
150
|
+
hookStats = { toolCalls: 0, toolFailures: 0, turns: 0 };
|
|
158
151
|
hookText = "";
|
|
159
152
|
hookConsecutiveFailures = 0;
|
|
160
153
|
hookRepeatedNoProgressCalls = 0;
|
|
@@ -164,20 +157,6 @@ export class AgentSession {
|
|
|
164
157
|
hookFileEditCounts = new Map();
|
|
165
158
|
hookToolCalls = new Map();
|
|
166
159
|
backgroundVerification = new Map();
|
|
167
|
-
idealReviewPhase = "idle";
|
|
168
|
-
/** Runtime-only suppression while Nolan owns verification in autopilot mode. */
|
|
169
|
-
idealReviewSuppressed = false;
|
|
170
|
-
/** Mirror of the last `hook_armed` value broadcast this run, so the event
|
|
171
|
-
* fires only on a real edge. */
|
|
172
|
-
idealReviewArmed = false;
|
|
173
|
-
/** Cached test-drift probe, keyed by the size of the edited-file set. Drift
|
|
174
|
-
* depends only on WHICH files were edited and that set only grows, so this
|
|
175
|
-
* keeps the arming check off the filesystem on most tool results — the probe
|
|
176
|
-
* is several sync existsSync calls per edited file. */
|
|
177
|
-
idealDriftProbe = null;
|
|
178
|
-
reviewCoverage;
|
|
179
|
-
/** Coverage follow-ups spent this run, capped by MAX_REVIEW_COVERAGE_INJECTIONS. */
|
|
180
|
-
reviewCoverageInjected = 0;
|
|
181
160
|
/** 0 = none; 1 = first nudge sent; 2 = final stop-and-report injected. */
|
|
182
161
|
loopBreakInjected = 0;
|
|
183
162
|
regroundingInjected = false;
|
|
@@ -187,8 +166,6 @@ export class AgentSession {
|
|
|
187
166
|
* injection at the next steering poll; judge failures fail open (no
|
|
188
167
|
* injection) and still consume budget + cooldown. */
|
|
189
168
|
semanticLoop = { checksUsed: 0, lastCheckTurn: 0, pending: false, verdict: null, injected: false };
|
|
190
|
-
/** Independent Ideal reviewer spawned once per run (score-gated). */
|
|
191
|
-
independentReviewStarted = false;
|
|
192
169
|
/**
|
|
193
170
|
* The environment as the cached system prompt currently describes it.
|
|
194
171
|
* Re-recorded on every prompt build, so a rebuild (e.g. `/add-dir`) needs no
|
|
@@ -199,11 +176,12 @@ export class AgentSession {
|
|
|
199
176
|
runStartedAt = 0;
|
|
200
177
|
/** Gate injections spent this run, capped by MAX_PROCESS_GATE_INJECTIONS. */
|
|
201
178
|
processGateInjected = 0;
|
|
202
|
-
/** Verification gate: code edited this run, nothing proved it since.
|
|
179
|
+
/** Verification gate: code edited this run, nothing proved it since. Always
|
|
180
|
+
* tracks evidence for run status; the `verificationGateEnabled` setting
|
|
181
|
+
* decides whether an unverified stop is also continued once. */
|
|
203
182
|
verificationGate = new VerificationGate();
|
|
204
|
-
/** Mirror of the last
|
|
205
|
-
|
|
206
|
-
verificationArmed = false;
|
|
183
|
+
/** Mirror of the last `hook_armed` value, so the event fires only on an edge. */
|
|
184
|
+
preFinalArmed = false;
|
|
207
185
|
compactionOccurred = false;
|
|
208
186
|
/**
|
|
209
187
|
* Re-grounding carry-over for post-turn compaction. `resetHookState` clears
|
|
@@ -216,6 +194,25 @@ export class AgentSession {
|
|
|
216
194
|
postTurnCompaction;
|
|
217
195
|
lastCompactionCompacted = false;
|
|
218
196
|
compactionRetryAfter = 0;
|
|
197
|
+
/**
|
|
198
|
+
* SoL-Pi plan-step compaction bookkeeping (see compaction/plan-step-policy.ts).
|
|
199
|
+
* Step progress comes from `[DONE:n]` markers in assistant text — the same
|
|
200
|
+
* contract the approved-plan UI tracks.
|
|
201
|
+
*/
|
|
202
|
+
planStepState = {
|
|
203
|
+
planPath: undefined,
|
|
204
|
+
scanIndex: 0,
|
|
205
|
+
doneSteps: new Set(),
|
|
206
|
+
requestsInCompletedSteps: 0,
|
|
207
|
+
requestsThisStep: 0,
|
|
208
|
+
requests: 0,
|
|
209
|
+
grownTokens: 0,
|
|
210
|
+
lastContextTokens: 0,
|
|
211
|
+
compactions: 0,
|
|
212
|
+
writeCostBalance: 0,
|
|
213
|
+
savingPerRequest: 0,
|
|
214
|
+
requestsSinceLastCompaction: 0,
|
|
215
|
+
};
|
|
219
216
|
/** A restored oversized checkpoint must be canonicalized before its first prompt is persisted. */
|
|
220
217
|
deferredCompactionPending = false;
|
|
221
218
|
/** Latest provider count, anchored to the assistant response it measured. */
|
|
@@ -231,8 +228,12 @@ export class AgentSession {
|
|
|
231
228
|
// different message (or past the end) by the time the cancel arrives.
|
|
232
229
|
userQueue = [];
|
|
233
230
|
queueSeq = 0;
|
|
231
|
+
/** Instant interrupt: the running loop's preempt listeners, fired on queueMessage. */
|
|
232
|
+
steeringListeners = new Set();
|
|
234
233
|
processManager;
|
|
235
234
|
lspManager;
|
|
235
|
+
testImpact;
|
|
236
|
+
debugManager;
|
|
236
237
|
subAgentManager;
|
|
237
238
|
/**
|
|
238
239
|
* Out-of-band push notifications (finished children, background-process
|
|
@@ -336,7 +337,6 @@ export class AgentSession {
|
|
|
336
337
|
this.provider = options.provider;
|
|
337
338
|
this.model = options.model;
|
|
338
339
|
this.cwd = options.cwd;
|
|
339
|
-
this.reviewCoverage = new ReviewCoverageTracker(this.cwd);
|
|
340
340
|
this.baseUrl = options.baseUrl;
|
|
341
341
|
this.maxTokens = this.resolveMaxTokens(options.model);
|
|
342
342
|
this.thinkingLevel = options.thinkingLevel;
|
|
@@ -402,7 +402,7 @@ export class AgentSession {
|
|
|
402
402
|
: this.opts.globalSubagents
|
|
403
403
|
? await discoverAgents({ globalAgentsDir: paths.agentsDir })
|
|
404
404
|
: [];
|
|
405
|
-
const { tools: builtInTools, processManager, rebuildReadTool, clearReadTracker, lspManager, subAgentManager, } = await createTools(this.cwd, {
|
|
405
|
+
const { tools: builtInTools, processManager, rebuildReadTool, clearReadTracker, recordBashReads, lspManager, testImpact, debugManager, subAgentManager, } = await createTools(this.cwd, {
|
|
406
406
|
agents,
|
|
407
407
|
skills: this.skills,
|
|
408
408
|
contextLimits: this.contextLimits,
|
|
@@ -433,17 +433,14 @@ export class AgentSession {
|
|
|
433
433
|
}),
|
|
434
434
|
getUseExternalGrep: () => this.settingsManager.get("grepUseRipgrep"),
|
|
435
435
|
authStorage: this.authStorage,
|
|
436
|
-
onFileRead: (filePath) => this.reviewCoverage.recordRead(filePath),
|
|
437
436
|
onFileMutated: (filePath) => {
|
|
438
437
|
const relative = path.relative(this.cwd, filePath) || path.basename(filePath);
|
|
439
438
|
this.hookFileEditCounts.set(relative, (this.hookFileEditCounts.get(relative) ?? 0) + 1);
|
|
440
|
-
this.reviewCoverage.recordChanged(filePath);
|
|
441
439
|
},
|
|
442
440
|
// Lazy — sessionId/model/provider can change after createTools() runs, so
|
|
443
441
|
// sub-agent spawns read the current parent state at execution time.
|
|
444
442
|
getProvider: () => this.provider,
|
|
445
443
|
getModel: () => this.model,
|
|
446
|
-
getThinkingLevel: () => this.thinkingLevel,
|
|
447
444
|
getBaseUrl: () => this.baseUrl,
|
|
448
445
|
getCacheKey: () => this.getPromptCacheKey(),
|
|
449
446
|
getMaxPerModel: () => this.settingsManager.get("subagentMaxPerModel"),
|
|
@@ -489,12 +486,16 @@ export class AgentSession {
|
|
|
489
486
|
this.mcpCatalog ??= new DeferredToolCatalog(this.contextLimits);
|
|
490
487
|
this.mcpCatalog.add(deferred);
|
|
491
488
|
this.ensureToolSearchTool();
|
|
489
|
+
this.promoteWaitAgentAfterSpawn();
|
|
492
490
|
}
|
|
493
491
|
}
|
|
494
492
|
this.rebuildReadTool = rebuildReadTool;
|
|
495
493
|
this.clearReadTracker = clearReadTracker;
|
|
494
|
+
this.recordBashReads = recordBashReads;
|
|
496
495
|
this.processManager = processManager;
|
|
497
496
|
this.lspManager = lspManager;
|
|
497
|
+
this.testImpact = testImpact;
|
|
498
|
+
this.debugManager = debugManager;
|
|
498
499
|
this.subAgentManager = subAgentManager;
|
|
499
500
|
this.bindManagerCancellation(this.opts.signal);
|
|
500
501
|
// Connect MCP servers. Child sessions skip user-configured servers to avoid
|
|
@@ -779,6 +780,29 @@ export class AgentSession {
|
|
|
779
780
|
: { serverName, ok: false, error: outcome.error };
|
|
780
781
|
}, this.contextLimits));
|
|
781
782
|
}
|
|
783
|
+
/**
|
|
784
|
+
* `wait_agent` is deferred, yet nearly every `spawn_agent` is followed by it,
|
|
785
|
+
* so the model spent a whole turn on `tool_search` just to load it (bench 41:
|
|
786
|
+
* ~6 s per fan-out). Promote it as soon as a spawn succeeds instead: the tool
|
|
787
|
+
* list grows exactly as it would after that `tool_search`, one turn earlier,
|
|
788
|
+
* and sessions that never spawn keep the smaller prefix.
|
|
789
|
+
*/
|
|
790
|
+
promoteWaitAgentAfterSpawn() {
|
|
791
|
+
const index = this.tools.findIndex((t) => t.name === "spawn_agent");
|
|
792
|
+
const spawn = index >= 0 ? this.tools[index] : undefined;
|
|
793
|
+
if (!spawn)
|
|
794
|
+
return;
|
|
795
|
+
this.tools[index] = {
|
|
796
|
+
...spawn,
|
|
797
|
+
execute: async (args, context) => {
|
|
798
|
+
const result = await spawn.execute(args, context);
|
|
799
|
+
if (!this.tools.some((t) => t.name === "wait_agent")) {
|
|
800
|
+
this.tools.push(...(this.mcpCatalog?.promote(["wait_agent"]) ?? []));
|
|
801
|
+
}
|
|
802
|
+
return result;
|
|
803
|
+
},
|
|
804
|
+
};
|
|
805
|
+
}
|
|
782
806
|
/** Append tools, replacing any same-named entry (cached stub → live tool). */
|
|
783
807
|
replaceOrPushTools(tools) {
|
|
784
808
|
for (const tool of tools) {
|
|
@@ -917,6 +941,7 @@ export class AgentSession {
|
|
|
917
941
|
kind: "prompt",
|
|
918
942
|
visibility: "transcript",
|
|
919
943
|
}, options = {}) {
|
|
944
|
+
this.prewarmController?.abort();
|
|
920
945
|
await this.settlePostTurnCompaction();
|
|
921
946
|
await this.adoptDeferredCheckpointBeforePrompt();
|
|
922
947
|
const slash = await this.resolveSlashInput(content);
|
|
@@ -952,6 +977,7 @@ export class AgentSession {
|
|
|
952
977
|
* attachments are always a direct conversational turn.
|
|
953
978
|
*/
|
|
954
979
|
async promptWithAttachments(text, attachments) {
|
|
980
|
+
this.prewarmController?.abort();
|
|
955
981
|
await this.settlePostTurnCompaction();
|
|
956
982
|
await this.adoptDeferredCheckpointBeforePrompt();
|
|
957
983
|
const parts = this.buildAttachmentParts(text, attachments);
|
|
@@ -1050,15 +1076,7 @@ export class AgentSession {
|
|
|
1050
1076
|
resetHookState(originalRequest) {
|
|
1051
1077
|
this.opts.completionReview?.begin(originalRequest);
|
|
1052
1078
|
this.lspManager?.clearPendingDiagnostics();
|
|
1053
|
-
this.hookStats = {
|
|
1054
|
-
changedLines: 0,
|
|
1055
|
-
toolCalls: 0,
|
|
1056
|
-
toolFailures: 0,
|
|
1057
|
-
turns: 0,
|
|
1058
|
-
writeCalls: 0,
|
|
1059
|
-
editCalls: 0,
|
|
1060
|
-
bashCalls: 0,
|
|
1061
|
-
};
|
|
1079
|
+
this.hookStats = { toolCalls: 0, toolFailures: 0, turns: 0 };
|
|
1062
1080
|
this.hookText = "";
|
|
1063
1081
|
this.hookConsecutiveFailures = 0;
|
|
1064
1082
|
this.hookRepeatedNoProgressCalls = 0;
|
|
@@ -1067,13 +1085,6 @@ export class AgentSession {
|
|
|
1067
1085
|
this.hookCyclicPattern = null;
|
|
1068
1086
|
this.hookFileEditCounts.clear();
|
|
1069
1087
|
this.hookToolCalls.clear();
|
|
1070
|
-
this.reviewCoverage.reset();
|
|
1071
|
-
this.reviewCoverageInjected = 0;
|
|
1072
|
-
this.idealReviewPhase = "idle";
|
|
1073
|
-
// No event here: clients reset their own hold on run_start.
|
|
1074
|
-
this.idealReviewArmed = false;
|
|
1075
|
-
this.verificationArmed = false;
|
|
1076
|
-
this.idealDriftProbe = null;
|
|
1077
1088
|
this.loopBreakInjected = 0;
|
|
1078
1089
|
this.regroundingInjected = false;
|
|
1079
1090
|
this.hookRecentCalls = [];
|
|
@@ -1084,7 +1095,6 @@ export class AgentSession {
|
|
|
1084
1095
|
verdict: null,
|
|
1085
1096
|
injected: false,
|
|
1086
1097
|
};
|
|
1087
|
-
this.independentReviewStarted = false;
|
|
1088
1098
|
this.runStartedAt = Date.now();
|
|
1089
1099
|
this.processGateInjected = 0;
|
|
1090
1100
|
this.verificationGate.beginRun();
|
|
@@ -1105,8 +1115,8 @@ export class AgentSession {
|
|
|
1105
1115
|
}
|
|
1106
1116
|
/**
|
|
1107
1117
|
* Fold one agent event into the hook stat accumulators. Pure bookkeeping —
|
|
1108
|
-
* the same signals the TUI's useAgentLoop collects, so
|
|
1109
|
-
*
|
|
1118
|
+
* the same signals the TUI's useAgentLoop collects, so loop-break decisions
|
|
1119
|
+
* match across the CLI and the app.
|
|
1110
1120
|
*/
|
|
1111
1121
|
async trackHookEvent(event) {
|
|
1112
1122
|
if (this.opts.completionReview) {
|
|
@@ -1172,12 +1182,6 @@ export class AgentSession {
|
|
|
1172
1182
|
this.hookStats.toolCalls += 1;
|
|
1173
1183
|
if (event.isError)
|
|
1174
1184
|
this.hookStats.toolFailures += 1;
|
|
1175
|
-
if (name === "write")
|
|
1176
|
-
this.hookStats.writeCalls += 1;
|
|
1177
|
-
if (name === "edit")
|
|
1178
|
-
this.hookStats.editCalls += 1;
|
|
1179
|
-
if (name === "bash")
|
|
1180
|
-
this.hookStats.bashCalls += 1;
|
|
1181
1185
|
this.hookConsecutiveFailures = event.isError ? this.hookConsecutiveFailures + 1 : 0;
|
|
1182
1186
|
this.hookRepeatedNoProgressCalls = this.hookProgressTracker.record(name, args, event.result, event.isError);
|
|
1183
1187
|
this.hookCyclicPattern = this.hookCycleDetector.record(name, args, event.result, event.isError);
|
|
@@ -1194,12 +1198,6 @@ export class AgentSession {
|
|
|
1194
1198
|
if (this.hookRecentCalls.length > MAX_SEMANTIC_LOOP_CALLS) {
|
|
1195
1199
|
this.hookRecentCalls.splice(0, this.hookRecentCalls.length - MAX_SEMANTIC_LOOP_CALLS);
|
|
1196
1200
|
}
|
|
1197
|
-
if (name === "edit" && !event.isError) {
|
|
1198
|
-
const diff = event.details?.diff ?? event.result;
|
|
1199
|
-
const added = (diff.match(/^\+[^+]/gm) ?? []).length;
|
|
1200
|
-
const removed = (diff.match(/^-[^-]/gm) ?? []).length;
|
|
1201
|
-
this.hookStats.changedLines += added + removed;
|
|
1202
|
-
}
|
|
1203
1201
|
// Only host-observed successful mutations and trustworthy check results
|
|
1204
1202
|
// affect approval. The model's text is never evidence.
|
|
1205
1203
|
let verificationChanged = false;
|
|
@@ -1222,8 +1220,13 @@ export class AgentSession {
|
|
|
1222
1220
|
const command = typeof args.command === "string" ? args.command : "";
|
|
1223
1221
|
const classification = classifyVerificationCommand(command);
|
|
1224
1222
|
if (classification.accepted || classification.snapshotEligible) {
|
|
1225
|
-
|
|
1226
|
-
|
|
1223
|
+
// A foreground check that outlived the default budget was moved to
|
|
1224
|
+
// the background, not failed: track it to its real exit the same way.
|
|
1225
|
+
const autoBackgroundId = event.isError ? undefined : autoBackgroundedId(event.result);
|
|
1226
|
+
if ((autoBackgroundId !== undefined || args.run_in_background === true) &&
|
|
1227
|
+
!event.isError &&
|
|
1228
|
+
args.persist !== true) {
|
|
1229
|
+
const id = autoBackgroundId ?? /^ID:\s*(\S+)/m.exec(event.result)?.[1];
|
|
1227
1230
|
// No parseable ID means the check cannot be tracked to a real exit
|
|
1228
1231
|
// code — no evidence either way. Recording a FAILURE here made
|
|
1229
1232
|
// every later green run of a different spelling look owed.
|
|
@@ -1294,10 +1297,35 @@ export class AgentSession {
|
|
|
1294
1297
|
// Tool results for this step are in the array and their side effects
|
|
1295
1298
|
// already hit the filesystem. Flushing here is what makes a crash lose
|
|
1296
1299
|
// at most the in-flight step instead of the entire turn.
|
|
1300
|
+
await this.creditBashReads();
|
|
1297
1301
|
await this.flushPendingMessages();
|
|
1298
1302
|
break;
|
|
1299
1303
|
}
|
|
1300
1304
|
}
|
|
1305
|
+
/**
|
|
1306
|
+
* Count full-file `cat` output from the step that just finished as reads, so
|
|
1307
|
+
* an edit after `cat` does not cost a second read. Uses the step's results as
|
|
1308
|
+
* stored in the transcript (after per-turn trimming): only bytes the model
|
|
1309
|
+
* actually received count.
|
|
1310
|
+
*/
|
|
1311
|
+
async creditBashReads() {
|
|
1312
|
+
if (!this.recordBashReads)
|
|
1313
|
+
return;
|
|
1314
|
+
const messages = this.activeLoopMessages ?? this.messages;
|
|
1315
|
+
const toolMessage = messages.at(-1);
|
|
1316
|
+
const assistant = messages.at(-2);
|
|
1317
|
+
if (toolMessage?.role !== "tool" || assistant?.role !== "assistant")
|
|
1318
|
+
return;
|
|
1319
|
+
if (typeof assistant.content === "string")
|
|
1320
|
+
return;
|
|
1321
|
+
const calls = assistant.content.filter((part) => part.type === "tool_call");
|
|
1322
|
+
try {
|
|
1323
|
+
await this.recordBashReads(calls, toolMessage.content);
|
|
1324
|
+
}
|
|
1325
|
+
catch (err) {
|
|
1326
|
+
log("WARN", "agent-session", "Crediting bash reads failed", { error: String(err) });
|
|
1327
|
+
}
|
|
1328
|
+
}
|
|
1301
1329
|
/**
|
|
1302
1330
|
* Append every message added since the last flush to the session file.
|
|
1303
1331
|
*
|
|
@@ -1360,7 +1388,7 @@ export class AgentSession {
|
|
|
1360
1388
|
const diagnosticText = this.lspManager?.drainDiagnostics(this.getVerificationProblem() !== null, { deferUnverified: true });
|
|
1361
1389
|
if (diagnosticText)
|
|
1362
1390
|
this.eventBus.emit("diagnostics", { text: diagnosticText });
|
|
1363
|
-
this.
|
|
1391
|
+
this.refreshHookArming();
|
|
1364
1392
|
const notified = this.notifications.drain();
|
|
1365
1393
|
const notificationMessage = notified.length > 0 || diagnosticText
|
|
1366
1394
|
? {
|
|
@@ -1423,6 +1451,7 @@ export class AgentSession {
|
|
|
1423
1451
|
}
|
|
1424
1452
|
if (this.opts.selfCorrectionHooks === false)
|
|
1425
1453
|
return null;
|
|
1454
|
+
// Legacy key: the user-facing switch for loop-break and re-grounding nudges.
|
|
1426
1455
|
if (!this.settingsManager.get("idealReviewEnabled"))
|
|
1427
1456
|
return null;
|
|
1428
1457
|
// Deterministic stuck verdict, computed once and shared: the semantic
|
|
@@ -1622,83 +1651,6 @@ export class AgentSession {
|
|
|
1622
1651
|
.join("\n")
|
|
1623
1652
|
: "";
|
|
1624
1653
|
}
|
|
1625
|
-
/** Independent fresh-context review of the finished work (Codex Guardian
|
|
1626
|
-
* pattern). Spawns a READ-ONLY child on the ACTIVE model, waits bounded,
|
|
1627
|
-
* and returns findings for the acting agent to address — or nothing when
|
|
1628
|
-
* the review passes, is unavailable, or fails (in-thread review remains the
|
|
1629
|
-
* fallback; the feature degrades, never blocks).
|
|
1630
|
-
*
|
|
1631
|
-
* Runs inside the pre-stop poll, so the candidate final answer is already
|
|
1632
|
-
* held by arming and this wait cannot race a streamed answer. */
|
|
1633
|
-
async runIndependentReview(decision) {
|
|
1634
|
-
if (!this.subAgentManager)
|
|
1635
|
-
return [];
|
|
1636
|
-
if (this.independentReviewStarted)
|
|
1637
|
-
return [];
|
|
1638
|
-
// An allow-listed session (a subagent worker itself) must not spawn
|
|
1639
|
-
// harness-owned grandchildren the tool policy never granted.
|
|
1640
|
-
if (this.opts.allowedTools && !this.opts.allowedTools.includes("spawn_agent"))
|
|
1641
|
-
return [];
|
|
1642
|
-
if (decision.score < INDEPENDENT_REVIEW_SCORE_THRESHOLD)
|
|
1643
|
-
return [];
|
|
1644
|
-
this.independentReviewStarted = true;
|
|
1645
|
-
const taskName = `ideal-reviewer-${Math.random().toString(36).slice(2, 8)}`;
|
|
1646
|
-
let agentId;
|
|
1647
|
-
try {
|
|
1648
|
-
const task = buildReviewerTask({
|
|
1649
|
-
originalRequest: this.originalRequest,
|
|
1650
|
-
changedFiles: [...this.hookFileEditCounts.keys()],
|
|
1651
|
-
stats: this.hookStats,
|
|
1652
|
-
triggerReasons: decision.reasons,
|
|
1653
|
-
});
|
|
1654
|
-
// Active model forced at spawn time — never routed to a fast/review model.
|
|
1655
|
-
// The reviewer's own time limit ends it with a verdict on what it read;
|
|
1656
|
-
// the wait below is only a backstop against a hung child.
|
|
1657
|
-
const snapshot = await this.subAgentManager.spawn(taskName, task, undefined, {
|
|
1658
|
-
model: this.model,
|
|
1659
|
-
tools: REVIEWER_TOOLS,
|
|
1660
|
-
turnTimeoutMs: REVIEWER_TURN_TIMEOUT_MS,
|
|
1661
|
-
});
|
|
1662
|
-
agentId = snapshot.agent_id;
|
|
1663
|
-
const waited = await this.subAgentManager.wait([agentId], "all", REVIEWER_WAIT_MS);
|
|
1664
|
-
const agent = waited.agents[0];
|
|
1665
|
-
if (!agent || !isTerminalSubAgentState(agent.state)) {
|
|
1666
|
-
// Timeout: collect the straggler so the completion gate cannot fire on
|
|
1667
|
-
// it later, then fall back to the in-thread review.
|
|
1668
|
-
await this.subAgentManager.interrupt(agentId, true).catch(() => { });
|
|
1669
|
-
log("WARN", "ideal", "Independent reviewer timed out; falling back to in-thread review", {
|
|
1670
|
-
agentId,
|
|
1671
|
-
});
|
|
1672
|
-
return [];
|
|
1673
|
-
}
|
|
1674
|
-
const findings = parseReviewerFindings(agent.output ?? "");
|
|
1675
|
-
if (!findings) {
|
|
1676
|
-
log("WARN", "ideal", "Independent reviewer output unparseable; falling back", {
|
|
1677
|
-
agentId,
|
|
1678
|
-
state: agent.state,
|
|
1679
|
-
...(agent.error ? { error: agent.error } : {}),
|
|
1680
|
-
});
|
|
1681
|
-
return [];
|
|
1682
|
-
}
|
|
1683
|
-
if (findings.clean) {
|
|
1684
|
-
log("INFO", "ideal", "Independent reviewer verdict: clean", { agentId });
|
|
1685
|
-
return [];
|
|
1686
|
-
}
|
|
1687
|
-
log("INFO", "ideal", "Independent reviewer flagged findings", {
|
|
1688
|
-
agentId,
|
|
1689
|
-
count: String(findings.findings.length),
|
|
1690
|
-
});
|
|
1691
|
-
return [buildIndependentReviewMessage(findings.findings)];
|
|
1692
|
-
}
|
|
1693
|
-
catch (error) {
|
|
1694
|
-
if (agentId)
|
|
1695
|
-
await this.subAgentManager.interrupt(agentId, true).catch(() => { });
|
|
1696
|
-
log("WARN", "ideal", "Independent reviewer failed; falling back to in-thread review", {
|
|
1697
|
-
error: error instanceof Error ? error.message : String(error),
|
|
1698
|
-
});
|
|
1699
|
-
return [];
|
|
1700
|
-
}
|
|
1701
|
-
}
|
|
1702
1654
|
/**
|
|
1703
1655
|
* Turn-budget extension gate. The loop consults this instead of stopping
|
|
1704
1656
|
* mid-task when it exhausts `maxTurns`. Grant ONLY on evidence of progress —
|
|
@@ -1732,98 +1684,45 @@ export class AgentSession {
|
|
|
1732
1684
|
});
|
|
1733
1685
|
return granted;
|
|
1734
1686
|
}
|
|
1735
|
-
/**
|
|
1736
|
-
*
|
|
1737
|
-
*
|
|
1738
|
-
|
|
1739
|
-
|
|
1740
|
-
* The turn count is looked ahead by one on purpose. `hookStats.turns` only
|
|
1741
|
-
* advances at `turn_end`, so while the model is writing the draft the counter
|
|
1742
|
-
* still reads the PREVIOUS turn; the real gate sees one more. Without the
|
|
1743
|
-
* lookahead a run sitting on score 3 crosses to 4 on the draft's own
|
|
1744
|
-
* `turn_end` — after the text already streamed — which is precisely the
|
|
1745
|
-
* appear-then-vanish flash. Over-arming by one turn point costs only live
|
|
1746
|
-
* token streaming on a final answer that then shows whole; under-arming costs
|
|
1747
|
-
* the flash, so this errs toward arming.
|
|
1748
|
-
*/
|
|
1749
|
-
wouldInjectIdealReview() {
|
|
1750
|
-
if (this.opts.completionReview?.armed)
|
|
1751
|
-
return true;
|
|
1752
|
-
if (this.opts.selfCorrectionHooks === false || this.idealReviewSuppressed)
|
|
1753
|
-
return false;
|
|
1754
|
-
// Mid-review a stop still injects: the coverage follow-up while files are
|
|
1755
|
-
// unread, or its escalation once the budget is spent. Both make the model
|
|
1756
|
-
// answer again, so the candidate answer is a draft exactly as it is before
|
|
1757
|
-
// the review starts — without arming here it paints and the reviewed answer
|
|
1758
|
-
// lands under it as a duplicate.
|
|
1759
|
-
if (this.idealReviewPhase === "reviewing") {
|
|
1760
|
-
return this.reviewCoverage.evidence().missing.length > 0;
|
|
1761
|
-
}
|
|
1762
|
-
if (this.idealReviewPhase !== "idle")
|
|
1763
|
-
return false;
|
|
1764
|
-
if (!this.settingsManager.get("idealReviewEnabled"))
|
|
1687
|
+
/** Is the pre-stop verification gate active for this session? Off by the
|
|
1688
|
+
* `verificationGateEnabled` setting, by `selfCorrectionHooks: false`, and for
|
|
1689
|
+
* allow-listed sessions that cannot run commands at all. */
|
|
1690
|
+
verificationGateActive() {
|
|
1691
|
+
if (this.opts.selfCorrectionHooks === false)
|
|
1765
1692
|
return false;
|
|
1766
|
-
if (
|
|
1767
|
-
return true;
|
|
1768
|
-
}
|
|
1769
|
-
const files = this.hookFileEditCounts.size;
|
|
1770
|
-
if (files === 0)
|
|
1693
|
+
if (!this.settingsManager.get("verificationGateEnabled"))
|
|
1771
1694
|
return false;
|
|
1772
|
-
|
|
1773
|
-
this.idealDriftProbe = {
|
|
1774
|
-
files,
|
|
1775
|
-
drifted: detectTestDrift(this.hookFileEditCounts.keys(), this.cwd).length > 0,
|
|
1776
|
-
};
|
|
1777
|
-
}
|
|
1778
|
-
return this.idealDriftProbe.drifted;
|
|
1695
|
+
return !this.opts.allowedTools || this.opts.allowedTools.includes("bash");
|
|
1779
1696
|
}
|
|
1780
|
-
/** Would a stop right now inject
|
|
1781
|
-
*
|
|
1782
|
-
|
|
1697
|
+
/** Would a stop right now inject a pre-final follow-up? Queued LSP
|
|
1698
|
+
* diagnostics (real errors injected below), a mode-owned completion review,
|
|
1699
|
+
* or the verification gate can, so clients hold the candidate answer only
|
|
1700
|
+
* then. Same conditions as the pre-stop branch, so arming and injection
|
|
1701
|
+
* cannot disagree. */
|
|
1702
|
+
wouldInjectBeforeFinal() {
|
|
1703
|
+
if (this.opts.completionReview?.armed)
|
|
1704
|
+
return true;
|
|
1783
1705
|
if (this.lspManager?.hasQueuedDiagnostics())
|
|
1784
1706
|
return true;
|
|
1785
|
-
|
|
1786
|
-
return false;
|
|
1787
|
-
if (!this.settingsManager.get("verificationGateEnabled"))
|
|
1788
|
-
return false;
|
|
1789
|
-
if (this.opts.allowedTools && !this.opts.allowedTools.includes("bash"))
|
|
1790
|
-
return false;
|
|
1791
|
-
return this.verificationGate.willInject();
|
|
1707
|
+
return this.verificationGateActive() && this.verificationGate.willInject();
|
|
1792
1708
|
}
|
|
1793
1709
|
/** Broadcast pre-final hook arming on change. Both edges matter: armed=false
|
|
1794
|
-
* after the hook fires is what lets a client stream the
|
|
1795
|
-
*
|
|
1796
|
-
*
|
|
1797
|
-
* Callable before `initialize()`: the sidecar sets Nolan's review suppression
|
|
1798
|
-
* on a freshly constructed session, and every arming predicate below reads
|
|
1799
|
-
* settings that `initialize()` has not loaded yet. Nothing can be armed
|
|
1800
|
-
* before the session can run a turn, and the first `tool_result`/`turn_end`
|
|
1801
|
-
* recomputes both edges — so skipping is the correct answer, not a patch. */
|
|
1710
|
+
* after the hook fires is what lets a client stream the final answer live
|
|
1711
|
+
* again. Callable before `initialize()`, when no manager exists yet. */
|
|
1802
1712
|
refreshHookArming() {
|
|
1713
|
+
// Before `initialize()` settings are not loaded, so nothing can be armed.
|
|
1803
1714
|
if (!this.settingsManager)
|
|
1804
1715
|
return;
|
|
1805
|
-
this.
|
|
1806
|
-
this.
|
|
1807
|
-
}
|
|
1808
|
-
refreshVerificationArmed() {
|
|
1809
|
-
if (!this.settingsManager)
|
|
1810
|
-
return;
|
|
1811
|
-
const armed = this.wouldInjectVerification();
|
|
1812
|
-
if (armed === this.verificationArmed)
|
|
1716
|
+
const armed = this.wouldInjectBeforeFinal();
|
|
1717
|
+
if (armed === this.preFinalArmed)
|
|
1813
1718
|
return;
|
|
1814
|
-
this.
|
|
1719
|
+
this.preFinalArmed = armed;
|
|
1815
1720
|
this.eventBus.emit("hook_armed", { kind: "verification", armed });
|
|
1816
1721
|
}
|
|
1817
|
-
refreshIdealReviewArmed() {
|
|
1818
|
-
const armed = this.wouldInjectIdealReview();
|
|
1819
|
-
if (armed === this.idealReviewArmed)
|
|
1820
|
-
return;
|
|
1821
|
-
this.idealReviewArmed = armed;
|
|
1822
|
-
this.eventBus.emit("hook_armed", { kind: "ideal", armed });
|
|
1823
|
-
}
|
|
1824
1722
|
/**
|
|
1825
|
-
* Pre-stop
|
|
1826
|
-
*
|
|
1723
|
+
* Pre-stop follow-ups: LSP errors, unread child agents and background
|
|
1724
|
+
* processes, the verification gate (when `verificationGateEnabled`), and a
|
|
1725
|
+
* mode-owned completion review.
|
|
1827
1726
|
*/
|
|
1828
1727
|
async getHookFollowUpMessages() {
|
|
1829
1728
|
// Exit notifications and task_output refer to the same host process record.
|
|
@@ -1836,14 +1735,15 @@ export class AgentSession {
|
|
|
1836
1735
|
if (backgroundChanged)
|
|
1837
1736
|
await this.persistVerificationState();
|
|
1838
1737
|
// Edits return immediately; only the completion boundary waits for remaining
|
|
1839
|
-
// checks. Queued timeouts stay explicitly unverified, never a false
|
|
1738
|
+
// checks. Queued timeouts stay explicitly unverified, never a false
|
|
1739
|
+
// all-clear; they join the verification demand below when the gate is on.
|
|
1840
1740
|
await this.lspManager?.flushDiagnostics(this.opts.signal);
|
|
1841
1741
|
if (this.opts.signal?.aborted)
|
|
1842
1742
|
return null;
|
|
1843
|
-
const diagnosticText = this.lspManager?.drainDiagnostics(this.getVerificationProblem() !== null);
|
|
1743
|
+
const diagnosticText = this.lspManager?.drainDiagnostics(this.verificationGateActive() && this.getVerificationProblem() !== null);
|
|
1844
1744
|
if (diagnosticText)
|
|
1845
1745
|
this.eventBus.emit("diagnostics", { text: diagnosticText });
|
|
1846
|
-
this.
|
|
1746
|
+
this.refreshHookArming();
|
|
1847
1747
|
const diagnosticMessages = diagnosticText
|
|
1848
1748
|
? [
|
|
1849
1749
|
{
|
|
@@ -1868,18 +1768,29 @@ export class AgentSession {
|
|
|
1868
1768
|
return [...diagnosticMessages, ...processFollowUp];
|
|
1869
1769
|
}
|
|
1870
1770
|
// Verification gate: code was edited but nothing verified since the last
|
|
1871
|
-
// edit.
|
|
1872
|
-
|
|
1873
|
-
if (this.opts.selfCorrectionHooks !== false &&
|
|
1874
|
-
this.settingsManager.get("verificationGateEnabled") &&
|
|
1875
|
-
(!this.opts.allowedTools || this.opts.allowedTools.includes("bash"))) {
|
|
1771
|
+
// edit. Off via the `verificationGateEnabled` setting.
|
|
1772
|
+
if (this.verificationGateActive()) {
|
|
1876
1773
|
const verificationReason = this.verificationGate.pendingReason();
|
|
1774
|
+
const pendingFiles = this.verificationGate.pendingFiles();
|
|
1877
1775
|
const verificationFollowUp = this.verificationGate.followUp();
|
|
1878
1776
|
if (verificationFollowUp) {
|
|
1777
|
+
// Name the tests that actually reach the unverified files, with the
|
|
1778
|
+
// command that runs exactly those, so the check is targeted rather
|
|
1779
|
+
// than guessed. Best-effort: no index or no runner leaves it unchanged.
|
|
1780
|
+
if (this.testImpact &&
|
|
1781
|
+
(verificationReason === "initial" || verificationReason === "recheck")) {
|
|
1782
|
+
const impactLine = await this.testImpact
|
|
1783
|
+
.impactFor(pendingFiles)
|
|
1784
|
+
.then(formatImpactForVerification)
|
|
1785
|
+
.catch(() => "");
|
|
1786
|
+
const first = verificationFollowUp[0];
|
|
1787
|
+
if (impactLine && first?.role === "user" && typeof first.content === "string") {
|
|
1788
|
+
verificationFollowUp[0] = { ...first, content: first.content + impactLine };
|
|
1789
|
+
}
|
|
1790
|
+
}
|
|
1879
1791
|
log("INFO", "verification-gate", "Injecting verification follow-up", {});
|
|
1880
1792
|
// Announce, THEN disarm: clients release held text on disarm, so the
|
|
1881
|
-
// reverse order paints the draft and immediately deletes it
|
|
1882
|
-
// flash arming exists to prevent.
|
|
1793
|
+
// reverse order paints the draft and immediately deletes it.
|
|
1883
1794
|
this.eventBus.emit("hook", {
|
|
1884
1795
|
kind: "verification",
|
|
1885
1796
|
...(verificationReason === "tamper"
|
|
@@ -1892,7 +1803,7 @@ export class AgentSession {
|
|
|
1892
1803
|
return [...diagnosticMessages, ...verificationFollowUp];
|
|
1893
1804
|
}
|
|
1894
1805
|
}
|
|
1895
|
-
// Address real errors before review; unavailable checks share the
|
|
1806
|
+
// Address real errors before review; unavailable checks share the
|
|
1896
1807
|
// verification demand above instead of manufacturing a separate hook.
|
|
1897
1808
|
if (diagnosticMessages.length > 0)
|
|
1898
1809
|
return diagnosticMessages;
|
|
@@ -1912,131 +1823,24 @@ export class AgentSession {
|
|
|
1912
1823
|
}
|
|
1913
1824
|
this.refreshHookArming();
|
|
1914
1825
|
}
|
|
1915
|
-
|
|
1916
|
-
|
|
1917
|
-
|
|
1918
|
-
|
|
1919
|
-
|
|
1920
|
-
|
|
1921
|
-
|
|
1922
|
-
|
|
1923
|
-
|
|
1924
|
-
|
|
1925
|
-
|
|
1926
|
-
|
|
1927
|
-
|
|
1928
|
-
// model answer again, so the answer it interrupts is a draft and the
|
|
1929
|
-
// hook event is what tells clients to discard it. Injecting silently is
|
|
1930
|
-
// what let the pre-coverage answer paint above the reviewed one.
|
|
1931
|
-
this.eventBus.emit("hook", {
|
|
1932
|
-
kind: "ideal",
|
|
1933
|
-
coverageExpected: coverage.expected,
|
|
1934
|
-
coverageMissing: coverage.missing,
|
|
1935
|
-
});
|
|
1936
|
-
if (this.reviewCoverageInjected < MAX_REVIEW_COVERAGE_INJECTIONS) {
|
|
1937
|
-
this.reviewCoverageInjected += 1;
|
|
1938
|
-
// Stays armed (coverage is still outstanding) — this call is here so a
|
|
1939
|
-
// client that missed the earlier edge is armed before the next draft.
|
|
1940
|
-
this.refreshIdealReviewArmed();
|
|
1941
|
-
return [
|
|
1942
|
-
this.withReviewLspEvidence(buildReviewCoverageMessage(coverage.missing), lspEvidence),
|
|
1943
|
-
];
|
|
1944
|
-
}
|
|
1945
|
-
// Budget spent: close the gate so the run cannot spin on a file that
|
|
1946
|
-
// never becomes readable, and require the gap be reported to the user.
|
|
1947
|
-
this.idealReviewPhase = "complete";
|
|
1948
|
-
// The gate is shut, so this is the real disarm: the answer to the
|
|
1949
|
-
// escalation is final and streams live.
|
|
1950
|
-
this.refreshIdealReviewArmed();
|
|
1951
|
-
log("INFO", "ideal", "Ideal review coverage escalated after retry budget", {
|
|
1952
|
-
injected: String(this.reviewCoverageInjected),
|
|
1953
|
-
missing: coverage.missing,
|
|
1954
|
-
});
|
|
1955
|
-
return [buildReviewCoverageEscalationMessage(coverage.missing)];
|
|
1956
|
-
}
|
|
1957
|
-
this.idealReviewPhase = "complete";
|
|
1958
|
-
return null;
|
|
1826
|
+
return null;
|
|
1827
|
+
}
|
|
1828
|
+
/** Wraps the real run: cancels an in-flight cache prewarm and tracks run
|
|
1829
|
+
* activity / last real request time for {@link prewarm}. */
|
|
1830
|
+
async runLoop(options = {}) {
|
|
1831
|
+
this.prewarmController?.abort();
|
|
1832
|
+
this.runLoopDepth++;
|
|
1833
|
+
try {
|
|
1834
|
+
await this.runLoopInner(options);
|
|
1835
|
+
}
|
|
1836
|
+
finally {
|
|
1837
|
+
this.runLoopDepth--;
|
|
1838
|
+
this.lastRealRequestAt = Date.now();
|
|
1959
1839
|
}
|
|
1960
|
-
if (this.idealReviewPhase === "complete")
|
|
1961
|
-
return null;
|
|
1962
|
-
if (!this.settingsManager.get("idealReviewEnabled"))
|
|
1963
|
-
return null;
|
|
1964
|
-
const decision = evaluateIdealReview(this.hookStats);
|
|
1965
|
-
// Test drift fires the review even on a small change the score would skip:
|
|
1966
|
-
// a green-but-stale test is exactly what the volume gate sleeps through.
|
|
1967
|
-
const driftedFiles = detectTestDrift(this.hookFileEditCounts.keys(), this.cwd).slice(0, 5);
|
|
1968
|
-
if (!decision.shouldReview && driftedFiles.length === 0)
|
|
1969
|
-
return null;
|
|
1970
|
-
// Independent reviewer first (async, bounded): its findings ride in the
|
|
1971
|
-
// SAME follow-up batch as the in-thread review + coverage requirements, so
|
|
1972
|
-
// addressing everything still costs one extra turn.
|
|
1973
|
-
this.reviewCoverage.start(this.hookFileEditCounts.keys());
|
|
1974
|
-
this.idealReviewPhase = "reviewing";
|
|
1975
|
-
const coverage = this.reviewCoverage.evidence();
|
|
1976
|
-
const lspEvidence = this.reviewLspEvidence(coverage.expected);
|
|
1977
|
-
this.eventBus.emit("hook", {
|
|
1978
|
-
kind: "ideal",
|
|
1979
|
-
coverageExpected: coverage.expected,
|
|
1980
|
-
coverageMissing: coverage.missing,
|
|
1981
|
-
});
|
|
1982
|
-
// Recompute strictly AFTER the hook event: clients release held text on
|
|
1983
|
-
// disarm, so the reverse order would paint the draft and then delete it —
|
|
1984
|
-
// the exact flash arming exists to prevent. Arming normally PERSISTS here,
|
|
1985
|
-
// because review starts with every changed file uncovered and a stop while
|
|
1986
|
-
// coverage is outstanding injects again. Disarm lands later, on the read
|
|
1987
|
-
// that closes the last gap (or when the retry budget escalates).
|
|
1988
|
-
this.refreshIdealReviewArmed();
|
|
1989
|
-
// Announce the phase before the reviewer starts, not after its bounded wait.
|
|
1990
|
-
const independentMessages = await this.runIndependentReview(decision);
|
|
1991
|
-
log("INFO", "ideal", "Injecting ideal review before final response", {
|
|
1992
|
-
coverageExpected: coverage.expected,
|
|
1993
|
-
coverageMissing: coverage.missing,
|
|
1994
|
-
lspLowConfidence: lspEvidence.lowConfidence,
|
|
1995
|
-
lspMissing: lspEvidence.missing,
|
|
1996
|
-
});
|
|
1997
|
-
return [
|
|
1998
|
-
...independentMessages,
|
|
1999
|
-
this.withReviewLspEvidence(withReviewCoverageRequirements(buildIdealReviewMessage(decision.reasons, driftedFiles), coverage.missing), lspEvidence),
|
|
2000
|
-
];
|
|
2001
|
-
}
|
|
2002
|
-
reviewLspEvidence(files) {
|
|
2003
|
-
const lowConfidence = [];
|
|
2004
|
-
const missing = [];
|
|
2005
|
-
for (const filePath of files) {
|
|
2006
|
-
const outcome = this.lspManager?.getLatestOutcome(filePath);
|
|
2007
|
-
if (outcome?.kind === "low_confidence")
|
|
2008
|
-
lowConfidence.push(filePath);
|
|
2009
|
-
else if (outcome?.kind !== "clean" && outcome?.kind !== "diagnostics")
|
|
2010
|
-
missing.push(filePath);
|
|
2011
|
-
}
|
|
2012
|
-
return { lowConfidence, missing };
|
|
2013
|
-
}
|
|
2014
|
-
withReviewLspEvidence(message, evidence) {
|
|
2015
|
-
if (evidence.lowConfidence.length === 0 && evidence.missing.length === 0)
|
|
2016
|
-
return message;
|
|
2017
|
-
const notes = [
|
|
2018
|
-
...(evidence.lowConfidence.length > 0
|
|
2019
|
-
? [`Diagnostics are low confidence while indexing: ${evidence.lowConfidence.join(", ")}.`]
|
|
2020
|
-
: []),
|
|
2021
|
-
...(evidence.missing.length > 0
|
|
2022
|
-
? [`Diagnostics evidence is unavailable or missing: ${evidence.missing.join(", ")}.`]
|
|
2023
|
-
: []),
|
|
2024
|
-
"Do not describe those files as compiler-clean without other evidence.",
|
|
2025
|
-
];
|
|
2026
|
-
return {
|
|
2027
|
-
role: "user",
|
|
2028
|
-
provenance: message.provenance,
|
|
2029
|
-
content: `${String(message.content)}\n\n${notes.join(" ")}`,
|
|
2030
|
-
};
|
|
2031
1840
|
}
|
|
2032
1841
|
/** Auto-compact if needed, run agent loop with auth retry, and persist messages. */
|
|
2033
|
-
async
|
|
2034
|
-
|
|
2035
|
-
// during the previous turn gets its packs; the prompt is rebuilt only when
|
|
2036
|
-
// the set grows, keeping the cached prefix stable otherwise.
|
|
2037
|
-
if (this.refreshActiveLanguages())
|
|
2038
|
-
await this.rebuildSystemPromptInPlace();
|
|
2039
|
-
this.refreshSystemPromptTail();
|
|
1842
|
+
async runLoopInner(options = {}) {
|
|
1843
|
+
await this.prepareSystemPromptForRequest();
|
|
2040
1844
|
// One-shot cache-key marker per session so turn_end cacheRead numbers
|
|
2041
1845
|
// in the log can be traced back to a specific routing namespace —
|
|
2042
1846
|
// particularly useful when sub-agents inherit `parentKey:subagent`.
|
|
@@ -2135,11 +1939,14 @@ export class AgentSession {
|
|
|
2135
1939
|
// so the new value transparently yields a correctly-identified client.
|
|
2136
1940
|
let userAgent = this.provider === "anthropic" ? await getClaudeCliUserAgent() : undefined;
|
|
2137
1941
|
const loopMessages = await this.prepareDynamicContext();
|
|
1942
|
+
// Re-read every run so rule-file edits apply on the next turn; a few small files.
|
|
1943
|
+
const streamRules = await loadStreamRules(this.cwd);
|
|
2138
1944
|
const runAgentLoop = async (apiKey, accountId, projectId) => {
|
|
2139
1945
|
lastResolvedAccessToken = apiKey;
|
|
2140
1946
|
const modelInfo = getModel(this.model);
|
|
2141
1947
|
const effectiveBaseUrl = this.baseUrl ?? creds.baseUrl;
|
|
2142
1948
|
const generator = agentLoop(loopMessages, {
|
|
1949
|
+
...(streamRules.length > 0 ? { streamRules: { rules: streamRules } } : {}),
|
|
2143
1950
|
provider: this.provider,
|
|
2144
1951
|
model: this.model,
|
|
2145
1952
|
tools: options.disableTools ? [] : this.tools,
|
|
@@ -2187,6 +1994,33 @@ export class AgentSession {
|
|
|
2187
1994
|
// + pre-warm before the first turn. "baseline": current 5-min default.
|
|
2188
1995
|
cacheRetention: this.isSpeedOptimized() ? "long" : "short",
|
|
2189
1996
|
promptCacheKey: this.getPromptCacheKey(),
|
|
1997
|
+
onContextPrepared: (context) => {
|
|
1998
|
+
const report = this.cacheDiagnostics.prepare(context, {
|
|
1999
|
+
provider: this.provider,
|
|
2000
|
+
model: this.model,
|
|
2001
|
+
at: Date.now(),
|
|
2002
|
+
cacheRetention: this.isSpeedOptimized() ? "long" : "short",
|
|
2003
|
+
route: { baseUrl: effectiveBaseUrl, accountId: this.lastAccountId ?? accountId },
|
|
2004
|
+
settings: {
|
|
2005
|
+
thinking: this.planModeRef.current || options.capThinking
|
|
2006
|
+
? clampThinkingForPlanMode(this.thinkingLevel)
|
|
2007
|
+
: this.thinkingLevel,
|
|
2008
|
+
webSearch: !options.disableTools,
|
|
2009
|
+
supportsImages: modelInfo?.supportsImages,
|
|
2010
|
+
promptCacheKey: this.getPromptCacheKey(),
|
|
2011
|
+
},
|
|
2012
|
+
});
|
|
2013
|
+
log("INFO", "cache", "Prepared context", {
|
|
2014
|
+
sessionId: this.sessionId || this.transportSessionId,
|
|
2015
|
+
data: JSON.stringify(report),
|
|
2016
|
+
});
|
|
2017
|
+
if (report.thinkingPrefixRiskBlocks > 0) {
|
|
2018
|
+
log("WARN", "cache", "Possible signed-thinking prefix mismatch; not server verified", {
|
|
2019
|
+
sessionId: this.sessionId || this.transportSessionId,
|
|
2020
|
+
blocks: String(report.thinkingPrefixRiskBlocks),
|
|
2021
|
+
});
|
|
2022
|
+
}
|
|
2023
|
+
},
|
|
2190
2024
|
supportsImages: modelInfo?.supportsImages,
|
|
2191
2025
|
supportsVideo: modelInfo?.supportsVideo,
|
|
2192
2026
|
userAgent,
|
|
@@ -2195,9 +2029,15 @@ export class AgentSession {
|
|
|
2195
2029
|
maxToolResultChars: resolveSessionToolResultCharLimit(this.model, this.provider, accountId),
|
|
2196
2030
|
// Aggregate per-turn budget across parallel tool results (fan-out guard).
|
|
2197
2031
|
maxTurnToolResultChars: resolveSessionTurnToolResultCharLimit(this.model, this.provider, accountId),
|
|
2032
|
+
// Warn when web/MCP output contains instruction-like text (see injection-detect.ts).
|
|
2033
|
+
transformToolResult: flagUntrustedToolResult,
|
|
2198
2034
|
// Self-correction hooks (same as the TUI): loop-break + re-grounding are
|
|
2199
2035
|
// polled mid-loop; the ideal review is polled when the agent would stop.
|
|
2200
2036
|
getSteeringMessages: () => this.getHookSteeringMessages(),
|
|
2037
|
+
onSteeringAvailable: (listener) => {
|
|
2038
|
+
this.steeringListeners.add(listener);
|
|
2039
|
+
return () => this.steeringListeners.delete(listener);
|
|
2040
|
+
},
|
|
2201
2041
|
getFollowUpMessages: () => this.getHookFollowUpMessages(),
|
|
2202
2042
|
onTurnBudgetExhausted: (ctx) => this.shouldExtendTurnBudget(ctx),
|
|
2203
2043
|
// Check authoritative provider usage before every model/tool step.
|
|
@@ -2223,6 +2063,7 @@ export class AgentSession {
|
|
|
2223
2063
|
// retained usage afterwards since it counted the pruned content.
|
|
2224
2064
|
const pruneResult = pruneStaleToolResults(messages);
|
|
2225
2065
|
if (pruneResult.pruned) {
|
|
2066
|
+
this.cacheDiagnostics.noteEdit("tool_prune", pruneResult.freedTokens);
|
|
2226
2067
|
this.providerContext = null;
|
|
2227
2068
|
log("INFO", "compaction", "Pruned stale tool outputs", {
|
|
2228
2069
|
prunedResults: String(pruneResult.prunedResults),
|
|
@@ -2263,6 +2104,9 @@ export class AgentSession {
|
|
|
2263
2104
|
usage,
|
|
2264
2105
|
pendingMessages,
|
|
2265
2106
|
});
|
|
2107
|
+
// An approved plan runs as ONE run, so step boundaries are only
|
|
2108
|
+
// visible here, between model steps — not after the run ends.
|
|
2109
|
+
const planStep = this.observePlanStepProgress(messages, contextWindow, activeTokens);
|
|
2266
2110
|
log("INFO", "compaction", "In-flight compaction decision", {
|
|
2267
2111
|
provider: this.provider,
|
|
2268
2112
|
model: this.model,
|
|
@@ -2270,8 +2114,10 @@ export class AgentSession {
|
|
|
2270
2114
|
contextWindow: String(contextWindow),
|
|
2271
2115
|
activeTokens: String(activeTokens),
|
|
2272
2116
|
triggerLimit: String(policy.targetTokens),
|
|
2117
|
+
...(planStep ? { planStep: `${planStep.compact} (${planStep.reason})` } : {}),
|
|
2273
2118
|
});
|
|
2274
|
-
if (!shouldCompact(messages, contextWindow, policy.threshold, activeTokens, policy.targetTokens)
|
|
2119
|
+
if (!shouldCompact(messages, contextWindow, policy.threshold, activeTokens, policy.targetTokens) &&
|
|
2120
|
+
!planStep?.compact)
|
|
2275
2121
|
return messages;
|
|
2276
2122
|
}
|
|
2277
2123
|
// compact() operates on this.messages, while an earlier transform may
|
|
@@ -2604,7 +2450,7 @@ export class AgentSession {
|
|
|
2604
2450
|
const canonicalPath = await this.sessionManager.resolveCanonicalSession(this.conversationId, this.cwd);
|
|
2605
2451
|
if (!canonicalPath || canonicalPath === this.sessionPath)
|
|
2606
2452
|
return;
|
|
2607
|
-
await this.adoptCompactionCheckpoint(await this.sessionManager.load(canonicalPath));
|
|
2453
|
+
await this.adoptCompactionCheckpoint(await this.sessionManager.load(canonicalPath, { canonical: true }));
|
|
2608
2454
|
}
|
|
2609
2455
|
async persistCompactionCheckpoint(sourceFingerprint, result) {
|
|
2610
2456
|
const parentSessionId = this.sessionId || undefined;
|
|
@@ -2653,6 +2499,99 @@ export class AgentSession {
|
|
|
2653
2499
|
if (this.postTurnCompaction)
|
|
2654
2500
|
await this.postTurnCompaction;
|
|
2655
2501
|
}
|
|
2502
|
+
/**
|
|
2503
|
+
* Advance plan-step bookkeeping over the messages added since the last
|
|
2504
|
+
* observation and, when a plan step was newly completed (`[DONE:n]`), run
|
|
2505
|
+
* the SoL-Pi cost rule. Called between model steps (in-flight) and once
|
|
2506
|
+
* after the run; `scanIndex` and `doneSteps` make each message and each
|
|
2507
|
+
* step count once across both paths. Returns undefined when no step
|
|
2508
|
+
* completed since the last observation.
|
|
2509
|
+
*/
|
|
2510
|
+
observePlanStepProgress(messages, contextWindow, activeTokens) {
|
|
2511
|
+
const st = this.planStepState;
|
|
2512
|
+
const planPath = this.approvedPlanPath;
|
|
2513
|
+
if (st.planPath !== planPath) {
|
|
2514
|
+
st.planPath = planPath;
|
|
2515
|
+
st.doneSteps = new Set();
|
|
2516
|
+
st.requestsInCompletedSteps = 0;
|
|
2517
|
+
st.requestsThisStep = 0;
|
|
2518
|
+
}
|
|
2519
|
+
if (st.scanIndex > messages.length)
|
|
2520
|
+
st.scanIndex = 0;
|
|
2521
|
+
let requests = 0;
|
|
2522
|
+
const newlyDone = [];
|
|
2523
|
+
for (let i = st.scanIndex; i < messages.length; i++) {
|
|
2524
|
+
const msg = messages[i];
|
|
2525
|
+
if (msg?.role !== "assistant")
|
|
2526
|
+
continue;
|
|
2527
|
+
requests++;
|
|
2528
|
+
const text = typeof msg.content === "string"
|
|
2529
|
+
? msg.content
|
|
2530
|
+
: msg.content.map((part) => (part.type === "text" ? part.text : "")).join("\n");
|
|
2531
|
+
for (const step of findCompletedMarkers(text)) {
|
|
2532
|
+
if (!st.doneSteps.has(step))
|
|
2533
|
+
newlyDone.push(step);
|
|
2534
|
+
}
|
|
2535
|
+
}
|
|
2536
|
+
st.scanIndex = messages.length;
|
|
2537
|
+
const contextTokens = activeTokens ?? estimateConversationTokens(messages);
|
|
2538
|
+
if (st.lastContextTokens > 0)
|
|
2539
|
+
st.grownTokens += Math.max(0, contextTokens - st.lastContextTokens);
|
|
2540
|
+
st.lastContextTokens = contextTokens;
|
|
2541
|
+
st.requests += requests;
|
|
2542
|
+
st.requestsThisStep += requests;
|
|
2543
|
+
st.requestsSinceLastCompaction += requests;
|
|
2544
|
+
if (st.writeCostBalance > 0)
|
|
2545
|
+
st.writeCostBalance -= st.savingPerRequest * requests;
|
|
2546
|
+
if (!planPath || newlyDone.length === 0)
|
|
2547
|
+
return undefined;
|
|
2548
|
+
for (const step of newlyDone)
|
|
2549
|
+
st.doneSteps.add(step);
|
|
2550
|
+
st.requestsInCompletedSteps += st.requestsThisStep;
|
|
2551
|
+
st.requestsThisStep = 0;
|
|
2552
|
+
let planText;
|
|
2553
|
+
try {
|
|
2554
|
+
planText = readFileSync(planPath, "utf8");
|
|
2555
|
+
}
|
|
2556
|
+
catch {
|
|
2557
|
+
return undefined;
|
|
2558
|
+
}
|
|
2559
|
+
const totalSteps = extractPlanSteps(planText).length;
|
|
2560
|
+
const decision = decidePlanStepCompaction({
|
|
2561
|
+
contextTokens,
|
|
2562
|
+
keptTailTokens: PLAN_STEP_KEEP_TOKENS,
|
|
2563
|
+
contextWindow,
|
|
2564
|
+
// The model registry carries no cache price fields; use the SoL-Pi default.
|
|
2565
|
+
cacheWriteReadRatio: DEFAULT_CACHE_WRITE_READ_RATIO,
|
|
2566
|
+
stepsCompleted: st.doneSteps.size,
|
|
2567
|
+
stepsRemaining: Math.max(0, totalSteps - st.doneSteps.size),
|
|
2568
|
+
requestsInCompletedSteps: st.requestsInCompletedSteps,
|
|
2569
|
+
requests: st.requests,
|
|
2570
|
+
grownTokens: st.grownTokens,
|
|
2571
|
+
priorCompactions: st.compactions,
|
|
2572
|
+
writeCostBalance: st.writeCostBalance,
|
|
2573
|
+
requestsSinceLastCompaction: st.compactions > 0 ? st.requestsSinceLastCompaction : undefined,
|
|
2574
|
+
});
|
|
2575
|
+
log("INFO", "compaction", "Plan-step compaction decision", {
|
|
2576
|
+
compact: String(decision.compact),
|
|
2577
|
+
reason: decision.reason,
|
|
2578
|
+
});
|
|
2579
|
+
return decision;
|
|
2580
|
+
}
|
|
2581
|
+
/**
|
|
2582
|
+
* Plan-step bookkeeping after ANY successful compaction (as in the bench:
|
|
2583
|
+
* every compaction leaves a cache-write cost to be repaid by later savings).
|
|
2584
|
+
*/
|
|
2585
|
+
recordPlanStepCompaction(contextBefore) {
|
|
2586
|
+
const st = this.planStepState;
|
|
2587
|
+
const after = estimateConversationTokens(this.messages);
|
|
2588
|
+
st.scanIndex = this.messages.length;
|
|
2589
|
+
st.lastContextTokens = after;
|
|
2590
|
+
st.requestsSinceLastCompaction = 0;
|
|
2591
|
+
st.compactions++;
|
|
2592
|
+
st.writeCostBalance += after * (DEFAULT_CACHE_WRITE_READ_RATIO - 1);
|
|
2593
|
+
st.savingPerRequest = Math.max(0, contextBefore - after);
|
|
2594
|
+
}
|
|
2656
2595
|
/**
|
|
2657
2596
|
* Post-turn compaction: once the final response has been delivered, compact
|
|
2658
2597
|
* in the background while the user reads the answer, instead of making the
|
|
@@ -2661,7 +2600,9 @@ export class AgentSession {
|
|
|
2661
2600
|
* Codex `model_post_turn_compact_threshold_percent` guards: skip when user
|
|
2662
2601
|
* input is already queued (it would race the next turn), when the run was
|
|
2663
2602
|
* aborted, or during the failure cooldown — and never let a compaction
|
|
2664
|
-
* error surface in the completed turn.
|
|
2603
|
+
* error surface in the completed turn. Besides the size trigger, a newly
|
|
2604
|
+
* completed approved-plan step may compact when the cache-cost rule in
|
|
2605
|
+
* compaction/plan-step-policy.ts says the shrink pays for itself.
|
|
2665
2606
|
*/
|
|
2666
2607
|
maybeCompactPostTurn(creds) {
|
|
2667
2608
|
if (!this.settingsManager.get("autoCompact"))
|
|
@@ -2674,15 +2615,18 @@ export class AgentSession {
|
|
|
2674
2615
|
return;
|
|
2675
2616
|
if (Date.now() < this.compactionRetryAfter)
|
|
2676
2617
|
return;
|
|
2677
|
-
// One compaction per turn boundary: a pre-run or overflow-recovery
|
|
2678
|
-
// compaction already shrank this run's history — re-probing right after
|
|
2679
|
-
// the final response would only re-derive that decision.
|
|
2680
|
-
if (this.compactionOccurred)
|
|
2681
|
-
return;
|
|
2682
2618
|
const contextWindow = getContextWindow(this.model, {
|
|
2683
2619
|
provider: this.provider,
|
|
2684
2620
|
accountId: creds.accountId,
|
|
2685
2621
|
});
|
|
2622
|
+
// One compaction per turn boundary: a pre-run, in-flight or overflow
|
|
2623
|
+
// compaction already shrank this run's history — re-probing right after
|
|
2624
|
+
// the final response would only re-derive that decision. Still record the
|
|
2625
|
+
// final response's `[DONE:n]` steps so plan bookkeeping stays current.
|
|
2626
|
+
if (this.compactionOccurred) {
|
|
2627
|
+
this.observePlanStepProgress(this.messages, contextWindow, undefined);
|
|
2628
|
+
return;
|
|
2629
|
+
}
|
|
2686
2630
|
const policy = resolveCompactionPolicy({
|
|
2687
2631
|
provider: this.provider,
|
|
2688
2632
|
model: this.model,
|
|
@@ -2701,9 +2645,12 @@ export class AgentSession {
|
|
|
2701
2645
|
});
|
|
2702
2646
|
}
|
|
2703
2647
|
}
|
|
2704
|
-
|
|
2648
|
+
const planStep = this.observePlanStepProgress(this.messages, contextWindow, activeTokens);
|
|
2649
|
+
if (!shouldCompact(this.messages, contextWindow, policy.threshold, activeTokens, policy.targetTokens) &&
|
|
2650
|
+
!planStep?.compact)
|
|
2705
2651
|
return;
|
|
2706
2652
|
log("INFO", "compaction", "Post-turn compaction decision — compacting in background", {
|
|
2653
|
+
trigger: planStep?.compact ? `plan-step (${planStep.reason})` : "size",
|
|
2707
2654
|
provider: this.provider,
|
|
2708
2655
|
model: this.model,
|
|
2709
2656
|
transport: this.provider === "openai" && creds.accountId ? "codex_oauth" : "public_api",
|
|
@@ -2757,6 +2704,7 @@ export class AgentSession {
|
|
|
2757
2704
|
approvedPlanPath: this.approvedPlanPath,
|
|
2758
2705
|
});
|
|
2759
2706
|
const originalCount = this.messages.length;
|
|
2707
|
+
const contextTokensBefore = estimateConversationTokens(this.messages);
|
|
2760
2708
|
this.eventBus.emit("compaction_start", { messageCount: originalCount });
|
|
2761
2709
|
let contextSelection;
|
|
2762
2710
|
const runCompactor = async () => {
|
|
@@ -2794,7 +2742,7 @@ export class AgentSession {
|
|
|
2794
2742
|
let sourceFingerprint = computeSourceFingerprint(this.messages);
|
|
2795
2743
|
const canonicalPath = await this.sessionManager.resolveCanonicalSession(conversationId, this.cwd);
|
|
2796
2744
|
if (canonicalPath && canonicalPath !== this.sessionPath) {
|
|
2797
|
-
const newest = await this.sessionManager.load(canonicalPath);
|
|
2745
|
+
const newest = await this.sessionManager.load(canonicalPath, { canonical: true });
|
|
2798
2746
|
if (newest.header.sourceFingerprint === sourceFingerprint) {
|
|
2799
2747
|
await this.adoptCompactionCheckpoint(newest);
|
|
2800
2748
|
this.lastCompactionCompacted = true;
|
|
@@ -2868,6 +2816,10 @@ export class AgentSession {
|
|
|
2868
2816
|
}
|
|
2869
2817
|
});
|
|
2870
2818
|
}
|
|
2819
|
+
if (this.lastCompactionCompacted) {
|
|
2820
|
+
this.cacheDiagnostics.noteEdit("compaction");
|
|
2821
|
+
this.recordPlanStepCompaction(contextTokensBefore);
|
|
2822
|
+
}
|
|
2871
2823
|
this.eventBus.emit("compaction_end", {
|
|
2872
2824
|
compacted: this.lastCompactionCompacted,
|
|
2873
2825
|
originalCount,
|
|
@@ -2887,6 +2839,7 @@ export class AgentSession {
|
|
|
2887
2839
|
});
|
|
2888
2840
|
}
|
|
2889
2841
|
async newSession(preserveConversation = false) {
|
|
2842
|
+
this.cacheDiagnostics.reset();
|
|
2890
2843
|
// Approved-plan execution is a clean checkpoint of the same conversation;
|
|
2891
2844
|
// explicit new sessions reset the conversation identity.
|
|
2892
2845
|
if (!preserveConversation) {
|
|
@@ -2934,6 +2887,7 @@ export class AgentSession {
|
|
|
2934
2887
|
}
|
|
2935
2888
|
async loadSession(sessionPath) {
|
|
2936
2889
|
await this.loadExistingSession(sessionPath);
|
|
2890
|
+
this.cacheDiagnostics.reset();
|
|
2937
2891
|
if (this.sessionId)
|
|
2938
2892
|
await this.subAgentManager?.hydrate(this.sessionId);
|
|
2939
2893
|
this.eventBus.emit("session_start", { sessionId: this.sessionId });
|
|
@@ -2964,6 +2918,7 @@ export class AgentSession {
|
|
|
2964
2918
|
const branchMessages = this.sessionManager.getMessages(loaded.entries, this.currentLeafId);
|
|
2965
2919
|
const systemMsg = this.messages[0];
|
|
2966
2920
|
this.messages = [systemMsg, ...branchMessages];
|
|
2921
|
+
this.cacheDiagnostics.reset();
|
|
2967
2922
|
this.lastPersistedIndex = this.messages.length;
|
|
2968
2923
|
// Reads made in the dropped messages are no longer in the model's context.
|
|
2969
2924
|
this.clearReadTracker?.();
|
|
@@ -3031,23 +2986,33 @@ export class AgentSession {
|
|
|
3031
2986
|
: undefined;
|
|
3032
2987
|
return costUsd === undefined ? { used, size } : { used, size, costUsd };
|
|
3033
2988
|
}
|
|
3034
|
-
getPlanMode() {
|
|
3035
|
-
return this.planModeRef.current;
|
|
3036
|
-
}
|
|
3037
2989
|
/**
|
|
3038
|
-
*
|
|
3039
|
-
*
|
|
3040
|
-
*
|
|
2990
|
+
* Whether the provider's prompt cache has likely lapsed since the last
|
|
2991
|
+
* successful request, and how many tokens the next message would re-read at
|
|
2992
|
+
* full price. Null when the route has no known TTL or the chat is empty.
|
|
3041
2993
|
*/
|
|
3042
|
-
|
|
3043
|
-
|
|
3044
|
-
|
|
3045
|
-
|
|
3046
|
-
|
|
3047
|
-
|
|
3048
|
-
|
|
3049
|
-
|
|
3050
|
-
|
|
2994
|
+
getCacheExpiryStatus(now = Date.now()) {
|
|
2995
|
+
const status = assessCacheExpiry({
|
|
2996
|
+
current: {
|
|
2997
|
+
provider: this.provider,
|
|
2998
|
+
model: this.model,
|
|
2999
|
+
policy: resolveCacheTtl({
|
|
3000
|
+
provider: this.provider,
|
|
3001
|
+
model: this.model,
|
|
3002
|
+
cacheRetention: this.isSpeedOptimized() ? "long" : "short",
|
|
3003
|
+
baseUrl: this.baseUrl,
|
|
3004
|
+
accountId: this.lastAccountId,
|
|
3005
|
+
}),
|
|
3006
|
+
},
|
|
3007
|
+
lastTouch: this.cacheDiagnostics.lastCacheTouch(),
|
|
3008
|
+
now,
|
|
3009
|
+
prefixTokens: this.getContextUsage().used,
|
|
3010
|
+
hasHistory: this.messages.some((m) => m.role === "user"),
|
|
3011
|
+
});
|
|
3012
|
+
return status && { ...status, sessionId: this.sessionId || this.transportSessionId };
|
|
3013
|
+
}
|
|
3014
|
+
getPlanMode() {
|
|
3015
|
+
return this.planModeRef.current;
|
|
3051
3016
|
}
|
|
3052
3017
|
/** Queue a user message (optionally with attachments) to be injected mid-run
|
|
3053
3018
|
* as steering. Returns the new queue length. No-op semantics are the caller's
|
|
@@ -3055,6 +3020,9 @@ export class AgentSession {
|
|
|
3055
3020
|
queueMessage(text, attachments = []) {
|
|
3056
3021
|
this.queueSeq += 1;
|
|
3057
3022
|
this.userQueue.push({ id: `q${this.queueSeq}`, text, attachments });
|
|
3023
|
+
// Instant interrupt: preempt running tools so the steer lands right away.
|
|
3024
|
+
for (const listener of [...this.steeringListeners])
|
|
3025
|
+
listener();
|
|
3058
3026
|
return this.userQueue.length;
|
|
3059
3027
|
}
|
|
3060
3028
|
/** Pending queued messages (id + text), oldest first, for client display. */
|
|
@@ -3107,6 +3075,15 @@ export class AgentSession {
|
|
|
3107
3075
|
return `No background process with id "${id}"`;
|
|
3108
3076
|
return this.processManager.stop(id);
|
|
3109
3077
|
}
|
|
3078
|
+
/**
|
|
3079
|
+
* Force-stop every background process tree, synchronously. Background
|
|
3080
|
+
* commands run in their own process group, so the daemon's group kill on
|
|
3081
|
+
* quit never reaches them: this is the only thing that does. Callers on a
|
|
3082
|
+
* shutdown deadline run it before awaiting anything that can hang.
|
|
3083
|
+
*/
|
|
3084
|
+
stopBackgroundProcesses() {
|
|
3085
|
+
this.processManager?.shutdownAll();
|
|
3086
|
+
}
|
|
3110
3087
|
/** Replace a host-owned system prompt in place without resetting conversation history. */
|
|
3111
3088
|
setCustomSystemPrompt(systemPrompt, promptCacheKeyPrefix) {
|
|
3112
3089
|
this.customSystemPrompt = systemPrompt;
|
|
@@ -3253,6 +3230,18 @@ export class AgentSession {
|
|
|
3253
3230
|
* the standard prompt therefore needs a rebuild. Custom and sub-agent
|
|
3254
3231
|
* prompts never render packs, so detection is skipped for them.
|
|
3255
3232
|
*/
|
|
3233
|
+
/**
|
|
3234
|
+
* Bring the system prompt to the exact state the next provider request will
|
|
3235
|
+
* send. Shared by real runs and {@link prewarm} so both see one prefix.
|
|
3236
|
+
*/
|
|
3237
|
+
async prepareSystemPromptForRequest() {
|
|
3238
|
+
// Languages are re-detected at each task boundary so a project scaffolded
|
|
3239
|
+
// during the previous turn gets its packs; the prompt is rebuilt only when
|
|
3240
|
+
// the set grows, keeping the cached prefix stable otherwise.
|
|
3241
|
+
if (this.refreshActiveLanguages())
|
|
3242
|
+
await this.rebuildSystemPromptInPlace();
|
|
3243
|
+
this.refreshSystemPromptTail();
|
|
3244
|
+
}
|
|
3256
3245
|
refreshActiveLanguages() {
|
|
3257
3246
|
if (this.customSystemPrompt || this.agentPrompt !== undefined)
|
|
3258
3247
|
return false;
|
|
@@ -3351,6 +3340,17 @@ export class AgentSession {
|
|
|
3351
3340
|
}));
|
|
3352
3341
|
}
|
|
3353
3342
|
async persistTurnMetric(event) {
|
|
3343
|
+
if (event.stopReason === "error")
|
|
3344
|
+
this.cacheDiagnostics.discardAttempt();
|
|
3345
|
+
const cache = this.cacheDiagnostics.complete(event.usage, event.timing);
|
|
3346
|
+
if (cache) {
|
|
3347
|
+
log("INFO", "cache", "Context cache outcome", {
|
|
3348
|
+
sessionId: this.sessionId || this.transportSessionId,
|
|
3349
|
+
provider: this.provider,
|
|
3350
|
+
model: this.model,
|
|
3351
|
+
data: JSON.stringify(cache),
|
|
3352
|
+
});
|
|
3353
|
+
}
|
|
3354
3354
|
const payload = {
|
|
3355
3355
|
version: 1,
|
|
3356
3356
|
turn: event.turn,
|
|
@@ -3736,6 +3736,138 @@ export class AgentSession {
|
|
|
3736
3736
|
if (signal?.aborted)
|
|
3737
3737
|
this.managerAbortHandler();
|
|
3738
3738
|
}
|
|
3739
|
+
runLoopDepth = 0;
|
|
3740
|
+
lastRealRequestAt = 0;
|
|
3741
|
+
lastPrewarmAt = 0;
|
|
3742
|
+
prewarmController = null;
|
|
3743
|
+
/**
|
|
3744
|
+
* Best-effort Anthropic prompt-cache prewarm before the user's next turn
|
|
3745
|
+
* (desktop app calls this on the first keystroke after opening a chat or an
|
|
3746
|
+
* idle pause). Sends the exact request prefix the next real turn will use —
|
|
3747
|
+
* same system/tools/thinking/cache options — with `max_tokens: 1`, so the
|
|
3748
|
+
* first real reply is a cache read instead of a cold write.
|
|
3749
|
+
*/
|
|
3750
|
+
async prewarm(signal) {
|
|
3751
|
+
if (this.provider !== "anthropic")
|
|
3752
|
+
return { ok: false, reason: "provider" };
|
|
3753
|
+
if (this.settingsManager?.get("cachePrewarm") === false) {
|
|
3754
|
+
return { ok: false, reason: "disabled" };
|
|
3755
|
+
}
|
|
3756
|
+
if (this.runLoopDepth > 0)
|
|
3757
|
+
return { ok: false, reason: "run_active" };
|
|
3758
|
+
if (this.prewarmController)
|
|
3759
|
+
return { ok: false, reason: "in_flight" };
|
|
3760
|
+
const cacheRetention = this.isSpeedOptimized() ? "long" : "short";
|
|
3761
|
+
const ttlMs = (cacheRetention === "long" ? 60 : 5) * 60_000;
|
|
3762
|
+
const now = Date.now();
|
|
3763
|
+
if (now - Math.max(this.lastPrewarmAt, this.lastRealRequestAt) < ttlMs) {
|
|
3764
|
+
return { ok: false, reason: "cache_fresh" };
|
|
3765
|
+
}
|
|
3766
|
+
// Same preparation as the real run, BEFORE copying history: otherwise the
|
|
3767
|
+
// warmed system block differs from the next request (language packs are
|
|
3768
|
+
// detected at run start) and everything after it misses the cache.
|
|
3769
|
+
await this.prepareSystemPromptForRequest();
|
|
3770
|
+
// The await above yields: re-check that no run or other prewarm began.
|
|
3771
|
+
if (this.runLoopDepth > 0)
|
|
3772
|
+
return { ok: false, reason: "run_active" };
|
|
3773
|
+
if (this.prewarmController)
|
|
3774
|
+
return { ok: false, reason: "in_flight" };
|
|
3775
|
+
// End at the last user/tool message: the next real turn appends a new user
|
|
3776
|
+
// message after the trailing assistant reply, and its cache lookback hits
|
|
3777
|
+
// the entry written at this boundary. A request must end on a user turn.
|
|
3778
|
+
let end = this.messages.length;
|
|
3779
|
+
while (end > 0 && this.messages[end - 1]?.role === "assistant")
|
|
3780
|
+
end--;
|
|
3781
|
+
// The loop repairs tool pairing in place before every request; apply the
|
|
3782
|
+
// same repair to a copy so the warmed prefix is byte-identical to it.
|
|
3783
|
+
const messages = structuredClone(this.messages.slice(0, end));
|
|
3784
|
+
repairToolPairingAdjacent(messages);
|
|
3785
|
+
if (!messages.some((m) => m.role === "user" || m.role === "tool")) {
|
|
3786
|
+
return { ok: false, reason: "no_history" };
|
|
3787
|
+
}
|
|
3788
|
+
const tokens = estimateConversationTokens(messages);
|
|
3789
|
+
if (tokens < 4_000)
|
|
3790
|
+
return { ok: false, reason: "too_small" };
|
|
3791
|
+
const controller = new AbortController();
|
|
3792
|
+
const onAbort = () => controller.abort();
|
|
3793
|
+
signal?.addEventListener("abort", onAbort, { once: true });
|
|
3794
|
+
this.prewarmController = controller;
|
|
3795
|
+
const started = Date.now();
|
|
3796
|
+
try {
|
|
3797
|
+
const creds = await this.authStorage.resolveCredentials(this.provider, {
|
|
3798
|
+
storageKeys: this.currentAuthStorageKeys(),
|
|
3799
|
+
});
|
|
3800
|
+
if (controller.signal.aborted || this.runLoopDepth > 0) {
|
|
3801
|
+
return { ok: false, reason: "aborted" };
|
|
3802
|
+
}
|
|
3803
|
+
const modelInfo = getModel(this.model);
|
|
3804
|
+
const result = stream({
|
|
3805
|
+
provider: this.provider,
|
|
3806
|
+
model: this.model,
|
|
3807
|
+
messages,
|
|
3808
|
+
tools: this.tools,
|
|
3809
|
+
webSearch: true,
|
|
3810
|
+
maxTokens: this.maxTokens,
|
|
3811
|
+
thinking: this.planModeRef.current
|
|
3812
|
+
? clampThinkingForPlanMode(this.thinkingLevel)
|
|
3813
|
+
: this.thinkingLevel,
|
|
3814
|
+
apiKey: creds.accessToken,
|
|
3815
|
+
baseUrl: this.baseUrl ?? creds.baseUrl,
|
|
3816
|
+
accountId: creds.accountId,
|
|
3817
|
+
transportSessionId: this.sessionId || this.transportSessionId,
|
|
3818
|
+
cacheRetention,
|
|
3819
|
+
promptCacheKey: this.getPromptCacheKey(),
|
|
3820
|
+
supportsImages: modelInfo?.supportsImages,
|
|
3821
|
+
supportsVideo: modelInfo?.supportsVideo,
|
|
3822
|
+
userAgent: await getClaudeCliUserAgent(),
|
|
3823
|
+
prewarm: true,
|
|
3824
|
+
signal: controller.signal,
|
|
3825
|
+
});
|
|
3826
|
+
const response = await result.response;
|
|
3827
|
+
const usage = response.usage;
|
|
3828
|
+
if (usage.inputTokens === 0 && usage.outputTokens === 0) {
|
|
3829
|
+
// @prestyj/ai sent nothing: budget thinking can't stay identical at max_tokens 1.
|
|
3830
|
+
log("INFO", "prewarm", "Cache prewarm skipped: budget thinking", {
|
|
3831
|
+
model: this.model,
|
|
3832
|
+
});
|
|
3833
|
+
return { ok: false, reason: "thinking_budget_incompatible" };
|
|
3834
|
+
}
|
|
3835
|
+
this.lastPrewarmAt = Date.now();
|
|
3836
|
+
const warmedPolicy = resolveCacheTtl({
|
|
3837
|
+
provider: this.provider,
|
|
3838
|
+
model: this.model,
|
|
3839
|
+
cacheRetention,
|
|
3840
|
+
baseUrl: this.baseUrl ?? creds.baseUrl,
|
|
3841
|
+
accountId: creds.accountId,
|
|
3842
|
+
});
|
|
3843
|
+
if (warmedPolicy) {
|
|
3844
|
+
this.cacheDiagnostics.noteCacheTouch({
|
|
3845
|
+
at: started,
|
|
3846
|
+
provider: this.provider,
|
|
3847
|
+
model: this.model,
|
|
3848
|
+
policy: warmedPolicy,
|
|
3849
|
+
});
|
|
3850
|
+
}
|
|
3851
|
+
log("INFO", "prewarm", "Cache prewarm complete", {
|
|
3852
|
+
tokens: String(tokens),
|
|
3853
|
+
cacheRead: String(usage.cacheRead ?? 0),
|
|
3854
|
+
cacheWrite: String(usage.cacheWrite ?? 0),
|
|
3855
|
+
ms: String(Date.now() - started),
|
|
3856
|
+
});
|
|
3857
|
+
return { ok: true, reason: "warmed", usage };
|
|
3858
|
+
}
|
|
3859
|
+
catch (error) {
|
|
3860
|
+
if (isAbortError(error) || controller.signal.aborted)
|
|
3861
|
+
return { ok: false, reason: "aborted" };
|
|
3862
|
+
log("WARN", "prewarm", `Cache prewarm failed: ${error instanceof Error ? error.message : String(error)}`);
|
|
3863
|
+
return { ok: false, reason: "error" };
|
|
3864
|
+
}
|
|
3865
|
+
finally {
|
|
3866
|
+
signal?.removeEventListener("abort", onAbort);
|
|
3867
|
+
if (this.prewarmController === controller)
|
|
3868
|
+
this.prewarmController = null;
|
|
3869
|
+
}
|
|
3870
|
+
}
|
|
3739
3871
|
/** True when speedProfile is "optimized" (1-h cache TTL + pre-warm), or the
|
|
3740
3872
|
* session was constructed with `forceLongCacheRetention` (Nolan sessions). */
|
|
3741
3873
|
isSpeedOptimized() {
|
|
@@ -3764,16 +3896,22 @@ export class AgentSession {
|
|
|
3764
3896
|
return this.getPromptCacheKey();
|
|
3765
3897
|
}
|
|
3766
3898
|
async dispose() {
|
|
3899
|
+
// First and synchronous: nothing below may delay this, or a hung teardown
|
|
3900
|
+
// step leaves background commands running after the app has quit.
|
|
3901
|
+
this.stopBackgroundProcesses();
|
|
3767
3902
|
// Quiesce any in-flight post-turn compaction BEFORE tearing down state:
|
|
3768
3903
|
// the background compact() snapshots and replaces `this.messages`, so
|
|
3769
3904
|
// letting it run past this point would checkpoint a near-empty history
|
|
3770
3905
|
// and leak a junk session file after teardown.
|
|
3771
3906
|
if (this.postTurnCompaction)
|
|
3772
3907
|
await this.postTurnCompaction;
|
|
3908
|
+
this.cacheDiagnostics.reset();
|
|
3773
3909
|
this.diagnosticsRecorder?.finalize();
|
|
3774
3910
|
this.managerAbortSignal?.removeEventListener("abort", this.managerAbortHandler);
|
|
3775
|
-
|
|
3911
|
+
// Again, in case a turn racing teardown started one while we awaited.
|
|
3912
|
+
this.stopBackgroundProcesses();
|
|
3776
3913
|
this.lspManager?.shutdownAll();
|
|
3914
|
+
this.debugManager?.shutdown();
|
|
3777
3915
|
await Promise.all([this.subAgentManager?.shutdownAll(), this.mcpManager?.dispose()]);
|
|
3778
3916
|
await this.extensionLoader.deactivateAll();
|
|
3779
3917
|
this.setSessionPath("");
|
|
@@ -3809,8 +3947,10 @@ export class AgentSession {
|
|
|
3809
3947
|
// A stale physical checkpoint is only an address, not the conversation tip.
|
|
3810
3948
|
// Resolve every resume—not just over-threshold/deferred compaction resumes—
|
|
3811
3949
|
// before reading history so the next prompt cannot continue an old branch.
|
|
3812
|
-
const
|
|
3813
|
-
const loaded =
|
|
3950
|
+
const resolvedPath = await this.sessionManager.resolveCanonicalSession(sessionPath, this.cwd);
|
|
3951
|
+
const loaded = resolvedPath
|
|
3952
|
+
? await this.sessionManager.load(resolvedPath, { canonical: true })
|
|
3953
|
+
: await this.sessionManager.load(sessionPath);
|
|
3814
3954
|
// Use the leaf from the header to walk the correct branch
|
|
3815
3955
|
const loadedMessages = this.sessionManager.getMessages(loaded.entries, loaded.header.leafId);
|
|
3816
3956
|
const savedCompletionReview = [...loaded.entries]
|