@securecode-ai/mcp 0.5.5 → 0.5.7

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (186) hide show
  1. package/README.md +52 -12
  2. package/dist/api/types.d.ts +7 -0
  3. package/dist/approval/auditLog.d.ts +1 -1
  4. package/dist/approval/auditLog.js +11 -3
  5. package/dist/approval/auditLog.js.map +1 -1
  6. package/dist/approval/broker.d.ts +7 -2
  7. package/dist/approval/broker.js +239 -110
  8. package/dist/approval/broker.js.map +1 -1
  9. package/dist/approval/policy.d.ts +10 -0
  10. package/dist/approval/policy.js +104 -0
  11. package/dist/approval/policy.js.map +1 -0
  12. package/dist/approval/types.d.ts +11 -2
  13. package/dist/approval/types.js +10 -1
  14. package/dist/approval/types.js.map +1 -1
  15. package/dist/attack/agentScanExecutor.d.ts +35 -1
  16. package/dist/attack/agentScanExecutor.js +405 -76
  17. package/dist/attack/agentScanExecutor.js.map +1 -1
  18. package/dist/attack/agentScanLoop.d.ts +20 -0
  19. package/dist/attack/agentScanLoop.js +1264 -60
  20. package/dist/attack/agentScanLoop.js.map +1 -1
  21. package/dist/attack/agentScanProtocol.d.ts +382 -6
  22. package/dist/attack/agentScanProtocol.js +110 -4
  23. package/dist/attack/agentScanProtocol.js.map +1 -1
  24. package/dist/attack/agentTrace.d.ts +73 -0
  25. package/dist/attack/agentTrace.js +228 -0
  26. package/dist/attack/agentTrace.js.map +1 -0
  27. package/dist/attack/architectureScoutExecutor.d.ts +18 -0
  28. package/dist/attack/architectureScoutExecutor.js +211 -0
  29. package/dist/attack/architectureScoutExecutor.js.map +1 -0
  30. package/dist/attack/architectureScoutLoop.d.ts +36 -0
  31. package/dist/attack/architectureScoutLoop.js +403 -0
  32. package/dist/attack/architectureScoutLoop.js.map +1 -0
  33. package/dist/attack/architectureScoutProtocol.d.ts +209 -0
  34. package/dist/attack/architectureScoutProtocol.js +47 -0
  35. package/dist/attack/architectureScoutProtocol.js.map +1 -0
  36. package/dist/attack/candidateStore.d.ts +95 -0
  37. package/dist/attack/candidateStore.js +231 -0
  38. package/dist/attack/candidateStore.js.map +1 -0
  39. package/dist/attack/evidenceLedger.d.ts +67 -0
  40. package/dist/attack/evidenceLedger.js +192 -0
  41. package/dist/attack/evidenceLedger.js.map +1 -0
  42. package/dist/attack/finishGate.d.ts +62 -0
  43. package/dist/attack/finishGate.js +209 -0
  44. package/dist/attack/finishGate.js.map +1 -0
  45. package/dist/attack/fixCodeMerge.d.ts +27 -0
  46. package/dist/attack/fixCodeMerge.js +42 -0
  47. package/dist/attack/fixCodeMerge.js.map +1 -0
  48. package/dist/attack/fixVerifyLoop.d.ts +55 -0
  49. package/dist/attack/fixVerifyLoop.js +187 -0
  50. package/dist/attack/fixVerifyLoop.js.map +1 -0
  51. package/dist/attack/investigationProfiles.d.ts +36 -0
  52. package/dist/attack/investigationProfiles.js +144 -0
  53. package/dist/attack/investigationProfiles.js.map +1 -0
  54. package/dist/attack/investigationState.d.ts +227 -0
  55. package/dist/attack/investigationState.js +666 -0
  56. package/dist/attack/investigationState.js.map +1 -0
  57. package/dist/attack/mutationOperators.d.ts +22 -0
  58. package/dist/attack/mutationOperators.js +170 -0
  59. package/dist/attack/mutationOperators.js.map +1 -0
  60. package/dist/attack/mutationTest.d.ts +33 -0
  61. package/dist/attack/mutationTest.js +110 -0
  62. package/dist/attack/mutationTest.js.map +1 -0
  63. package/dist/attack/proofGate.d.ts +20 -0
  64. package/dist/attack/proofGate.js +104 -0
  65. package/dist/attack/proofGate.js.map +1 -0
  66. package/dist/attack/proofTypes.d.ts +57 -0
  67. package/dist/attack/proofTypes.js +32 -0
  68. package/dist/attack/proofTypes.js.map +1 -0
  69. package/dist/attack/protocolValidator.d.ts +50 -0
  70. package/dist/attack/protocolValidator.js +427 -0
  71. package/dist/attack/protocolValidator.js.map +1 -0
  72. package/dist/attack/qualityMetrics.d.ts +133 -0
  73. package/dist/attack/qualityMetrics.js +226 -0
  74. package/dist/attack/qualityMetrics.js.map +1 -0
  75. package/dist/attack/scanScheduler.d.ts +52 -0
  76. package/dist/attack/scanScheduler.js +265 -0
  77. package/dist/attack/scanScheduler.js.map +1 -0
  78. package/dist/attack/scanState.d.ts +58 -0
  79. package/dist/attack/scanState.js +102 -0
  80. package/dist/attack/scanState.js.map +1 -0
  81. package/dist/attack/searchIntent.d.ts +16 -0
  82. package/dist/attack/searchIntent.js +57 -0
  83. package/dist/attack/searchIntent.js.map +1 -0
  84. package/dist/attack/verifyLoop.d.ts +43 -0
  85. package/dist/attack/verifyLoop.js +314 -14
  86. package/dist/attack/verifyLoop.js.map +1 -1
  87. package/dist/attack/workItem.d.ts +57 -0
  88. package/dist/attack/workItem.js +225 -0
  89. package/dist/attack/workItem.js.map +1 -0
  90. package/dist/audit/findingReviewQueue.d.ts +109 -0
  91. package/dist/audit/findingReviewQueue.js +335 -0
  92. package/dist/audit/findingReviewQueue.js.map +1 -0
  93. package/dist/audit/scanAuditLog.d.ts +160 -0
  94. package/dist/audit/scanAuditLog.js +404 -0
  95. package/dist/audit/scanAuditLog.js.map +1 -0
  96. package/dist/dependency/dependencyChecker.js +35 -9
  97. package/dist/dependency/dependencyChecker.js.map +1 -1
  98. package/dist/dependency/exploitPriority.d.ts +33 -0
  99. package/dist/dependency/exploitPriority.js +78 -0
  100. package/dist/dependency/exploitPriority.js.map +1 -0
  101. package/dist/dependency/finding.d.ts +16 -0
  102. package/dist/dependency/osvClient.js +2 -0
  103. package/dist/dependency/osvClient.js.map +1 -1
  104. package/dist/dependency/types.d.ts +12 -1
  105. package/dist/mcp/server.js +13 -2
  106. package/dist/mcp/server.js.map +1 -1
  107. package/dist/mcp/tools.d.ts +1 -0
  108. package/dist/mcp/tools.js +127 -6
  109. package/dist/mcp/tools.js.map +1 -1
  110. package/dist/project-map/agentMemory.d.ts +83 -1
  111. package/dist/project-map/agentMemory.js +197 -7
  112. package/dist/project-map/agentMemory.js.map +1 -1
  113. package/dist/project-map/architectureContext.d.ts +202 -0
  114. package/dist/project-map/architectureContext.js +421 -0
  115. package/dist/project-map/architectureContext.js.map +1 -0
  116. package/dist/project-map/blastRadius.d.ts +54 -0
  117. package/dist/project-map/blastRadius.js +196 -0
  118. package/dist/project-map/blastRadius.js.map +1 -0
  119. package/dist/project-map/callGraphExtractor.d.ts +13 -0
  120. package/dist/project-map/callGraphExtractor.js +222 -0
  121. package/dist/project-map/callGraphExtractor.js.map +1 -0
  122. package/dist/project-map/capabilityRegistry.d.ts +46 -0
  123. package/dist/project-map/capabilityRegistry.js +91 -0
  124. package/dist/project-map/capabilityRegistry.js.map +1 -0
  125. package/dist/project-map/findTests.d.ts +29 -0
  126. package/dist/project-map/findTests.js +236 -0
  127. package/dist/project-map/findTests.js.map +1 -0
  128. package/dist/project-map/handlerInventory.d.ts +112 -0
  129. package/dist/project-map/handlerInventory.js +184 -0
  130. package/dist/project-map/handlerInventory.js.map +1 -0
  131. package/dist/project-map/implementationResolver.d.ts +43 -0
  132. package/dist/project-map/implementationResolver.js +217 -0
  133. package/dist/project-map/implementationResolver.js.map +1 -0
  134. package/dist/project-map/mapContext.d.ts +6 -1
  135. package/dist/project-map/mapContext.js +1 -0
  136. package/dist/project-map/mapContext.js.map +1 -1
  137. package/dist/project-map/scanCache.d.ts +57 -3
  138. package/dist/project-map/scanCache.js +61 -2
  139. package/dist/project-map/scanCache.js.map +1 -1
  140. package/dist/project-map/symbolIndex.d.ts +39 -0
  141. package/dist/project-map/symbolIndex.js +385 -0
  142. package/dist/project-map/symbolIndex.js.map +1 -0
  143. package/dist/tooling/agentEvalScoring.d.ts +77 -0
  144. package/dist/tooling/agentEvalScoring.js +140 -0
  145. package/dist/tooling/agentEvalScoring.js.map +1 -0
  146. package/dist/tooling/agentRegression.d.ts +53 -0
  147. package/dist/tooling/agentRegression.js +99 -0
  148. package/dist/tooling/agentRegression.js.map +1 -0
  149. package/dist/tools/agentScan.d.ts +69 -0
  150. package/dist/tools/agentScan.js +702 -43
  151. package/dist/tools/agentScan.js.map +1 -1
  152. package/dist/tools/findingReviewTools.d.ts +22 -0
  153. package/dist/tools/findingReviewTools.js +121 -0
  154. package/dist/tools/findingReviewTools.js.map +1 -0
  155. package/dist/tools/fix.js +1 -1
  156. package/dist/tools/fix.js.map +1 -1
  157. package/dist/tools/map.js +191 -1
  158. package/dist/tools/map.js.map +1 -1
  159. package/dist/tools/runTests.d.ts +2 -0
  160. package/dist/tools/runTests.js +29 -0
  161. package/dist/tools/runTests.js.map +1 -0
  162. package/dist/utils/effectMock.d.ts +1 -0
  163. package/dist/utils/effectMock.js +162 -0
  164. package/dist/utils/effectMock.js.map +1 -0
  165. package/dist/utils/gitContext.d.ts +26 -0
  166. package/dist/utils/gitContext.js +317 -0
  167. package/dist/utils/gitContext.js.map +1 -0
  168. package/dist/utils/localTestRunner.d.ts +57 -2
  169. package/dist/utils/localTestRunner.js +122 -115
  170. package/dist/utils/localTestRunner.js.map +1 -1
  171. package/dist/utils/securityConfig.d.ts +17 -0
  172. package/dist/utils/securityConfig.js +192 -0
  173. package/dist/utils/securityConfig.js.map +1 -0
  174. package/dist/utils/testCommandPolicy.d.ts +36 -0
  175. package/dist/utils/testCommandPolicy.js +252 -0
  176. package/dist/utils/testCommandPolicy.js.map +1 -0
  177. package/dist/utils/testRunner.d.ts +56 -0
  178. package/dist/utils/testRunner.js +246 -0
  179. package/dist/utils/testRunner.js.map +1 -0
  180. package/dist/utils/testSafety.d.ts +40 -2
  181. package/dist/utils/testSafety.js +205 -26
  182. package/dist/utils/testSafety.js.map +1 -1
  183. package/dist/utils/verificationSandbox.d.ts +113 -0
  184. package/dist/utils/verificationSandbox.js +656 -0
  185. package/dist/utils/verificationSandbox.js.map +1 -0
  186. package/package.json +1 -1
@@ -19,93 +19,374 @@
19
19
  */
20
20
  Object.defineProperty(exports, "__esModule", { value: true });
21
21
  exports.runAgentScan = runAgentScan;
22
+ exports.isCoherentText = isCoherentText;
23
+ exports.estimateTranscriptSize = estimateTranscriptSize;
24
+ exports.compactTranscript = compactTranscript;
25
+ exports.compactTranscriptAggressive = compactTranscriptAggressive;
26
+ exports.sanitizeFindings = sanitizeFindings;
22
27
  const client_1 = require("../api/client");
23
28
  const agentScanExecutor_1 = require("./agentScanExecutor");
29
+ const agentTrace_1 = require("./agentTrace");
30
+ const investigationState_1 = require("./investigationState");
31
+ const investigationProfiles_1 = require("./investigationProfiles");
32
+ const architectureContext_1 = require("../project-map/architectureContext");
33
+ const endpointDiscovery_1 = require("../project-map/endpointDiscovery");
34
+ const implementationResolver_1 = require("../project-map/implementationResolver");
35
+ const searchIntent_1 = require("./searchIntent");
36
+ const scanState_1 = require("./scanState");
37
+ const evidenceLedger_1 = require("./evidenceLedger");
38
+ const workItem_1 = require("./workItem");
39
+ const handlerInventory_1 = require("../project-map/handlerInventory");
40
+ const candidateStore_1 = require("./candidateStore");
41
+ const scanScheduler_1 = require("./scanScheduler");
42
+ const finishGate_1 = require("./finishGate");
43
+ const qualityMetrics_1 = require("./qualityMetrics");
44
+ const agentScanExecutor_2 = require("./agentScanExecutor");
45
+ const protocolValidator_1 = require("./protocolValidator");
24
46
  const agentScanProtocol_1 = require("./agentScanProtocol");
25
47
  async function runAgentScan(ctx, target, options = {}) {
26
48
  const client = new client_1.ApiClient({ baseUrl: ctx.apiUrl, token: ctx.apiToken });
27
49
  const budget = {
28
- stepsRemaining: options.budget?.stepsRemaining ?? agentScanProtocol_1.AGENT_SCAN_DEFAULTS.maxSteps,
50
+ stepsRemaining: options.budget?.stepsRemaining ?? agentScanProtocol_1.AGENT_SCAN_DEFAULTS.initialSteps,
29
51
  costSpentUsd: options.budget?.costSpentUsd ?? 0,
30
52
  costCapUsd: options.budget?.costCapUsd ?? agentScanProtocol_1.AGENT_SCAN_DEFAULTS.costCapUsd,
53
+ stepsGranted: options.budget?.stepsGranted ?? agentScanProtocol_1.AGENT_SCAN_DEFAULTS.initialSteps,
54
+ hardMaxSteps: options.budget?.hardMaxSteps ?? agentScanProtocol_1.AGENT_SCAN_DEFAULTS.hardMaxSteps,
55
+ extensionsGranted: options.budget?.extensionsGranted ?? 0,
31
56
  };
32
57
  const startTime = Date.now();
33
58
  const wallClockMs = agentScanProtocol_1.AGENT_SCAN_DEFAULTS.wallClockMs;
34
59
  let stepsTaken = 0;
35
60
  let costSpentUsd = 0;
61
+ let stepsGranted = budget.stepsGranted;
62
+ let extensionsGranted = budget.extensionsGranted;
63
+ let meaningfulProgressSinceLastExtension = false;
36
64
  try {
37
- const startResp = await client.postJson('/agent/scan/start', {}, options.signal);
65
+ const startRespRaw = await client.postJson('/agent/scan/start', {}, options.signal);
66
+ const startValidation = (0, protocolValidator_1.validateStartResponse)(startRespRaw);
67
+ if (!startValidation.ok) {
68
+ return {
69
+ status: 'spawn_failed',
70
+ findings: [],
71
+ transcript: [],
72
+ stepsUsed: 0,
73
+ stepsGranted: agentScanProtocol_1.AGENT_SCAN_DEFAULTS.initialSteps,
74
+ extensionsGranted: 0,
75
+ costSpentUsd: 0,
76
+ terminationReason: 'api_error',
77
+ error: `API returned an invalid start response: ${startValidation.error}`,
78
+ investigationNotes: [],
79
+ coverageGaps: [],
80
+ };
81
+ }
82
+ const startResp = startValidation.value;
83
+ const trace = new agentTrace_1.AgentTraceLogger(ctx.workspaceRoot, startResp.runId);
84
+ trace.logRunStarted();
85
+ // Authoritative scan state — the single source of truth for phase,
86
+ // budget, recovery, and lifecycle. Local counters below are kept in
87
+ // sync with this state until they are fully replaced.
88
+ const scanState = (0, scanState_1.createScanRunState)(startResp.runId, target, budget);
89
+ scanState.budget.wallClockMs = wallClockMs;
90
+ const qualityTracker = new qualityMetrics_1.QualityMetricsTracker();
38
91
  const transcript = [];
92
+ const investigationState = new investigationState_1.InvestigationState();
93
+ // Select a target-specific investigation profile to set the required
94
+ // checklist steps. This prevents the agent from wasting steps on
95
+ // irrelevant tools (e.g., get_endpoints on a utility file) and ensures
96
+ // required steps for the target type are not skipped.
97
+ const profile = (0, investigationProfiles_1.selectInvestigationProfile)({
98
+ filePath: target.filePath,
99
+ architectureContext: target.architectureContext,
100
+ endpointContext: target.endpointContext,
101
+ });
102
+ investigationState.setRequiredSteps(profile.requiredSteps);
103
+ scanState.profileId = profile.name;
104
+ // Implementation resolution: if the target file looks like a
105
+ // contract-only file (interface, type declaration, abstract class),
106
+ // resolve the actual implementation so the investigation can follow
107
+ // the real code, not just signatures.
108
+ let implementationResolution = null;
109
+ try {
110
+ const fs = require('fs');
111
+ const absPath = require('path').resolve(ctx.workspaceRoot, target.filePath);
112
+ const content = fs.existsSync(absPath) ? fs.readFileSync(absPath, 'utf8') : '';
113
+ if (content && (0, implementationResolver_1.looksLikeContractOnly)(content)) {
114
+ const symbol = target.filePath.replace(/\.ts$/, '').replace(/.*\//, '');
115
+ const resolution = await (0, implementationResolver_1.resolveImplementation)(ctx.workspaceRoot, target.filePath, symbol);
116
+ if (!resolution.unresolved && resolution.implementationLocations.length > 0) {
117
+ implementationResolution = (0, implementationResolver_1.formatImplementationResolution)(resolution);
118
+ target.implementationHint = implementationResolution;
119
+ }
120
+ }
121
+ }
122
+ catch { /* best-effort */ }
123
+ // Control plane infrastructure — evidence ledger, work items, handler
124
+ // inventory, candidate store, and scheduler work together to ensure
125
+ // the investigation is complete before finish is accepted.
126
+ const evidenceLedger = new evidenceLedger_1.EvidenceLedger();
127
+ evidenceLedger.addRequirements(profile.requirements);
128
+ const workItemQueue = new workItem_1.WorkItemQueue();
129
+ // Note: profile requirements are tracked by the evidence ledger and
130
+ // checklist, not as work items. Work items are for architecture-risk
131
+ // tasks, handler reviews, and implementation reviews — structured
132
+ // work that the scheduler can prioritize and attempt-count.
133
+ const handlerInventory = new handlerInventory_1.HandlerInventory();
134
+ const candidateStore = new candidateStore_1.CandidateStore();
135
+ // Convert architecture risks relevant to this target into tracked
136
+ // investigation tasks. Unresolved tasks will appear as coverage gaps.
137
+ const archTasks = (0, architectureContext_1.createInvestigationTasksFromRisks)(target.architectureContext, target.filePath);
138
+ investigationState.addInvestigationTasks(archTasks);
139
+ // Create work items for architecture-risk tasks so the scheduler
140
+ // can prioritize them and suggest deterministic actions.
141
+ for (const task of archTasks) {
142
+ const requirements = [];
143
+ if (task.requiredProofDimensions?.includes('source') || task.requiredProofDimensions?.includes('reachability')) {
144
+ requirements.push({
145
+ id: `${task.id}-req-source`,
146
+ description: 'Trace the code path from input to the sensitive operation',
147
+ acceptedKinds: ['source-range', 'cross-file-flow'],
148
+ targetFiles: task.targetFiles,
149
+ requiredTools: ['read_file', 'trace_flow_cross_file', 'trace_flow'],
150
+ minimumCount: 1,
151
+ });
152
+ }
153
+ if (task.requiredProofDimensions?.includes('control')) {
154
+ requirements.push({
155
+ id: `${task.id}-req-control`,
156
+ description: 'Determine whether the control is present, bypassed, or missing',
157
+ acceptedKinds: ['guard-result', 'policy-result', 'config-result'],
158
+ targetFiles: task.targetFiles,
159
+ requiredTools: ['check_guard', 'check_policy', 'read_config'],
160
+ minimumCount: 1,
161
+ acceptsNegative: true,
162
+ });
163
+ }
164
+ if (task.requiredProofDimensions?.includes('threat-model')) {
165
+ requirements.push({
166
+ id: `${task.id}-req-threat-model`,
167
+ description: 'Establish the threat model (is the attacker in scope?)',
168
+ acceptedKinds: ['threat-model-result', 'config-result', 'source-range'],
169
+ targetFiles: task.targetFiles,
170
+ requiredTools: ['read_config', 'read_file', 'search_code'],
171
+ minimumCount: 1,
172
+ acceptsNegative: true,
173
+ });
174
+ }
175
+ if (task.requiredProofDimensions?.includes('impact')) {
176
+ requirements.push({
177
+ id: `${task.id}-req-impact`,
178
+ description: 'Verify the impact: trace to the actual sensitive sink and confirm reachability',
179
+ acceptedKinds: ['cross-file-flow', 'capability-result', 'source-range'],
180
+ targetFiles: task.targetFiles,
181
+ requiredTools: ['trace_flow_cross_file', 'trace_flow'],
182
+ minimumCount: 1,
183
+ });
184
+ }
185
+ if (task.requiredProofDimensions?.includes('verification')) {
186
+ requirements.push({
187
+ id: `${task.id}-req-verification`,
188
+ description: 'Verify the vulnerability with a test or exploit proof',
189
+ acceptedKinds: ['test-location', 'test-result', 'proof-result'],
190
+ targetFiles: task.targetFiles,
191
+ requiredTools: ['find_tests', 'run_tests'],
192
+ minimumCount: 1,
193
+ acceptsNegative: true,
194
+ });
195
+ }
196
+ if (requirements.length === 0) {
197
+ requirements.push({
198
+ id: `${task.id}-req-0`,
199
+ description: task.requiredEvidence[0] || 'Investigate the risk',
200
+ acceptedKinds: ['source-range', 'cross-file-flow', 'policy-result'],
201
+ minimumCount: 1,
202
+ });
203
+ }
204
+ workItemQueue.add((0, workItem_1.createArchitectureRiskWorkItem)(task.claim, task.targetFiles, requirements));
205
+ }
39
206
  const readFiles = new Set();
40
207
  const readFileCounts = new Map();
208
+ // Pre-compute function boundaries for chunk alignment
209
+ let functionBoundaries;
210
+ try {
211
+ const fb = await (0, agentScanExecutor_2.extractFunctionBoundaries)(target.fileContent, target.filePath);
212
+ if (fb)
213
+ functionBoundaries = fb;
214
+ }
215
+ catch { /* best-effort */ }
41
216
  // Dynamic read cap based on file size:
42
217
  // < 200 lines → 5 reads (small file, 5 chunks is enough)
43
- // < 1000 lines → 10 reads (medium file)
44
- // < 5000 lines → 20 reads (large file, needs many sections)
45
- // >= 5000 lines → 30 reads (very large, allow thorough coverage)
218
+ // < 1000 lines → 8 reads (medium file)
219
+ // < 5000 lines → 12 reads (large file — use search_code/trace_flow, not brute reading)
220
+ // >= 5000 lines → 15 reads (very large — still capped; the agent must use
221
+ // search_code and trace_flow for pattern discovery, not
222
+ // read the entire file section by section)
46
223
  function maxReadsForFile(filePath) {
47
224
  try {
48
225
  const fs = require('fs');
49
226
  const abs = require('path').resolve(ctx.workspaceRoot, filePath);
50
227
  const stat = fs.statSync(abs);
51
228
  if (stat.size > 200_000)
52
- return 30;
229
+ return 15;
53
230
  const content = fs.readFileSync(abs, 'utf8');
54
231
  const lines = content.split('\n').length;
55
232
  if (lines < 200)
56
233
  return 5;
57
234
  if (lines < 1000)
58
- return 10;
235
+ return 8;
59
236
  if (lines < 5000)
60
- return 20;
61
- return 30;
237
+ return 12;
238
+ return 15;
62
239
  }
63
240
  catch {
64
- return 10;
241
+ return 8;
65
242
  }
66
243
  }
67
244
  // Track non-read tool calls to prevent the agent from looping on the
68
245
  // same search_code/trace_flow call repeatedly. Keyed by (type + args).
69
246
  const toolCallCounts = new Map();
70
247
  const MAX_SAME_TOOL_CALL = 2;
248
+ // Track normalized search patterns to detect equivalent searches
249
+ // (same terms, different order) that the raw toolKey dedup misses.
250
+ const searchedPatterns = new Set();
251
+ let equivalentSearchCount = 0;
71
252
  let consecutiveErrors = 0;
72
253
  let consecutiveBlockedReads = 0;
254
+ let meaningfulProgressSinceRecovery = true;
255
+ let aggressiveCompaction = false;
73
256
  while (true) {
74
257
  // Wall clock check
75
258
  if (Date.now() - startTime > wallClockMs) {
259
+ (0, scanState_1.terminateScan)(scanState, 'wall_clock', `Wall clock limit (${wallClockMs}ms) exceeded.`);
76
260
  return {
77
261
  status: 'capped',
78
262
  findings: [],
79
263
  transcript,
80
264
  stepsUsed: stepsTaken,
265
+ stepsGranted,
266
+ extensionsGranted,
81
267
  costSpentUsd,
268
+ terminationReason: 'wall_clock',
82
269
  summary: `Wall clock limit (${wallClockMs}ms) exceeded.`,
270
+ investigationNotes: [],
271
+ coverageGaps: [],
83
272
  };
84
273
  }
85
274
  // Abort check
86
275
  if (options.signal?.aborted) {
276
+ (0, scanState_1.terminateScan)(scanState, 'cancelled', 'Cancelled by user.');
87
277
  return {
88
278
  status: 'cancelled',
89
279
  findings: [],
90
280
  transcript,
91
281
  stepsUsed: stepsTaken,
282
+ stepsGranted,
283
+ extensionsGranted,
92
284
  costSpentUsd,
285
+ terminationReason: 'cancelled',
93
286
  summary: 'Cancelled by user.',
287
+ investigationNotes: [],
288
+ coverageGaps: [],
94
289
  };
95
290
  }
291
+ // Build action constraint for blocked-read recovery.
292
+ // 0-1 blocked reads: normal (no constraint)
293
+ // 2 blocked reads: recovery mode — if unread ranges exist,
294
+ // require the next unread range; if no unread ranges remain,
295
+ // forbid read_file entirely and require an analysis tool.
296
+ // 3 blocked reads: MCP selects a deterministic recovery action (skip API)
297
+ let actionConstraint;
298
+ if (consecutiveBlockedReads >= 2 && consecutiveBlockedReads < 3) {
299
+ const nextRange = target.fileContent
300
+ ? investigationState.getPrioritizedUnreadRange(target.filePath, target.fileContent)
301
+ : investigationState.getNextUnreadRange(target.filePath);
302
+ if (nextRange) {
303
+ actionConstraint = {
304
+ mode: 'recovery',
305
+ requiredAction: {
306
+ type: 'read_file',
307
+ path: target.filePath,
308
+ startLine: nextRange.start,
309
+ endLine: nextRange.end,
310
+ rationale: `Read next unread range: lines ${nextRange.start}-${nextRange.end}`,
311
+ },
312
+ reason: `You have been blocked by duplicate/overlapping reads. Read the next unread range: lines ${nextRange.start}-${nextRange.end}, or use an analysis tool (trace_flow, check_guard, check_policy).`,
313
+ };
314
+ }
315
+ else {
316
+ const recoveryAction = investigationState.getRecommendedRecoveryAction();
317
+ actionConstraint = {
318
+ mode: 'recovery',
319
+ forbiddenActions: ['read_file'],
320
+ requiredAction: recoveryAction || undefined,
321
+ reason: 'You have been blocked by duplicate/overlapping reads and no unread ranges remain. Switch to an analysis tool.',
322
+ };
323
+ }
324
+ }
325
+ const compactedTranscript = aggressiveCompaction
326
+ ? compactTranscriptAggressive(transcript)
327
+ : compactTranscript(transcript);
96
328
  const stepReq = {
97
329
  runId: startResp.runId,
98
330
  target,
99
- transcript,
331
+ transcript: compactedTranscript,
100
332
  budget: {
101
333
  stepsRemaining: budget.stepsRemaining,
102
334
  costSpentUsd,
103
335
  costCapUsd: budget.costCapUsd,
336
+ stepsGranted,
337
+ hardMaxSteps: budget.hardMaxSteps,
338
+ extensionsGranted,
339
+ },
340
+ clientCapabilities: (0, agentScanProtocol_1.defaultClientCapabilities)(),
341
+ investigationProgress: {
342
+ completedSteps: investigationState.getCompletedSteps(),
343
+ incompleteSteps: investigationState.getIncompleteSteps(),
344
+ consecutiveBlockedReads,
345
+ meaningfulProgressSinceLastExtension,
104
346
  },
347
+ actionConstraint,
105
348
  };
106
349
  let stepResp;
107
350
  try {
108
- stepResp = await client.postJson('/agent/scan/step', stepReq, options.signal);
351
+ const stepRespRaw = await client.postJson('/agent/scan/step', stepReq, options.signal);
352
+ const stepValidation = (0, protocolValidator_1.validateStepResponse)(stepRespRaw);
353
+ if (!stepValidation.ok) {
354
+ // Wire-level malformed response. Don't execute the
355
+ // potentially-dangerous `next` action; treat it as a
356
+ // controlled error so the agent can retry against a
357
+ // well-formed response on the next call.
358
+ const vErr = stepValidation.error;
359
+ console.warn(`[Agent Scan Loop] Step ${stepsTaken + 1} returned a malformed response: ${vErr}`);
360
+ consecutiveErrors++;
361
+ stepsTaken++;
362
+ budget.stepsRemaining--;
363
+ const errMsg = `API returned a malformed step response: ${vErr}`;
364
+ transcript.push({
365
+ action: {
366
+ type: 'system_event',
367
+ eventType: 'error',
368
+ message: errMsg,
369
+ },
370
+ observation: errMsg,
371
+ });
372
+ if (consecutiveErrors >= 3 || budget.stepsRemaining <= 0) {
373
+ return {
374
+ status: 'capped',
375
+ findings: [],
376
+ transcript,
377
+ stepsUsed: stepsTaken,
378
+ stepsGranted,
379
+ extensionsGranted,
380
+ costSpentUsd,
381
+ terminationReason: 'api_error',
382
+ summary: `Step budget exhausted after ${consecutiveErrors} consecutive malformed API responses. Last error: ${vErr}`,
383
+ investigationNotes: [],
384
+ coverageGaps: [],
385
+ };
386
+ }
387
+ continue;
388
+ }
389
+ stepResp = stepValidation.value;
109
390
  }
110
391
  catch (stepErr) {
111
392
  const errMsg = stepErr?.message || String(stepErr);
@@ -119,8 +400,13 @@ async function runAgentScan(ctx, target, options = {}) {
119
400
  findings: [],
120
401
  transcript,
121
402
  stepsUsed: stepsTaken,
403
+ stepsGranted,
404
+ extensionsGranted,
122
405
  costSpentUsd,
406
+ terminationReason: 'api_error',
123
407
  error: 'API server restarted mid-scan — the run was lost. Please retry the scan.',
408
+ investigationNotes: [],
409
+ coverageGaps: [],
124
410
  };
125
411
  }
126
412
  // Detect abort — the user cancelled. Don't treat as error.
@@ -130,26 +416,38 @@ async function runAgentScan(ctx, target, options = {}) {
130
416
  findings: [],
131
417
  transcript,
132
418
  stepsUsed: stepsTaken,
419
+ stepsGranted,
420
+ extensionsGranted,
133
421
  costSpentUsd,
422
+ terminationReason: 'cancelled',
134
423
  summary: 'Cancelled by user.',
424
+ investigationNotes: [],
425
+ coverageGaps: [],
135
426
  };
136
427
  }
428
+ // Detect timeout / network errors — retry once with aggressive compaction
429
+ if (!aggressiveCompaction && /timeout|ETIMEDOUT|ESOCKETTIMEDOUT|socket hang up|ECONNRESET|fetch failed/i.test(errMsg)) {
430
+ console.warn(`[Agent Scan Loop] Step ${stepsTaken + 1} timeout/network error — retrying with aggressive compaction: ${errMsg}`);
431
+ aggressiveCompaction = true;
432
+ continue;
433
+ }
137
434
  // The API rejected the action (malformed, missing field, etc).
138
- // Add the error to the transcript so the LLM sees it on the
139
- // next step and can correct itself. Then retry — don't waste
140
- // the step silently.
435
+ // Add a first-class system_event to the transcript so the LLM
436
+ // sees the error on the next step and can correct itself.
437
+ // Previously this piggybacked on `read_file('__ERROR__')`,
438
+ // which the executor would have tried to open and failed.
141
439
  console.warn(`[Agent Scan Loop] Step ${stepsTaken + 1} error: ${errMsg}`);
142
440
  consecutiveErrors++;
143
441
  stepsTaken++;
144
442
  budget.stepsRemaining--;
145
- // Push the error into the transcript so the LLM sees it
443
+ const errorMsg = `ERROR: Your previous action was rejected: ${errMsg}. Please try a DIFFERENT action with ALL required fields. Set unused fields to null.`;
146
444
  transcript.push({
147
445
  action: {
148
- type: 'read_file',
149
- path: '__ERROR__',
150
- rationale: 'Previous action was invalid',
446
+ type: 'system_event',
447
+ eventType: 'error',
448
+ message: errorMsg,
151
449
  },
152
- observation: `ERROR: Your previous action was rejected: ${errMsg}. Please try a DIFFERENT action with ALL required fields. Set unused fields to null.`,
450
+ observation: errorMsg,
153
451
  });
154
452
  if (consecutiveErrors >= 3 || budget.stepsRemaining <= 0) {
155
453
  return {
@@ -157,74 +455,367 @@ async function runAgentScan(ctx, target, options = {}) {
157
455
  findings: [],
158
456
  transcript,
159
457
  stepsUsed: stepsTaken,
458
+ stepsGranted,
459
+ extensionsGranted,
160
460
  costSpentUsd,
461
+ terminationReason: 'api_error',
161
462
  summary: `Step budget exhausted after ${consecutiveErrors} consecutive API errors. Last error: ${errMsg}`,
463
+ investigationNotes: [],
464
+ coverageGaps: [],
162
465
  };
163
466
  }
164
467
  continue;
165
468
  }
166
469
  costSpentUsd += stepResp.costUsd || 0;
470
+ scanState.budget.costSpentUsd = costSpentUsd;
167
471
  consecutiveErrors = 0;
472
+ trace.logStepRequested(stepResp.model, stepResp.tokens, stepResp.costUsd, stepResp.latencyMs);
473
+ // Transition from planning to surveying on first successful step
474
+ if (scanState.phase === 'planning') {
475
+ (0, scanState_1.transitionScanPhase)(scanState, 'surveying', 'first step received');
476
+ }
477
+ // Budget extension — the API may grant additional steps when the
478
+ // agent demonstrates meaningful progress. Update our local tracking.
479
+ if (stepResp.budgetExtension) {
480
+ const ext = stepResp.budgetExtension;
481
+ stepsGranted = ext.totalGranted;
482
+ extensionsGranted = ext.granted > 0 ? extensionsGranted + 1 : extensionsGranted;
483
+ budget.stepsRemaining = stepResp.stepsRemaining;
484
+ budget.stepsGranted = ext.totalGranted;
485
+ budget.extensionsGranted = extensionsGranted;
486
+ scanState.budget.stepsGranted = ext.totalGranted;
487
+ scanState.budget.extensionsGranted = extensionsGranted;
488
+ scanState.budget.stepsRemaining = stepResp.stepsRemaining;
489
+ meaningfulProgressSinceLastExtension = false;
490
+ if (options.onProgress) {
491
+ options.onProgress(stepsTaken, ext.hardMaxSteps, `Budget extended +${ext.granted} steps (${ext.totalGranted}/${ext.hardMaxSteps}): ${ext.reason}`);
492
+ }
493
+ }
494
+ // System event (e.g., critique from the senior reviewer) — append
495
+ // to the transcript WITHOUT executing it, then continue the loop
496
+ // so the next step's prompt sees the event and re-plans. This
497
+ // replaces the previous `read_file('__CRITIQUE__')` pattern that
498
+ // lost the critique content over the wire.
499
+ if (stepResp.systemEvent) {
500
+ const ev = stepResp.systemEvent;
501
+ if (ev.eventType === 'critique')
502
+ qualityTracker.recordCritique();
503
+ if (ev.eventType === 'finish_gate')
504
+ qualityTracker.recordFinishGate();
505
+ transcript.push({
506
+ action: ev,
507
+ observation: ev.message,
508
+ });
509
+ stepsTaken++;
510
+ budget.stepsRemaining = stepResp.stepsRemaining;
511
+ if (options.onProgress) {
512
+ options.onProgress(stepsTaken, stepsGranted, `Step ${stepsTaken}: system_event(${ev.eventType})`);
513
+ }
514
+ // Re-loop: ask the API for the next action now that the
515
+ // critique is in the transcript. The agent will see the
516
+ // critique in the next step's prompt and either re-investigate
517
+ // or call finish again with an updated selfCritique.
518
+ continue;
519
+ }
168
520
  // Null next = done (cost capped or steps exhausted)
169
521
  if (!stepResp.next) {
170
522
  const status = stepResp.costCapped
171
523
  ? 'capped'
172
524
  : (stepResp.degraded ? 'degraded' : 'completed');
525
+ trace.logRunCompleted(status);
173
526
  return {
174
527
  status,
175
528
  findings: [],
176
529
  transcript,
177
530
  stepsUsed: stepsTaken,
531
+ stepsGranted,
532
+ extensionsGranted,
178
533
  costSpentUsd,
534
+ terminationReason: stepResp.costCapped ? 'cost_cap' : 'budget_exhausted',
179
535
  summary: stepResp.costCapped
180
536
  ? `Cost cap ($${budget.costCapUsd.toFixed(2)}) reached.`
181
537
  : 'Agent completed without explicit finish.',
538
+ investigationNotes: [],
539
+ coverageGaps: [],
182
540
  };
183
541
  }
184
542
  const action = stepResp.next;
185
543
  stepsTaken++;
186
544
  budget.stepsRemaining = stepResp.stepsRemaining;
545
+ scanState.budget.stepsUsed = stepsTaken;
546
+ scanState.budget.stepsRemaining = stepResp.stepsRemaining;
547
+ trace.nextStep();
548
+ trace.logActionSelected(action.type, stepResp.model, stepResp.degraded || stepResp.fallbackFired);
187
549
  // Progress callback
188
550
  if (options.onProgress) {
189
551
  const actionDesc = describeAction(action);
190
- options.onProgress(stepsTaken, agentScanProtocol_1.AGENT_SCAN_DEFAULTS.maxSteps, `Step ${stepsTaken}: ${actionDesc}`);
552
+ options.onProgress(stepsTaken, stepsGranted, `Step ${stepsTaken}: ${actionDesc}`);
191
553
  }
192
554
  // Finish = done with findings
193
555
  if (action.type === 'finish') {
194
- return {
195
- status: 'completed',
196
- findings: action.findings,
197
- transcript,
198
- stepsUsed: stepsTaken,
199
- costSpentUsd,
200
- summary: action.summary,
201
- };
556
+ scanState.finishAttempts++;
557
+ qualityTracker.recordFinishAttempt();
558
+ // Register candidates from findings — each finding becomes a
559
+ // tracked candidate. If the candidate is new (discovered), the
560
+ // finish gate will reject and the agent must gather evidence.
561
+ for (const finding of action.findings) {
562
+ const rootCauseId = finding.rootCause?.rootCauseId || `${finding.type}:${finding.line}`;
563
+ const existing = candidateStore.getCandidatesByRootCause(rootCauseId);
564
+ if (existing.length === 0) {
565
+ const candidateId = candidateStore.register({
566
+ rootCauseId,
567
+ type: finding.type,
568
+ severity: finding.severity,
569
+ locations: [{ filePath: target.filePath, line: finding.line }],
570
+ claim: finding.why,
571
+ requiredEvidence: [
572
+ {
573
+ id: `${rootCauseId}-flow`,
574
+ description: 'Verify data flow to the vulnerable location',
575
+ acceptedKinds: ['cross-file-flow'],
576
+ targetFiles: [target.filePath],
577
+ requiredTools: ['trace_flow_cross_file', 'trace_flow'],
578
+ minimumCount: 1,
579
+ },
580
+ {
581
+ id: `${rootCauseId}-guard`,
582
+ description: 'Check for guards/controls on the vulnerable location',
583
+ acceptedKinds: ['guard-result', 'policy-result'],
584
+ targetFiles: [target.filePath],
585
+ requiredTools: ['check_guard', 'check_policy'],
586
+ minimumCount: 1,
587
+ },
588
+ ],
589
+ });
590
+ // Backfill evidence: scan the transcript for prior
591
+ // verification actions on this candidate's location.
592
+ // If the agent already ran trace_flow/check_guard on
593
+ // the file before calling finish, credit that evidence.
594
+ const targetFileNorm = target.filePath.replace(/\\/g, '/').toLowerCase();
595
+ const verificationActions = new Set([
596
+ 'trace_flow', 'trace_flow_cross_file', 'check_guard', 'check_policy',
597
+ ]);
598
+ for (const step of transcript) {
599
+ if (verificationActions.has(step.action.type)) {
600
+ const stepFile = step.action.filePath || step.action.path || target.filePath;
601
+ const stepFileNorm = String(stepFile).replace(/\\/g, '/').toLowerCase();
602
+ if (stepFileNorm === targetFileNorm) {
603
+ const cat = step.action.type === 'trace_flow' || step.action.type === 'trace_flow_cross_file' ? 'flow'
604
+ : step.action.type === 'check_guard' ? 'guard'
605
+ : step.action.type === 'check_policy' ? 'policy'
606
+ : undefined;
607
+ const dim = step.action.type === 'trace_flow' || step.action.type === 'trace_flow_cross_file' ? 'reachability'
608
+ : step.action.type === 'check_guard' || step.action.type === 'check_policy' ? 'control'
609
+ : undefined;
610
+ candidateStore.addEvidence(candidateId, `backfill:${step.action.type}:${stepFileNorm}`, cat, dim);
611
+ }
612
+ }
613
+ if (step.action.type === 'read_file') {
614
+ const stepFile = step.action.path || target.filePath;
615
+ const stepFileNorm = String(stepFile).replace(/\\/g, '/').toLowerCase();
616
+ if (stepFileNorm === targetFileNorm) {
617
+ candidateStore.addEvidence(candidateId, `backfill:source:${stepFileNorm}`, 'source', 'source');
618
+ }
619
+ }
620
+ if (step.action.type === 'read_config') {
621
+ const stepFile = target.filePath;
622
+ const stepFileNorm = stepFile.replace(/\\/g, '/').toLowerCase();
623
+ candidateStore.addEvidence(candidateId, `backfill:config:${stepFileNorm}`, undefined, 'threat-model');
624
+ }
625
+ }
626
+ }
627
+ }
628
+ // Evaluate the finish proposal through the hard finish gate
629
+ const schedDecision = (0, scanScheduler_1.schedule)({
630
+ state: scanState,
631
+ evidence: evidenceLedger,
632
+ workItems: workItemQueue,
633
+ handlers: handlerInventory,
634
+ candidates: candidateStore,
635
+ investigation: investigationState,
636
+ target,
637
+ functionBoundaries,
638
+ });
639
+ const gateResult = (0, finishGate_1.evaluateFinishGate)({
640
+ proposal: action,
641
+ state: scanState,
642
+ evidence: evidenceLedger,
643
+ workItems: workItemQueue,
644
+ handlers: handlerInventory,
645
+ candidates: candidateStore,
646
+ investigation: investigationState,
647
+ scheduler: schedDecision,
648
+ target: { filePath: target.filePath, fileContent: target.fileContent },
649
+ });
650
+ if (gateResult.accepted) {
651
+ trace.logRunCompleted('completed');
652
+ if (gateResult.mode === 'forced-incomplete') {
653
+ (0, scanState_1.terminateScan)(scanState, 'forced_incomplete', 'Budget exhausted with incomplete investigation');
654
+ qualityTracker.recordForcedTermination('forced_incomplete');
655
+ }
656
+ else {
657
+ (0, scanState_1.terminateScan)(scanState, 'agent_finish', gateResult.normalizedFinish?.summary);
658
+ qualityTracker.recordTermination('agent_finish');
659
+ }
660
+ // Merge gate-generated coverage gaps with model-provided gaps
661
+ let coverageGaps = action.coverageGaps ?? [];
662
+ if (gateResult.coverageGaps) {
663
+ coverageGaps = [...coverageGaps, ...gateResult.coverageGaps];
664
+ }
665
+ const covSummary = investigationState.getCoverageSummary(target.filePath);
666
+ qualityTracker.recordCoverage(covSummary.totalLines, covSummary.coveredLines, covSummary.uncoveredRangeCount, covSummary.largestUncoveredRange);
667
+ qualityTracker.recordBudget(stepsTaken, stepsGranted, extensionsGranted, costSpentUsd, budget.costCapUsd, wallClockMs, Date.now() - startTime);
668
+ const investigationNotes = [...(action.investigationNotes ?? [])];
669
+ for (const candidate of candidateStore.getActive()) {
670
+ if (candidate.status === 'discovered' || candidate.status === 'investigating' || candidate.status === 'supported') {
671
+ const missingDims = candidate.requiredProofDimensions?.filter(d => !candidate.satisfiedDimensions?.includes(d)) || [];
672
+ investigationNotes.push({
673
+ title: candidate.claim,
674
+ detail: `Candidate was not fully verified. Status: ${candidate.status}. Missing proof dimensions: ${missingDims.join(', ') || 'none'}. Evidence refs: ${candidate.evidenceRefs.length}.`,
675
+ file: candidate.locations[0]?.filePath || target.filePath,
676
+ line: candidate.locations[0]?.line,
677
+ verificationLevel: 'logic-confirmed',
678
+ rootCauseId: candidate.rootCauseId,
679
+ requiredEvidence: missingDims.map(d => `Satisfy the ${d} proof dimension`),
680
+ priority: candidate.severity === 'critical' || candidate.severity === 'high' ? 'high' : 'medium',
681
+ });
682
+ candidateStore.setUnproven(candidate.id, `Converted to note: missing proof dimensions ${missingDims.join(', ')}`);
683
+ }
684
+ }
685
+ return {
686
+ status: 'completed',
687
+ findings: sanitizeFindings(action.findings),
688
+ investigationNotes,
689
+ coverageGaps,
690
+ transcript,
691
+ stepsUsed: stepsTaken,
692
+ stepsGranted,
693
+ extensionsGranted,
694
+ costSpentUsd,
695
+ terminationReason: gateResult.mode === 'forced-incomplete' ? 'forced_incomplete' : 'agent_finish',
696
+ summary: action.summary,
697
+ qualityMetrics: qualityTracker.getMetrics(),
698
+ };
699
+ }
700
+ // Finish rejected — continue investigation
701
+ trace.logToolBlocked('finish', `Finish rejected: ${gateResult.reasons.map(r => r.description).join('; ')}`);
702
+ qualityTracker.recordFinishRejection();
703
+ // Add a system event so the model sees the rejection
704
+ const rejectionMessage = `FINISH REJECTED — your investigation is incomplete:\n${gateResult.reasons.map(r => ` - ${r.description}`).join('\n')}\n\nYou must complete the remaining investigation steps before calling finish.${gateResult.recoveryAction ? `\n\nYour next action MUST be: ${gateResult.recoveryAction.type}` : ''}`;
705
+ transcript.push({
706
+ action: {
707
+ type: 'system_event',
708
+ eventType: 'finish_rejected',
709
+ message: rejectionMessage,
710
+ },
711
+ observation: rejectionMessage,
712
+ });
713
+ // Execute the recovery action if the scheduler has one
714
+ if (gateResult.recoveryAction) {
715
+ const recoveryAction = gateResult.recoveryAction;
716
+ stepsTaken++;
717
+ budget.stepsRemaining--;
718
+ scanState.budget.stepsUsed = stepsTaken;
719
+ scanState.budget.stepsRemaining = budget.stepsRemaining;
720
+ trace.nextStep();
721
+ trace.logActionSelected(recoveryAction.type, 'finish-gate-recovery', false);
722
+ if (options.onProgress) {
723
+ options.onProgress(stepsTaken, stepsGranted, `Step ${stepsTaken}: finish-gate recovery: ${describeAction(recoveryAction)}`);
724
+ }
725
+ let recoveryObservation;
726
+ if (recoveryAction.type === 'read_file') {
727
+ const readResult = await (0, agentScanExecutor_1.executeReadFileAction)(recoveryAction, ctx);
728
+ recoveryObservation = readResult.observation;
729
+ if (readResult.totalLines > 0) {
730
+ investigationState.recordActualRead(recoveryAction.path, readResult.actualStart, readResult.actualEnd, readResult.totalLines, readResult.truncated);
731
+ }
732
+ }
733
+ else {
734
+ recoveryObservation = await (0, agentScanExecutor_1.executeAction)(recoveryAction, ctx, startResp.runId, client, target);
735
+ }
736
+ qualityTracker.recordToolUse(recoveryAction.type);
737
+ investigationState.recordToolUse(recoveryAction.type);
738
+ trace.logToolCompleted(recoveryAction.type, recoveryObservation);
739
+ transcript.push({ action: recoveryAction, observation: `[FINISH GATE RECOVERY] ${recoveryObservation}` });
740
+ }
741
+ continue;
742
+ }
743
+ // Sync candidates-verified: when all candidates are ready for the
744
+ // Juror (supported or terminal) or there are none, the
745
+ // candidates-verified step is complete.
746
+ if (candidateStore.allReadyForJuror()) {
747
+ investigationState.markCandidatesVerified();
202
748
  }
203
749
  // Execute the action locally
204
750
  let observation;
205
751
  let wasBlocked = false;
752
+ let flowResult = null;
753
+ // Reset progress flag — only set to true when actual progress is made
754
+ meaningfulProgressSinceRecovery = false;
206
755
  if (action.type === 'read_file') {
207
756
  const normalizedPath = action.path.replace(/\\/g, '/').toLowerCase();
208
- // Track by (path, startLine, endLine) — allow re-reading with different ranges
209
- const rangeKey = `${normalizedPath}:${action.startLine || 0}:${action.endLine || 0}`;
210
- if (readFiles.has(rangeKey)) {
211
- observation = `File "${action.path}" (lines ${action.startLine || 'all'}-${action.endLine || 'all'}) was already read. The content is in the transcript above. Use a DIFFERENT line range or a different tool.`;
757
+ let totalLines;
758
+ try {
759
+ const fs = require('fs');
760
+ const abs = require('path').resolve(ctx.workspaceRoot, action.path);
761
+ const content = fs.readFileSync(abs, 'utf8');
762
+ totalLines = content.split('\n').length;
763
+ }
764
+ catch { /* best-effort */ }
765
+ const readValue = investigationState.classifyRead(action.path, action.startLine, action.endLine, totalLines);
766
+ const checklist = investigationState.formatChecklistForPrompt();
767
+ const fileMax = maxReadsForFile(action.path);
768
+ const count = investigationState.getReadCount(action.path);
769
+ if (readValue.classification === 'duplicate') {
770
+ qualityTracker.recordRead('duplicate', false);
771
+ investigationState.recordBlockedRead(action.path);
772
+ const nextHint = readValue.nextUnreadRange
773
+ ? `\nNext unread range: lines ${readValue.nextUnreadRange.start}-${readValue.nextUnreadRange.end}. Use read_file with startLine=${readValue.nextUnreadRange.start} and endLine=${readValue.nextUnreadRange.end}.`
774
+ : '';
775
+ observation = `BLOCKED: Lines ${action.startLine || 'all'}-${action.endLine || 'all'} of "${action.path}" were already read. The content is in the transcript above.${nextHint}\n\n${checklist}`;
776
+ wasBlocked = true;
777
+ }
778
+ else if (readValue.classification === 'high-overlap') {
779
+ qualityTracker.recordRead('high-overlap', false);
780
+ investigationState.recordBlockedRead(action.path);
781
+ const nextHint = readValue.nextUnreadRange
782
+ ? `\nNext unread range: lines ${readValue.nextUnreadRange.start}-${readValue.nextUnreadRange.end}. Use read_file with startLine=${readValue.nextUnreadRange.start} and endLine=${readValue.nextUnreadRange.end}.`
783
+ : '';
784
+ observation = `BLOCKED: Lines ${action.startLine || 1}-${action.endLine || totalLines || '?'} of "${action.path}" substantially overlap already-read ranges (${readValue.newLines} new lines of ${(action.endLine || 0) - (action.startLine || 1) + 1} requested). The content is in the transcript above. Read a DIFFERENT section or use trace_flow/check_guard/check_policy to analyze what you've already read.${nextHint}\n\n${checklist}`;
785
+ wasBlocked = true;
786
+ }
787
+ else if (readValue.classification === 'invalid') {
788
+ qualityTracker.recordRead('invalid', false);
789
+ investigationState.recordBlockedRead(action.path);
790
+ observation = `BLOCKED: Invalid range for "${action.path}" (startLine=${action.startLine}, endLine=${action.endLine}). The requested range is inverted or out of bounds. Use a valid line range.\n\n${checklist}`;
212
791
  wasBlocked = true;
213
792
  }
214
793
  else {
215
- // Per-file read cap: dynamic based on file size.
216
- // Small files (7 lines) get 5 reads; large files (5000 lines)
217
- // get up to 30 reads so the agent can cover the whole file.
218
- const fileMax = maxReadsForFile(action.path);
219
- const count = (readFileCounts.get(normalizedPath) || 0) + 1;
220
- if (count > fileMax) {
221
- observation = `BLOCKED: You have already read "${action.path}" ${fileMax} times. Further read_file calls on this file will also be blocked. You MUST use a different tool (search_code, trace_flow, check_guard, check_policy, list_imports, or finish) to proceed.`;
222
- wasBlocked = true;
794
+ qualityTracker.recordRead(readValue.classification, false);
795
+ const readResult = await (0, agentScanExecutor_1.executeReadFileAction)(action, ctx);
796
+ observation = readResult.observation;
797
+ qualityTracker.recordRead(readValue.classification, readResult.truncated);
798
+ if (readResult.totalLines > 0 && !readResult.truncated) {
799
+ investigationState.recordActualRead(action.path, readResult.actualStart, readResult.actualEnd, readResult.totalLines, readResult.truncated);
800
+ const rangeKey = `${normalizedPath}:${readResult.actualStart || 0}:${readResult.actualEnd || 0}`;
801
+ readFiles.add(rangeKey);
802
+ qualityTracker.recordToolUse('read_file');
803
+ investigationState.recordToolUse('read_file');
804
+ meaningfulProgressSinceLastExtension = true;
805
+ meaningfulProgressSinceRecovery = true;
806
+ if (count >= fileMax) {
807
+ observation += `\n\nNOTE: You have read "${action.path}" ${count + 1} times. Consider using search_code, trace_flow, check_guard, or check_policy to analyze the code you've read. If you have enough evidence, call finish to report your findings.\n\n${checklist}`;
808
+ }
809
+ }
810
+ else if (readResult.truncated) {
811
+ investigationState.recordActualRead(action.path, readResult.actualStart, readResult.actualEnd, readResult.totalLines, readResult.truncated);
812
+ qualityTracker.recordToolUse('read_file');
813
+ investigationState.recordToolUse('read_file');
814
+ meaningfulProgressSinceLastExtension = false;
815
+ meaningfulProgressSinceRecovery = false;
223
816
  }
224
817
  else {
225
- readFileCounts.set(normalizedPath, count);
226
- readFiles.add(rangeKey);
227
- observation = await (0, agentScanExecutor_1.executeAction)(action, ctx, startResp.runId, client, target);
818
+ investigationState.recordBlockedRead(action.path);
228
819
  }
229
820
  }
230
821
  }
@@ -245,7 +836,73 @@ async function runAgentScan(ctx, target, options = {}) {
245
836
  else if (action.type === 'check_policy') {
246
837
  toolKey = `check_policy:${action.filePath || ''}`;
247
838
  }
839
+ else if (action.type === 'call_graph') {
840
+ toolKey = `call_graph:${action.filePath || ''}:${action.functionName || ''}`;
841
+ }
842
+ else if (action.type === 'git_blame') {
843
+ toolKey = `git_blame:${action.filePath || ''}:${action.startLine || 0}:${action.endLine || 0}`;
844
+ }
845
+ else if (action.type === 'git_history') {
846
+ toolKey = `git_history:${action.filePath || ''}:${action.functionName || ''}`;
847
+ }
848
+ else if (action.type === 'git_diff') {
849
+ toolKey = `git_diff:${action.baseRef || ''}:${action.headRef || 'HEAD'}`;
850
+ }
851
+ else if (action.type === 'check_dependencies') {
852
+ toolKey = `check_dependencies`;
853
+ }
854
+ else if (action.type === 'read_config') {
855
+ toolKey = `read_config:${action.configKind || 'all'}`;
856
+ }
857
+ else if (action.type === 'find_definition') {
858
+ toolKey = `find_definition:${action.filePath || ''}:${action.symbol || ''}`;
859
+ }
860
+ else if (action.type === 'find_references') {
861
+ toolKey = `find_references:${action.filePath || ''}:${action.symbol || ''}`;
862
+ }
863
+ else if (action.type === 'find_tests') {
864
+ toolKey = `find_tests:${action.filePath || ''}:${action.symbol || ''}`;
865
+ }
866
+ else if (action.type === 'run_tests') {
867
+ const a = action;
868
+ if (a.mode === 'existing') {
869
+ toolKey = `run_tests:existing:${(a.testFiles || []).join(',')}:${a.testPattern || ''}:${a.packageManager || ''}`;
870
+ }
871
+ else {
872
+ const crypto = require('crypto');
873
+ const scriptHash = a.script ? crypto.createHash('sha256').update(a.script).digest('hex').substring(0, 16) : '';
874
+ toolKey = `run_tests:generated:${a.runner || ''}:${scriptHash}`;
875
+ }
876
+ }
248
877
  if (toolKey) {
878
+ // Check for equivalent search intent (same terms, different
879
+ // order) before the raw toolKey dedup.
880
+ if (action.type === 'search_code') {
881
+ const pattern = action.pattern || '';
882
+ const normalized = (0, searchIntent_1.normalizeSearchPattern)(pattern);
883
+ let isEquivalent = false;
884
+ for (const prev of searchedPatterns) {
885
+ if ((0, searchIntent_1.isEquivalentSearchIntent)(prev, pattern)) {
886
+ isEquivalent = true;
887
+ break;
888
+ }
889
+ }
890
+ if (isEquivalent) {
891
+ equivalentSearchCount++;
892
+ if (equivalentSearchCount >= 2) {
893
+ observation = `BLOCKED: This search is equivalent to a previous search (same terms, different order). The results are in the transcript above. Use find_definition, find_references, call_graph, a targeted read_file with specific line numbers, or trace_flow_cross_file instead.`;
894
+ wasBlocked = true;
895
+ const checklist = investigationState.formatChecklistForPrompt();
896
+ observation += `\n\n${checklist}`;
897
+ // Skip execution — jump to the blocked handling
898
+ transcript.push({ action, observation });
899
+ trace.logToolBlocked(action.type, observation.slice(0, 200));
900
+ consecutiveBlockedReads++;
901
+ continue;
902
+ }
903
+ }
904
+ searchedPatterns.add(normalized);
905
+ }
249
906
  const count = (toolCallCounts.get(toolKey) || 0) + 1;
250
907
  toolCallCounts.set(toolKey, count);
251
908
  if (count > MAX_SAME_TOOL_CALL) {
@@ -253,34 +910,374 @@ async function runAgentScan(ctx, target, options = {}) {
253
910
  wasBlocked = true;
254
911
  }
255
912
  else {
256
- observation = await (0, agentScanExecutor_1.executeAction)(action, ctx, startResp.runId, client, target);
913
+ if (action.type === 'trace_flow' || action.type === 'trace_flow_cross_file') {
914
+ const flowAction = await (0, agentScanExecutor_1.executeFlowAction)(action, ctx, startResp.runId, client, target);
915
+ observation = flowAction.observation;
916
+ flowResult = flowAction.flowResult;
917
+ }
918
+ else {
919
+ observation = await (0, agentScanExecutor_1.executeAction)(action, ctx, startResp.runId, client, target);
920
+ }
921
+ qualityTracker.recordToolUse(action.type);
922
+ investigationState.recordToolUse(action.type);
923
+ if (action.type === 'search_code') {
924
+ investigationState.recordSymbolSearch(action.pattern || '');
925
+ }
926
+ // First call with these args = meaningful progress.
927
+ // Repeated call (count > 1) does NOT reset recovery state.
928
+ const isFirstCall = count === 1;
929
+ meaningfulProgressSinceLastExtension = isFirstCall;
930
+ if (isFirstCall) {
931
+ meaningfulProgressSinceRecovery = true;
932
+ }
257
933
  }
258
934
  }
259
935
  else {
260
- observation = await (0, agentScanExecutor_1.executeAction)(action, ctx, startResp.runId, client, target);
936
+ if (action.type === 'trace_flow' || action.type === 'trace_flow_cross_file') {
937
+ const flowAction = await (0, agentScanExecutor_1.executeFlowAction)(action, ctx, startResp.runId, client, target);
938
+ observation = flowAction.observation;
939
+ flowResult = flowAction.flowResult;
940
+ }
941
+ else {
942
+ observation = await (0, agentScanExecutor_1.executeAction)(action, ctx, startResp.runId, client, target);
943
+ }
944
+ qualityTracker.recordToolUse(action.type);
945
+ investigationState.recordToolUse(action.type);
946
+ meaningfulProgressSinceLastExtension = true;
947
+ meaningfulProgressSinceRecovery = true;
261
948
  }
262
949
  }
263
- // Track consecutive blocked reads — if the agent keeps requesting
264
- // read_file on blocked files, force-finish to stop wasting steps.
950
+ // Track consecutive blocked reads — quality-first: don't
951
+ // force-finish while unread ranges remain and budget is available.
952
+ // Instead, force a deterministic recovery action.
265
953
  if (wasBlocked) {
954
+ trace.logToolBlocked(action.type, observation.slice(0, 200));
266
955
  consecutiveBlockedReads++;
267
- if (consecutiveBlockedReads >= 8) {
268
- console.warn(`[Agent Scan Loop] ${consecutiveBlockedReads} consecutive blocked reads — force-finishing to stop wasting steps.`);
269
- transcript.push({ action, observation });
270
- return {
271
- status: 'completed',
272
- findings: [],
273
- transcript,
274
- stepsUsed: stepsTaken,
275
- costSpentUsd,
276
- summary: `Investigation cut short — agent was stuck re-reading files. No findings reported.`,
277
- };
956
+ scanState.recovery.consecutiveBlockedActions = consecutiveBlockedReads;
957
+ scanState.recovery.totalBlockedActions++;
958
+ scanState.recovery.lastBlockedAction = action.type;
959
+ scanState.recovery.meaningfulProgressSinceRecovery = false;
960
+ const recoveryLimit = agentScanProtocol_1.AGENT_SCAN_DEFAULTS.blockedReadRecoveryLimit;
961
+ if (consecutiveBlockedReads >= 2 && consecutiveBlockedReads < 3) {
962
+ const nextRange = target.fileContent
963
+ ? investigationState.getPrioritizedUnreadRange(target.filePath, target.fileContent)
964
+ : investigationState.getNextUnreadRange(target.filePath);
965
+ if (nextRange) {
966
+ observation += `\n\nRECOVERY REQUIRED: You have been blocked ${consecutiveBlockedReads} time(s). Read the next unread range: lines ${nextRange.start}-${nextRange.end}, or use an analysis tool (trace_flow, check_guard, check_policy).`;
967
+ }
968
+ else {
969
+ const recommendedTool = investigationState.getRecommendedRecoveryAction();
970
+ if (recommendedTool) {
971
+ observation += `\n\nRECOVERY REQUIRED: You have been blocked ${consecutiveBlockedReads} time(s). No unread ranges remain. Your next action MUST be: ${recommendedTool}. If you have enough evidence, call finish.`;
972
+ }
973
+ }
974
+ }
975
+ if (consecutiveBlockedReads >= recoveryLimit) {
976
+ const nextRange = target.fileContent
977
+ ? investigationState.getPrioritizedUnreadRange(target.filePath, target.fileContent)
978
+ : investigationState.getNextUnreadRange(target.filePath);
979
+ const hasBudget = budget.stepsRemaining > 0 && costSpentUsd < budget.costCapUsd;
980
+ const hasWallClock = Date.now() - startTime < wallClockMs;
981
+ if (nextRange && hasBudget && hasWallClock) {
982
+ console.warn(`[Agent Scan Loop] ${consecutiveBlockedReads} consecutive blocked reads — forcing deterministic recovery to unread range ${nextRange.start}-${nextRange.end}.`);
983
+ }
984
+ else {
985
+ console.warn(`[Agent Scan Loop] ${consecutiveBlockedReads} consecutive blocked reads — no recovery available, force-finishing.`);
986
+ transcript.push({ action, observation });
987
+ trace.logRunCompleted('completed');
988
+ const incompleteSteps = investigationState.getIncompleteSteps();
989
+ const autoGaps = incompleteSteps.map(step => ({
990
+ title: `Investigation step not completed: ${step}`,
991
+ detail: `The agent was force-finished after repeated blocked reads without completing this required investigation step: ${step}. The investigation was incomplete and vulnerabilities may have been missed.`,
992
+ file: target.filePath,
993
+ requiredEvidence: [`Complete the ${step} step before concluding no vulnerabilities exist`],
994
+ suggestedNextAction: step === 'config-inspection' ? 'read_config'
995
+ : step === 'policy-check' ? 'check_policy'
996
+ : step === 'cross-file-flow' ? 'trace_flow_cross_file'
997
+ : step === 'route-discovery' ? 'get_endpoints'
998
+ : step === 'auth-symbol-search' ? 'search_code'
999
+ : 'continue investigation',
1000
+ priority: 'high',
1001
+ }));
1002
+ const unresolvedTasks = investigationState.getUnresolvedTasks();
1003
+ const taskGaps = unresolvedTasks.map(task => ({
1004
+ title: `Architecture risk unresolved: ${task.claim}`,
1005
+ detail: `This architecture-risk investigation task was not resolved: ${task.claim}. Required evidence: ${task.requiredEvidence.join('; ')}.`,
1006
+ file: task.targetFiles[0] || target.filePath,
1007
+ requiredEvidence: task.requiredEvidence,
1008
+ suggestedNextAction: task.requiredTools[0] || 'continue investigation',
1009
+ priority: 'high',
1010
+ }));
1011
+ const uncoveredRanges = investigationState.getUncoveredRanges(target.filePath);
1012
+ const rangeGaps = uncoveredRanges.map(r => ({
1013
+ title: `Unread range: lines ${r.start}-${r.end} of ${target.filePath}`,
1014
+ detail: `This range was never read during the investigation. Vulnerabilities in this range were not checked.`,
1015
+ file: target.filePath,
1016
+ requiredEvidence: [`Read lines ${r.start}-${r.end} and analyze for vulnerabilities`],
1017
+ suggestedNextAction: 'read_file',
1018
+ priority: 'high',
1019
+ }));
1020
+ const allGaps = [...autoGaps, ...taskGaps, ...rangeGaps];
1021
+ (0, scanState_1.terminateScan)(scanState, 'blocked_read_recovery', `Agent stuck re-reading files. ${allGaps.length} coverage gaps.`);
1022
+ qualityTracker.recordForcedTermination('blocked_read_recovery');
1023
+ const covSummary = investigationState.getCoverageSummary(target.filePath);
1024
+ qualityTracker.recordCoverage(covSummary.totalLines, covSummary.coveredLines, covSummary.uncoveredRangeCount, covSummary.largestUncoveredRange);
1025
+ qualityTracker.recordBudget(stepsTaken, stepsGranted, extensionsGranted, costSpentUsd, budget.costCapUsd, wallClockMs, Date.now() - startTime);
1026
+ return {
1027
+ status: 'completed',
1028
+ findings: [],
1029
+ transcript,
1030
+ stepsUsed: stepsTaken,
1031
+ stepsGranted,
1032
+ extensionsGranted,
1033
+ costSpentUsd,
1034
+ terminationReason: 'blocked_read_recovery',
1035
+ summary: `Investigation cut short — agent was stuck re-reading files. ${allGaps.length} coverage gaps identified.`,
1036
+ investigationNotes: [],
1037
+ coverageGaps: allGaps,
1038
+ };
1039
+ }
278
1040
  }
279
1041
  }
280
1042
  else {
281
- consecutiveBlockedReads = 0;
1043
+ trace.logToolCompleted(action.type, observation);
1044
+ // Only reset the blocked counter when meaningful progress was
1045
+ // made. Repeated searches and truncated reads do NOT reset
1046
+ // recovery state — only new coverage, new symbols, or new
1047
+ // tool calls (first invocation with these args) do.
1048
+ if (meaningfulProgressSinceRecovery) {
1049
+ consecutiveBlockedReads = 0;
1050
+ scanState.recovery.consecutiveBlockedActions = 0;
1051
+ scanState.recovery.meaningfulProgressSinceRecovery = true;
1052
+ }
282
1053
  }
1054
+ // The agent needs both the action it tried and the block/observation
1055
+ // message — the original {action, observation} pair carries both.
1056
+ // We don't need a separate system_event for blocked (unlike
1057
+ // critique, the blocked case is tied to an action the agent took).
283
1058
  transcript.push({ action, observation });
1059
+ // Link evidence to candidates: when the agent runs a verification
1060
+ // action (trace_flow, check_guard, check_policy, trace_flow_cross_file)
1061
+ // on a file that matches a candidate's location, add evidence to
1062
+ // that candidate. This auto-transitions candidates through
1063
+ // discovered → investigating → supported as evidence accumulates.
1064
+ if (!wasBlocked && candidateStore.size() > 0) {
1065
+ const verificationActions = new Set([
1066
+ 'trace_flow', 'trace_flow_cross_file', 'check_guard', 'check_policy',
1067
+ ]);
1068
+ if (verificationActions.has(action.type)) {
1069
+ const actionFile = action.filePath || action.path || target.filePath;
1070
+ const actionFileNorm = String(actionFile).replace(/\\/g, '/').toLowerCase();
1071
+ for (const candidate of candidateStore.getAll()) {
1072
+ const matchesLocation = candidate.locations.some(loc => loc.filePath.replace(/\\/g, '/').toLowerCase() === actionFileNorm);
1073
+ if (matchesLocation) {
1074
+ const evidenceId = `${action.type}:${actionFileNorm}:${stepsTaken}`;
1075
+ const category = action.type === 'trace_flow' || action.type === 'trace_flow_cross_file' ? 'flow'
1076
+ : action.type === 'check_guard' ? 'guard'
1077
+ : action.type === 'check_policy' ? 'policy'
1078
+ : undefined;
1079
+ candidateStore.addEvidence(candidate.id, evidenceId, category);
1080
+ }
1081
+ }
1082
+ }
1083
+ if (action.type === 'read_file' && candidateStore.size() > 0) {
1084
+ const actionFileNorm = String(action.path || target.filePath).replace(/\\/g, '/').toLowerCase();
1085
+ for (const candidate of candidateStore.getAll()) {
1086
+ const matchesLocation = candidate.locations.some(loc => loc.filePath.replace(/\\/g, '/').toLowerCase() === actionFileNorm);
1087
+ if (matchesLocation && candidate.evidenceCategories?.includes('source') !== true) {
1088
+ candidateStore.addEvidence(candidate.id, `source:${actionFileNorm}:${stepsTaken}`, 'source');
1089
+ }
1090
+ }
1091
+ }
1092
+ }
1093
+ // Link evidence to work items: when the agent runs an action on
1094
+ // a file that matches a work item's target files, add evidence
1095
+ // and resolve the work item if it has enough evidence.
1096
+ if (!wasBlocked && workItemQueue.size() > 0) {
1097
+ const actionFile = action.filePath || action.path || target.filePath;
1098
+ const actionFileNorm = String(actionFile).replace(/\\/g, '/').toLowerCase();
1099
+ const evidenceKindMap = {
1100
+ read_file: 'source-range',
1101
+ search_code: 'symbol-reference',
1102
+ trace_flow: 'cross-file-flow',
1103
+ trace_flow_cross_file: 'cross-file-flow',
1104
+ check_guard: 'guard-result',
1105
+ check_policy: 'policy-result',
1106
+ get_endpoints: 'handler-inventory',
1107
+ list_imports: 'symbol-reference',
1108
+ find_definition: 'symbol-definition',
1109
+ find_references: 'symbol-reference',
1110
+ find_tests: 'test-location',
1111
+ run_tests: 'test-result',
1112
+ read_config: 'config-result',
1113
+ call_graph: 'cross-file-flow',
1114
+ };
1115
+ const evidenceKind = evidenceKindMap[action.type] || 'source-range';
1116
+ for (const item of workItemQueue.getExecutable()) {
1117
+ const matchesFile = item.targetFiles.length === 0 ||
1118
+ item.targetFiles.some(f => f.replace(/\\/g, '/').toLowerCase() === actionFileNorm);
1119
+ if (matchesFile) {
1120
+ const evidenceId = `${action.type}:${actionFileNorm}:${stepsTaken}`;
1121
+ workItemQueue.addEvidence(item.id, evidenceId);
1122
+ for (const req of item.requirements) {
1123
+ if (!req.acceptedKinds.includes(evidenceKind))
1124
+ continue;
1125
+ if (req.targetFiles && req.targetFiles.length > 0) {
1126
+ const reqMatches = req.targetFiles.some(f => f.replace(/\\/g, '/').toLowerCase() === actionFileNorm);
1127
+ if (!reqMatches)
1128
+ continue;
1129
+ }
1130
+ if (req.requiredTools && req.requiredTools.length > 0) {
1131
+ if (!req.requiredTools.includes(action.type))
1132
+ continue;
1133
+ }
1134
+ workItemQueue.addEvidenceForRequirement(item.id, req.id, evidenceId);
1135
+ }
1136
+ if (workItemQueue.isFullyResolved(item.id)) {
1137
+ workItemQueue.resolve(item.id);
1138
+ }
1139
+ }
1140
+ }
1141
+ }
1142
+ // Re-sync candidates-verified after evidence linking
1143
+ if (candidateStore.allReadyForJuror()) {
1144
+ investigationState.markCandidatesVerified();
1145
+ }
1146
+ // Evidence ledger: record evidence for each action type. This
1147
+ // populates the evidence ledger so the scheduler can suggest
1148
+ // actions for unsatisfied requirements and the finish gate can
1149
+ // verify all evidence requirements are met.
1150
+ if (!wasBlocked) {
1151
+ const actionFile = String(action.filePath || action.path || target.filePath);
1152
+ const evidenceKindMap = {
1153
+ read_file: 'source-range',
1154
+ search_code: 'symbol-reference',
1155
+ trace_flow: 'cross-file-flow',
1156
+ trace_flow_cross_file: 'cross-file-flow',
1157
+ check_guard: 'guard-result',
1158
+ check_policy: 'policy-result',
1159
+ get_endpoints: 'handler-inventory',
1160
+ list_imports: 'symbol-reference',
1161
+ find_definition: 'symbol-definition',
1162
+ find_references: 'symbol-reference',
1163
+ find_tests: 'test-location',
1164
+ run_tests: 'test-result',
1165
+ read_config: 'config-result',
1166
+ call_graph: 'cross-file-flow',
1167
+ };
1168
+ const kind = evidenceKindMap[action.type];
1169
+ if (kind) {
1170
+ evidenceLedger.recordEvidence({
1171
+ kind: kind,
1172
+ tool: action.type,
1173
+ filePath: actionFile,
1174
+ range: action.type === 'read_file'
1175
+ ? { start: action.startLine || 1, end: action.endLine || 1 }
1176
+ : undefined,
1177
+ symbol: action.symbol || action.pattern || undefined,
1178
+ outcome: wasBlocked ? 'blocked' : 'positive',
1179
+ transcriptStep: stepsTaken,
1180
+ });
1181
+ }
1182
+ }
1183
+ // Flow verification: classify trace_flow/trace_flow_cross_file
1184
+ // results using structured data from the executor, not observation
1185
+ // text heuristics. This prevents the cross-file-flow checklist step
1186
+ // from completing until a flow result has been explicitly classified.
1187
+ if (action.type === 'trace_flow' || action.type === 'trace_flow_cross_file') {
1188
+ const flowFile = String(action.filePath || target.filePath);
1189
+ if (wasBlocked) {
1190
+ investigationState.recordFlowVerification(flowFile, action.type, 'blocked', 0, 'Action was blocked');
1191
+ }
1192
+ else if (flowResult) {
1193
+ investigationState.recordFlowVerification(flowFile, action.type, flowResult.status, flowResult.hops.length, flowResult.error || `${flowResult.hops.length} hop(s)`, {
1194
+ source: flowResult.source,
1195
+ sink: flowResult.sink,
1196
+ hops: flowResult.hops,
1197
+ truncated: flowResult.truncated,
1198
+ error: flowResult.error,
1199
+ });
1200
+ }
1201
+ else {
1202
+ investigationState.recordFlowVerification(flowFile, action.type, 'inconclusive', 0, 'No structured flow result available');
1203
+ }
1204
+ }
1205
+ // handlers and populate the handler inventory. Create handler
1206
+ // review work items for security-sensitive handlers.
1207
+ if (!wasBlocked && action.type === 'get_endpoints') {
1208
+ try {
1209
+ const endpoints = await (0, endpointDiscovery_1.discoverEndpoints)(ctx.workspaceRoot, action.glob);
1210
+ if (endpoints.length > 0) {
1211
+ handlerInventory.addFromEndpoints(endpoints, target.filePath);
1212
+ // Create work items for newly discovered handlers
1213
+ for (const handler of handlerInventory.getAll()) {
1214
+ if (!workItemQueue.all().some(w => w.title === `Review handler: ${handler.symbol || 'unknown'}`)) {
1215
+ workItemQueue.add((0, workItem_1.createHandlerReviewWorkItem)(handler.filePath, handler.symbol || 'unknown'));
1216
+ }
1217
+ }
1218
+ }
1219
+ else {
1220
+ // No handlers found — auto-complete all-handlers-reviewed
1221
+ investigationState.markAllHandlersReviewed();
1222
+ }
1223
+ }
1224
+ catch { /* best-effort */ }
1225
+ }
1226
+ // Auto-complete all-handlers-reviewed if handler inventory is
1227
+ // empty (no handlers to review for this file type)
1228
+ if (handlerInventory.size() === 0 && !investigationState.getCompletedSteps().includes('all-handlers-reviewed')) {
1229
+ investigationState.markAllHandlersReviewed();
1230
+ }
1231
+ // Handler review tracking: when the agent reads a file range that
1232
+ // covers a handler's range, mark that handler as reviewed.
1233
+ if (!wasBlocked && action.type === 'read_file' && handlerInventory.size() > 0) {
1234
+ const readStart = action.startLine || 1;
1235
+ const readEnd = action.endLine || 9999;
1236
+ const readFileNorm = String(action.path || target.filePath).replace(/\\/g, '/').toLowerCase();
1237
+ for (const handler of handlerInventory.getAll()) {
1238
+ if (!handler.reviewed) {
1239
+ const handlerFileNorm = handler.filePath.replace(/\\/g, '/').toLowerCase();
1240
+ if (handlerFileNorm === readFileNorm) {
1241
+ if (handler.range.start >= readStart && handler.range.end <= readEnd) {
1242
+ handlerInventory.markReviewed(handler.id);
1243
+ }
1244
+ }
1245
+ }
1246
+ }
1247
+ }
1248
+ // Deterministic recovery: on the third consecutive blocked read,
1249
+ // skip the API and directly execute a recovery action selected by
1250
+ // the MCP. This prevents wasting API calls on an agent that is
1251
+ // stuck re-reading files.
1252
+ if (wasBlocked && consecutiveBlockedReads >= 3) {
1253
+ const recoveryAction = selectDeterministicRecoveryAction(investigationState, target, ctx);
1254
+ if (recoveryAction) {
1255
+ stepsTaken++;
1256
+ budget.stepsRemaining--;
1257
+ trace.nextStep();
1258
+ trace.logActionSelected(recoveryAction.type, 'deterministic-recovery', false);
1259
+ if (options.onProgress) {
1260
+ options.onProgress(stepsTaken, stepsGranted, `Step ${stepsTaken}: deterministic recovery: ${describeAction(recoveryAction)}`);
1261
+ }
1262
+ let recoveryObservation;
1263
+ if (recoveryAction.type === 'read_file') {
1264
+ const readResult = await (0, agentScanExecutor_1.executeReadFileAction)(recoveryAction, ctx);
1265
+ recoveryObservation = readResult.observation;
1266
+ if (readResult.totalLines > 0) {
1267
+ investigationState.recordActualRead(recoveryAction.path, readResult.actualStart, readResult.actualEnd, readResult.totalLines, readResult.truncated);
1268
+ }
1269
+ }
1270
+ else {
1271
+ recoveryObservation = await (0, agentScanExecutor_1.executeAction)(recoveryAction, ctx, startResp.runId, client, target);
1272
+ }
1273
+ qualityTracker.recordToolUse(recoveryAction.type);
1274
+ investigationState.recordToolUse(recoveryAction.type);
1275
+ consecutiveBlockedReads = 0; // recovery resets the counter
1276
+ meaningfulProgressSinceRecovery = true; // recovery is meaningful progress
1277
+ trace.logToolCompleted(recoveryAction.type, recoveryObservation);
1278
+ transcript.push({ action: recoveryAction, observation: `[DETERMINISTIC RECOVERY] ${recoveryObservation}` });
1279
+ }
1280
+ }
284
1281
  }
285
1282
  }
286
1283
  catch (err) {
@@ -289,11 +1286,162 @@ async function runAgentScan(ctx, target, options = {}) {
289
1286
  findings: [],
290
1287
  transcript: [],
291
1288
  stepsUsed: stepsTaken,
1289
+ stepsGranted,
1290
+ extensionsGranted,
292
1291
  costSpentUsd,
1292
+ terminationReason: 'api_error',
293
1293
  error: err.message || String(err),
1294
+ investigationNotes: [],
1295
+ coverageGaps: [],
294
1296
  };
295
1297
  }
296
1298
  }
1299
+ function isCoherentText(text) {
1300
+ if (!text || text.length < 5)
1301
+ return false;
1302
+ const asciiLetters = (text.match(/[a-zA-Z]/g) || []).length;
1303
+ const totalChars = text.length;
1304
+ if (totalChars > 0 && asciiLetters / totalChars < 0.3)
1305
+ return false;
1306
+ const controlChars = (text.match(/[\x00-\x08\x0B\x0C\x0E-\x1F\uFFFD]/g) || []).length;
1307
+ if (controlChars > 0)
1308
+ return false;
1309
+ const words = text.split(/\s+/).filter(w => w.length > 1);
1310
+ if (words.length < 3)
1311
+ return false;
1312
+ return true;
1313
+ }
1314
+ const TRANSCRIPT_CHAR_BUDGET = 120_000;
1315
+ const TRANSCRIPT_KEEP_RECENT = 12;
1316
+ const TRANSCRIPT_SUMMARY_LEN = 200;
1317
+ function estimateTranscriptSize(transcript) {
1318
+ let size = 0;
1319
+ for (const step of transcript) {
1320
+ size += (step.observation || '').length;
1321
+ size += JSON.stringify(step.action || {}).length;
1322
+ }
1323
+ return size;
1324
+ }
1325
+ function compactTranscript(transcript) {
1326
+ const totalSize = estimateTranscriptSize(transcript);
1327
+ if (totalSize <= TRANSCRIPT_CHAR_BUDGET || transcript.length <= TRANSCRIPT_KEEP_RECENT) {
1328
+ return transcript;
1329
+ }
1330
+ const cutoff = transcript.length - TRANSCRIPT_KEEP_RECENT;
1331
+ const compacted = [];
1332
+ for (let i = 0; i < transcript.length; i++) {
1333
+ if (i < cutoff) {
1334
+ const obs = transcript[i].observation || '';
1335
+ const actionType = transcript[i].action?.type || 'unknown';
1336
+ if (actionType === 'system_event' || actionType === 'finish') {
1337
+ compacted.push(transcript[i]);
1338
+ }
1339
+ else if (obs.length > TRANSCRIPT_SUMMARY_LEN) {
1340
+ compacted.push({
1341
+ action: transcript[i].action,
1342
+ observation: obs.slice(0, TRANSCRIPT_SUMMARY_LEN) + '\n...[compacted]',
1343
+ });
1344
+ }
1345
+ else {
1346
+ compacted.push(transcript[i]);
1347
+ }
1348
+ }
1349
+ else {
1350
+ compacted.push(transcript[i]);
1351
+ }
1352
+ }
1353
+ return compacted;
1354
+ }
1355
+ function compactTranscriptAggressive(transcript) {
1356
+ const keepRecent = 6;
1357
+ const summaryLen = 100;
1358
+ if (transcript.length <= keepRecent) {
1359
+ return transcript;
1360
+ }
1361
+ const cutoff = transcript.length - keepRecent;
1362
+ const compacted = [];
1363
+ for (let i = 0; i < transcript.length; i++) {
1364
+ if (i < cutoff) {
1365
+ const obs = transcript[i].observation || '';
1366
+ const actionType = transcript[i].action?.type || 'unknown';
1367
+ if (actionType === 'system_event' || actionType === 'finish') {
1368
+ compacted.push(transcript[i]);
1369
+ }
1370
+ else if (obs.length > summaryLen) {
1371
+ compacted.push({
1372
+ action: transcript[i].action,
1373
+ observation: obs.slice(0, summaryLen) + '\n...[compacted]',
1374
+ });
1375
+ }
1376
+ else {
1377
+ compacted.push(transcript[i]);
1378
+ }
1379
+ }
1380
+ else {
1381
+ compacted.push(transcript[i]);
1382
+ }
1383
+ }
1384
+ return compacted;
1385
+ }
1386
+ const VALID_SEVERITIES = new Set(['critical', 'high', 'medium', 'low']);
1387
+ function sanitizeFindingText(text) {
1388
+ if (!text)
1389
+ return text;
1390
+ return text.replace(/[\x00-\x08\x0B\x0C\x0E-\x1F\uFFFD]/g, '').trim();
1391
+ }
1392
+ function validateFinding(f, index) {
1393
+ const issues = [];
1394
+ const sanitized = { ...f };
1395
+ if (typeof sanitized.line !== 'number' || sanitized.line < 1 || !Number.isFinite(sanitized.line)) {
1396
+ issues.push(`finding[${index}]: invalid line "${sanitized.line}" — defaulting to 0`);
1397
+ sanitized.line = 0;
1398
+ }
1399
+ if (typeof sanitized.type !== 'string' || sanitized.type.trim().length === 0) {
1400
+ issues.push(`finding[${index}]: missing or empty type — defaulting to "unknown"`);
1401
+ sanitized.type = 'unknown';
1402
+ }
1403
+ if (typeof sanitized.severity !== 'string' || !VALID_SEVERITIES.has(sanitized.severity)) {
1404
+ issues.push(`finding[${index}]: invalid severity "${sanitized.severity}" — defaulting to "low"`);
1405
+ sanitized.severity = 'low';
1406
+ }
1407
+ if (typeof sanitized.confidence !== 'number' || !Number.isFinite(sanitized.confidence)) {
1408
+ issues.push(`finding[${index}]: invalid confidence — defaulting to 0.3`);
1409
+ sanitized.confidence = 0.3;
1410
+ }
1411
+ else {
1412
+ sanitized.confidence = Math.max(0, Math.min(1, sanitized.confidence));
1413
+ }
1414
+ if (sanitized.lineEnd !== undefined && (typeof sanitized.lineEnd !== 'number' || sanitized.lineEnd < sanitized.line)) {
1415
+ delete sanitized.lineEnd;
1416
+ }
1417
+ sanitized.why = sanitizeFindingText(sanitized.why || '');
1418
+ if (!isCoherentText(sanitized.why)) {
1419
+ issues.push(`finding[${index}]: incoherent "why" — downgrading severity`);
1420
+ sanitized.why = `[Quality warning: model produced incoherent explanation] ${sanitized.why || '(empty)'}`;
1421
+ if (sanitized.severity === 'high' || sanitized.severity === 'critical') {
1422
+ sanitized.severity = 'medium';
1423
+ }
1424
+ sanitized.confidence = Math.min(sanitized.confidence, 0.3);
1425
+ }
1426
+ sanitized.evidence = sanitizeFindingText(sanitized.evidence || '');
1427
+ return { finding: sanitized, issues };
1428
+ }
1429
+ function sanitizeFindings(findings) {
1430
+ if (!Array.isArray(findings))
1431
+ return [];
1432
+ const seen = new Set();
1433
+ const result = [];
1434
+ for (let i = 0; i < findings.length; i++) {
1435
+ const { finding, issues } = validateFinding(findings[i], i);
1436
+ const dedupKey = `${finding.type}:${finding.line}`;
1437
+ if (seen.has(dedupKey)) {
1438
+ continue;
1439
+ }
1440
+ seen.add(dedupKey);
1441
+ result.push(finding);
1442
+ }
1443
+ return result;
1444
+ }
297
1445
  function describeAction(action) {
298
1446
  switch (action.type) {
299
1447
  case 'read_file': return `read_file(${action.path})`;
@@ -305,7 +1453,63 @@ function describeAction(action) {
305
1453
  case 'get_endpoints': return `get_endpoints(${action.glob || 'all'})`;
306
1454
  case 'list_imports': return `list_imports(${action.filePath})`;
307
1455
  case 'list_files': return `list_files(${action.path || 'root'})`;
1456
+ case 'call_graph': return `call_graph(${action.filePath})`;
1457
+ case 'git_blame': return `git_blame(${action.filePath})`;
1458
+ case 'git_history': return `git_history(${action.filePath || 'repo'})`;
1459
+ case 'git_diff': return `git_diff(${action.baseRef}..${action.headRef || 'HEAD'})`;
1460
+ case 'check_dependencies': return 'check_dependencies';
1461
+ case 'read_config': return `read_config(${action.configKind || 'all'})`;
1462
+ case 'find_definition': return `find_definition(${action.symbol})`;
1463
+ case 'find_references': return `find_references(${action.symbol})`;
1464
+ case 'find_tests': return `find_tests(${action.filePath}${action.symbol ? ':' + action.symbol : ''})`;
1465
+ case 'run_tests': return `run_tests(${action.mode}${action.testFiles?.length ? ':' + action.testFiles.length + ' files' : ''})`;
308
1466
  case 'finish': return 'finish';
1467
+ case 'system_event': return `system_event(${action.eventType})`;
1468
+ }
1469
+ }
1470
+ /**
1471
+ * Select a deterministic recovery action based on investigation state.
1472
+ *
1473
+ * Priority:
1474
+ * 1. Unread range exists on the target file → read_file(next unread range)
1475
+ * 2. route-discovery missing → get_endpoints
1476
+ * 3. policy-check missing → check_policy
1477
+ * 4. auth-symbol-search missing → search_code
1478
+ * 5. cross-file-flow missing → trace_flow_cross_file
1479
+ * 6. config-inspection missing → read_config
1480
+ * 7. tests-found missing → find_tests
1481
+ * 8. otherwise → null (force-finish will handle it)
1482
+ */
1483
+ function selectDeterministicRecoveryAction(state, target, ctx) {
1484
+ // Priority 1: unread range on the target file
1485
+ const nextRange = state.getNextUnreadRange(target.filePath);
1486
+ if (nextRange) {
1487
+ return {
1488
+ type: 'read_file',
1489
+ path: target.filePath,
1490
+ startLine: nextRange.start,
1491
+ endLine: nextRange.end,
1492
+ rationale: 'Deterministic recovery: reading next unread range',
1493
+ };
1494
+ }
1495
+ // Priority 2-7: incomplete investigation steps
1496
+ const incomplete = state.getIncompleteSteps();
1497
+ for (const step of incomplete) {
1498
+ switch (step) {
1499
+ case 'route-discovery':
1500
+ return { type: 'get_endpoints', rationale: 'Deterministic recovery: route discovery' };
1501
+ case 'policy-check':
1502
+ return { type: 'check_policy', filePath: target.filePath, rationale: 'Deterministic recovery: policy check' };
1503
+ case 'auth-symbol-search':
1504
+ return { type: 'search_code', pattern: 'auth|require|guard|permission|owner', rationale: 'Deterministic recovery: auth symbol search' };
1505
+ case 'cross-file-flow':
1506
+ return { type: 'trace_flow_cross_file', filePath: target.filePath, rationale: 'Deterministic recovery: cross-file flow' };
1507
+ case 'config-inspection':
1508
+ return { type: 'read_config', configKind: 'all', rationale: 'Deterministic recovery: config inspection' };
1509
+ case 'tests-found':
1510
+ return { type: 'find_tests', filePath: target.filePath, rationale: 'Deterministic recovery: find tests' };
1511
+ }
309
1512
  }
1513
+ return null;
310
1514
  }
311
1515
  //# sourceMappingURL=agentScanLoop.js.map