@securecode-ai/mcp 0.5.5 → 0.5.6

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (185) hide show
  1. package/dist/api/types.d.ts +7 -0
  2. package/dist/approval/auditLog.d.ts +1 -1
  3. package/dist/approval/auditLog.js +11 -3
  4. package/dist/approval/auditLog.js.map +1 -1
  5. package/dist/approval/broker.d.ts +7 -2
  6. package/dist/approval/broker.js +239 -110
  7. package/dist/approval/broker.js.map +1 -1
  8. package/dist/approval/policy.d.ts +10 -0
  9. package/dist/approval/policy.js +104 -0
  10. package/dist/approval/policy.js.map +1 -0
  11. package/dist/approval/types.d.ts +11 -2
  12. package/dist/approval/types.js +10 -1
  13. package/dist/approval/types.js.map +1 -1
  14. package/dist/attack/agentScanExecutor.d.ts +35 -1
  15. package/dist/attack/agentScanExecutor.js +405 -76
  16. package/dist/attack/agentScanExecutor.js.map +1 -1
  17. package/dist/attack/agentScanLoop.d.ts +20 -0
  18. package/dist/attack/agentScanLoop.js +1264 -60
  19. package/dist/attack/agentScanLoop.js.map +1 -1
  20. package/dist/attack/agentScanProtocol.d.ts +382 -6
  21. package/dist/attack/agentScanProtocol.js +110 -4
  22. package/dist/attack/agentScanProtocol.js.map +1 -1
  23. package/dist/attack/agentTrace.d.ts +73 -0
  24. package/dist/attack/agentTrace.js +228 -0
  25. package/dist/attack/agentTrace.js.map +1 -0
  26. package/dist/attack/architectureScoutExecutor.d.ts +18 -0
  27. package/dist/attack/architectureScoutExecutor.js +211 -0
  28. package/dist/attack/architectureScoutExecutor.js.map +1 -0
  29. package/dist/attack/architectureScoutLoop.d.ts +36 -0
  30. package/dist/attack/architectureScoutLoop.js +403 -0
  31. package/dist/attack/architectureScoutLoop.js.map +1 -0
  32. package/dist/attack/architectureScoutProtocol.d.ts +209 -0
  33. package/dist/attack/architectureScoutProtocol.js +47 -0
  34. package/dist/attack/architectureScoutProtocol.js.map +1 -0
  35. package/dist/attack/candidateStore.d.ts +95 -0
  36. package/dist/attack/candidateStore.js +231 -0
  37. package/dist/attack/candidateStore.js.map +1 -0
  38. package/dist/attack/evidenceLedger.d.ts +67 -0
  39. package/dist/attack/evidenceLedger.js +192 -0
  40. package/dist/attack/evidenceLedger.js.map +1 -0
  41. package/dist/attack/finishGate.d.ts +62 -0
  42. package/dist/attack/finishGate.js +209 -0
  43. package/dist/attack/finishGate.js.map +1 -0
  44. package/dist/attack/fixCodeMerge.d.ts +27 -0
  45. package/dist/attack/fixCodeMerge.js +42 -0
  46. package/dist/attack/fixCodeMerge.js.map +1 -0
  47. package/dist/attack/fixVerifyLoop.d.ts +55 -0
  48. package/dist/attack/fixVerifyLoop.js +187 -0
  49. package/dist/attack/fixVerifyLoop.js.map +1 -0
  50. package/dist/attack/investigationProfiles.d.ts +36 -0
  51. package/dist/attack/investigationProfiles.js +144 -0
  52. package/dist/attack/investigationProfiles.js.map +1 -0
  53. package/dist/attack/investigationState.d.ts +227 -0
  54. package/dist/attack/investigationState.js +666 -0
  55. package/dist/attack/investigationState.js.map +1 -0
  56. package/dist/attack/mutationOperators.d.ts +22 -0
  57. package/dist/attack/mutationOperators.js +170 -0
  58. package/dist/attack/mutationOperators.js.map +1 -0
  59. package/dist/attack/mutationTest.d.ts +33 -0
  60. package/dist/attack/mutationTest.js +110 -0
  61. package/dist/attack/mutationTest.js.map +1 -0
  62. package/dist/attack/proofGate.d.ts +20 -0
  63. package/dist/attack/proofGate.js +104 -0
  64. package/dist/attack/proofGate.js.map +1 -0
  65. package/dist/attack/proofTypes.d.ts +57 -0
  66. package/dist/attack/proofTypes.js +32 -0
  67. package/dist/attack/proofTypes.js.map +1 -0
  68. package/dist/attack/protocolValidator.d.ts +50 -0
  69. package/dist/attack/protocolValidator.js +427 -0
  70. package/dist/attack/protocolValidator.js.map +1 -0
  71. package/dist/attack/qualityMetrics.d.ts +133 -0
  72. package/dist/attack/qualityMetrics.js +226 -0
  73. package/dist/attack/qualityMetrics.js.map +1 -0
  74. package/dist/attack/scanScheduler.d.ts +52 -0
  75. package/dist/attack/scanScheduler.js +265 -0
  76. package/dist/attack/scanScheduler.js.map +1 -0
  77. package/dist/attack/scanState.d.ts +58 -0
  78. package/dist/attack/scanState.js +102 -0
  79. package/dist/attack/scanState.js.map +1 -0
  80. package/dist/attack/searchIntent.d.ts +16 -0
  81. package/dist/attack/searchIntent.js +57 -0
  82. package/dist/attack/searchIntent.js.map +1 -0
  83. package/dist/attack/verifyLoop.d.ts +43 -0
  84. package/dist/attack/verifyLoop.js +314 -14
  85. package/dist/attack/verifyLoop.js.map +1 -1
  86. package/dist/attack/workItem.d.ts +57 -0
  87. package/dist/attack/workItem.js +225 -0
  88. package/dist/attack/workItem.js.map +1 -0
  89. package/dist/audit/findingReviewQueue.d.ts +109 -0
  90. package/dist/audit/findingReviewQueue.js +335 -0
  91. package/dist/audit/findingReviewQueue.js.map +1 -0
  92. package/dist/audit/scanAuditLog.d.ts +160 -0
  93. package/dist/audit/scanAuditLog.js +404 -0
  94. package/dist/audit/scanAuditLog.js.map +1 -0
  95. package/dist/dependency/dependencyChecker.js +35 -9
  96. package/dist/dependency/dependencyChecker.js.map +1 -1
  97. package/dist/dependency/exploitPriority.d.ts +33 -0
  98. package/dist/dependency/exploitPriority.js +78 -0
  99. package/dist/dependency/exploitPriority.js.map +1 -0
  100. package/dist/dependency/finding.d.ts +16 -0
  101. package/dist/dependency/osvClient.js +2 -0
  102. package/dist/dependency/osvClient.js.map +1 -1
  103. package/dist/dependency/types.d.ts +12 -1
  104. package/dist/mcp/server.js +13 -2
  105. package/dist/mcp/server.js.map +1 -1
  106. package/dist/mcp/tools.d.ts +1 -0
  107. package/dist/mcp/tools.js +127 -6
  108. package/dist/mcp/tools.js.map +1 -1
  109. package/dist/project-map/agentMemory.d.ts +83 -1
  110. package/dist/project-map/agentMemory.js +197 -7
  111. package/dist/project-map/agentMemory.js.map +1 -1
  112. package/dist/project-map/architectureContext.d.ts +202 -0
  113. package/dist/project-map/architectureContext.js +421 -0
  114. package/dist/project-map/architectureContext.js.map +1 -0
  115. package/dist/project-map/blastRadius.d.ts +54 -0
  116. package/dist/project-map/blastRadius.js +196 -0
  117. package/dist/project-map/blastRadius.js.map +1 -0
  118. package/dist/project-map/callGraphExtractor.d.ts +13 -0
  119. package/dist/project-map/callGraphExtractor.js +222 -0
  120. package/dist/project-map/callGraphExtractor.js.map +1 -0
  121. package/dist/project-map/capabilityRegistry.d.ts +46 -0
  122. package/dist/project-map/capabilityRegistry.js +91 -0
  123. package/dist/project-map/capabilityRegistry.js.map +1 -0
  124. package/dist/project-map/findTests.d.ts +29 -0
  125. package/dist/project-map/findTests.js +236 -0
  126. package/dist/project-map/findTests.js.map +1 -0
  127. package/dist/project-map/handlerInventory.d.ts +112 -0
  128. package/dist/project-map/handlerInventory.js +184 -0
  129. package/dist/project-map/handlerInventory.js.map +1 -0
  130. package/dist/project-map/implementationResolver.d.ts +43 -0
  131. package/dist/project-map/implementationResolver.js +217 -0
  132. package/dist/project-map/implementationResolver.js.map +1 -0
  133. package/dist/project-map/mapContext.d.ts +6 -1
  134. package/dist/project-map/mapContext.js +1 -0
  135. package/dist/project-map/mapContext.js.map +1 -1
  136. package/dist/project-map/scanCache.d.ts +57 -3
  137. package/dist/project-map/scanCache.js +61 -2
  138. package/dist/project-map/scanCache.js.map +1 -1
  139. package/dist/project-map/symbolIndex.d.ts +39 -0
  140. package/dist/project-map/symbolIndex.js +385 -0
  141. package/dist/project-map/symbolIndex.js.map +1 -0
  142. package/dist/tooling/agentEvalScoring.d.ts +77 -0
  143. package/dist/tooling/agentEvalScoring.js +140 -0
  144. package/dist/tooling/agentEvalScoring.js.map +1 -0
  145. package/dist/tooling/agentRegression.d.ts +53 -0
  146. package/dist/tooling/agentRegression.js +99 -0
  147. package/dist/tooling/agentRegression.js.map +1 -0
  148. package/dist/tools/agentScan.d.ts +69 -0
  149. package/dist/tools/agentScan.js +702 -43
  150. package/dist/tools/agentScan.js.map +1 -1
  151. package/dist/tools/findingReviewTools.d.ts +22 -0
  152. package/dist/tools/findingReviewTools.js +121 -0
  153. package/dist/tools/findingReviewTools.js.map +1 -0
  154. package/dist/tools/fix.js +1 -1
  155. package/dist/tools/fix.js.map +1 -1
  156. package/dist/tools/map.js +191 -1
  157. package/dist/tools/map.js.map +1 -1
  158. package/dist/tools/runTests.d.ts +2 -0
  159. package/dist/tools/runTests.js +29 -0
  160. package/dist/tools/runTests.js.map +1 -0
  161. package/dist/utils/effectMock.d.ts +1 -0
  162. package/dist/utils/effectMock.js +162 -0
  163. package/dist/utils/effectMock.js.map +1 -0
  164. package/dist/utils/gitContext.d.ts +26 -0
  165. package/dist/utils/gitContext.js +317 -0
  166. package/dist/utils/gitContext.js.map +1 -0
  167. package/dist/utils/localTestRunner.d.ts +57 -2
  168. package/dist/utils/localTestRunner.js +122 -115
  169. package/dist/utils/localTestRunner.js.map +1 -1
  170. package/dist/utils/securityConfig.d.ts +17 -0
  171. package/dist/utils/securityConfig.js +192 -0
  172. package/dist/utils/securityConfig.js.map +1 -0
  173. package/dist/utils/testCommandPolicy.d.ts +36 -0
  174. package/dist/utils/testCommandPolicy.js +252 -0
  175. package/dist/utils/testCommandPolicy.js.map +1 -0
  176. package/dist/utils/testRunner.d.ts +56 -0
  177. package/dist/utils/testRunner.js +246 -0
  178. package/dist/utils/testRunner.js.map +1 -0
  179. package/dist/utils/testSafety.d.ts +40 -2
  180. package/dist/utils/testSafety.js +205 -26
  181. package/dist/utils/testSafety.js.map +1 -1
  182. package/dist/utils/verificationSandbox.d.ts +113 -0
  183. package/dist/utils/verificationSandbox.js +656 -0
  184. package/dist/utils/verificationSandbox.js.map +1 -0
  185. package/package.json +1 -1
@@ -51,20 +51,115 @@ var __importStar = (this && this.__importStar) || (function () {
51
51
  })();
52
52
  Object.defineProperty(exports, "__esModule", { value: true });
53
53
  exports.toolAgentScan = toolAgentScan;
54
+ exports.confidenceCeilingForFinding = confidenceCeilingForFinding;
55
+ exports.applyConfidenceClamp = applyConfidenceClamp;
54
56
  const client_1 = require("../api/client");
55
57
  const path = __importStar(require("path"));
56
58
  const files_1 = require("../utils/files");
57
59
  const mapContext_1 = require("../project-map/mapContext");
58
60
  const scanCache_1 = require("../project-map/scanCache");
59
61
  const agentMemory_1 = require("../project-map/agentMemory");
62
+ const capabilityRegistry_1 = require("../project-map/capabilityRegistry");
63
+ const cache_1 = require("../project-map/cache");
64
+ const architectureContext_1 = require("../project-map/architectureContext");
60
65
  const agentScanLoop_1 = require("../attack/agentScanLoop");
61
66
  const verifyLoop_1 = require("../attack/verifyLoop");
67
+ const fixVerifyLoop_1 = require("../attack/fixVerifyLoop");
68
+ const agentScanProtocol_1 = require("../attack/agentScanProtocol");
69
+ const localTestRunner_1 = require("../utils/localTestRunner");
70
+ const broker_1 = require("../approval/broker");
71
+ const gitContext_1 = require("../utils/gitContext");
72
+ const blastRadius_1 = require("../project-map/blastRadius");
73
+ const scanAuditLog_1 = require("../audit/scanAuditLog");
74
+ const findingReviewQueue_1 = require("../audit/findingReviewQueue");
62
75
  /** Prove high/critical/medium findings — skip only low. */
63
76
  function shouldProve(finding) {
64
77
  return finding.severity === 'critical' || finding.severity === 'high' || finding.severity === 'medium';
65
78
  }
79
+ /**
80
+ * Map a verification verdict to a precision verification level.
81
+ *
82
+ * PROVEN via local sandbox → exploit-confirmed (end-to-end test ran)
83
+ * PROVEN via API sandbox → impact-confirmed (server-side test ran)
84
+ * UNPROVEN → logic-confirmed (test proved behavior is safe)
85
+ * INCONCLUSIVE/SKIPPED → logic-confirmed (couldn't determine)
86
+ *
87
+ * If the finding already has a verificationLevel from the agent, keep the
88
+ * higher of the two (agent's level vs verify-mapped level).
89
+ */
90
+ function mapVerificationLevel(proven, viaApiSandbox, agentLevel) {
91
+ let mapped;
92
+ if (proven === 'PROVEN') {
93
+ mapped = viaApiSandbox ? 'impact-confirmed' : 'exploit-confirmed';
94
+ }
95
+ else if (proven === 'UNPROVEN') {
96
+ mapped = 'logic-confirmed';
97
+ }
98
+ else {
99
+ mapped = 'logic-confirmed';
100
+ }
101
+ const order = ['logic-confirmed', 'path-confirmed', 'impact-confirmed', 'exploit-confirmed'];
102
+ if (agentLevel) {
103
+ const agentIdx = order.indexOf(agentLevel);
104
+ const mappedIdx = order.indexOf(mapped);
105
+ return agentIdx > mappedIdx ? agentLevel : mapped;
106
+ }
107
+ return mapped;
108
+ }
109
+ /**
110
+ * Map a verify loop sub-verdict to a human-readable review reason.
111
+ * Falls back to 'inconclusive-verification' when no specific reason is known.
112
+ */
113
+ function mapToReviewReason(finding) {
114
+ const sv = finding.verifySubVerdict;
115
+ switch (sv) {
116
+ case 'sandbox-unavailable': return 'sandbox-unavailable';
117
+ case 'budget-exhausted': return 'verification-budget-exhausted';
118
+ case 'runtime-blocked': return 'runtime-blocked';
119
+ case 'cannot-test': return 'test-generation-failed';
120
+ case 'blocked': return 'runtime-blocked';
121
+ default: return 'inconclusive-verification';
122
+ }
123
+ }
124
+ /**
125
+ * Determine whether a finding should be queued for human review.
126
+ *
127
+ * Queue when:
128
+ * - proven === 'INCONCLUSIVE' (verification couldn't decide)
129
+ * - proven === 'SKIPPED' for medium/high/critical (intentionally skipped but
130
+ * still potentially real)
131
+ * - proven === 'UNPROVEN' but confidence remains high (≥60) — the LLM still
132
+ * believes it's real despite the verify test not reproducing
133
+ *
134
+ * Do NOT queue:
135
+ * - Low-severity SKIPPED findings (intentionally not verified)
136
+ * - PROVEN findings (already confirmed by exploit)
137
+ * - UNPROVEN findings with low confidence (likely false positive)
138
+ * - Cancelled/aborted findings (user ended the scan)
139
+ */
140
+ function shouldQueueForHumanReview(finding) {
141
+ if (finding.proven === 'PROVEN')
142
+ return false;
143
+ if (finding.proven === 'NOT_REPRODUCIBLE')
144
+ return false;
145
+ // Don't queue aborted/cancelled findings.
146
+ if (finding.verifySubVerdict === 'cancelled' || finding.verifySubVerdict === 'aborted') {
147
+ return false;
148
+ }
149
+ if (finding.proven === 'INCONCLUSIVE')
150
+ return true;
151
+ // SKIPPED: queue medium/high/critical, skip low.
152
+ if (finding.proven === 'SKIPPED') {
153
+ return finding.severity === 'critical' || finding.severity === 'high' || finding.severity === 'medium';
154
+ }
155
+ // UNPROVEN: queue only if high confidence remains after clamping.
156
+ if (finding.proven === 'UNPROVEN' && finding.confidence >= 60)
157
+ return true;
158
+ return false;
159
+ }
66
160
  async function toolAgentScan(ctx, args) {
67
161
  const progress = args._progress;
162
+ const skipFix = !!args._skipFix;
68
163
  // 1. Resolve code + language
69
164
  let code;
70
165
  let language;
@@ -130,25 +225,121 @@ async function toolAgentScan(ctx, args) {
130
225
  // best-effort — proceed without context
131
226
  }
132
227
  }
133
- // 2b. Check scan cache — if the file hasn't changed, return cached results
228
+ // 2a. Diff-aware blast radius scoping — if baseRef is provided, compute
229
+ // the set of changed files and their blast radius from the project map.
230
+ // The scope is passed to the agent target so the API prompt can guide
231
+ // the agent to focus on changed files and their dependents.
232
+ let scope;
233
+ const baseRef = args.baseRef;
234
+ if (baseRef) {
235
+ try {
236
+ const diffResult = await (0, gitContext_1.getGitChangedFiles)(ctx.workspaceRoot, baseRef, args.headRef);
237
+ if (diffResult.ok && diffResult.files.length > 0) {
238
+ let blastFiles = diffResult.files;
239
+ try {
240
+ const map = await (0, mapContext_1.getMap)(ctx.workspaceRoot);
241
+ if (map && map.files) {
242
+ const blastResult = (0, blastRadius_1.computeBlastRadius)({
243
+ changedFiles: diffResult.files,
244
+ map,
245
+ });
246
+ blastFiles = blastResult.files;
247
+ }
248
+ }
249
+ catch {
250
+ // map not available — use just the changed files
251
+ }
252
+ scope = {
253
+ changedFiles: diffResult.files,
254
+ blastRadius: blastFiles,
255
+ baseRef: diffResult.baseRef,
256
+ headRef: diffResult.headRef,
257
+ };
258
+ }
259
+ }
260
+ catch {
261
+ // best-effort — proceed without scope
262
+ }
263
+ }
264
+ // 2b. Load workspace memory BEFORE the cache check.
265
+ //
266
+ // Memory (dismissed false positives + known facts) is part of the cache
267
+ // key now — a cache hit on an unchanged file must still respect findings
268
+ // the user has dismissed since the cache was written. We load memory
269
+ // first, compute its fingerprint, and:
270
+ // - On cache hit with a matching fingerprint: return as-is.
271
+ // - On cache hit with a different fingerprint (or no fingerprint, for
272
+ // pre-v22 entries): filter the cached findings against current
273
+ // memory before returning.
274
+ // - On cache miss: pass the memory into the agent target as before.
275
+ let workspaceMemory;
276
+ let memoryFingerprint = '';
277
+ let falsePositives = [];
278
+ try {
279
+ const memory = (0, agentMemory_1.loadAgentMemory)(ctx.workspaceRoot);
280
+ workspaceMemory = (0, agentMemory_1.formatMemoryForPrompt)(memory) || undefined;
281
+ falsePositives = memory.falsePositives.map(fp => ({ findingType: fp.findingType, evidenceHash: fp.evidenceHash }));
282
+ memoryFingerprint = (0, scanCache_1.computeMemoryFingerprint)(falsePositives);
283
+ if (filePath) {
284
+ const fileHash = require('crypto').createHash('sha256').update(code).digest('hex').substring(0, 16);
285
+ (0, agentMemory_1.invalidateStaleEntries)(ctx.workspaceRoot, new Map([[filePath, fileHash]]));
286
+ }
287
+ }
288
+ catch {
289
+ // best-effort — proceed without memory
290
+ }
291
+ // 2c. Check scan cache — if the file hasn't changed, return cached results
134
292
  const skipCacheRead = !!args._noCache;
135
293
  const useCache = !!filePath;
136
294
  if (useCache && !skipCacheRead) {
137
295
  try {
138
296
  const cached = (0, scanCache_1.getCachedScan)(ctx.workspaceRoot, filePath, code);
139
297
  if (cached) {
298
+ // Filter against current memory. If the fingerprint matches
299
+ // what was stored at write time, no findings need to be
300
+ // dropped — we can return the cached list as-is. If it
301
+ // differs (or the entry predates memoryHash), filter.
302
+ let filteredFindings = cached.findings;
303
+ if (cached.memoryHash !== memoryFingerprint) {
304
+ filteredFindings = (0, scanCache_1.filterCachedFindingsAgainstMemory)(cached.findings, falsePositives);
305
+ }
140
306
  if (progress)
141
307
  progress(1, 1, 'Cached result — file unchanged since last scan.');
308
+ try {
309
+ const fileHash = require('crypto').createHash('sha256').update(code).digest('hex').substring(0, 16);
310
+ (0, scanAuditLog_1.recordScanAuditSample)(ctx.workspaceRoot, {
311
+ filePath: filePath,
312
+ fileHash,
313
+ language,
314
+ scanStatus: cached.status,
315
+ stepsUsed: cached.stepsUsed,
316
+ costSpentUsd: 0,
317
+ agentFindings: filteredFindings,
318
+ transcript: [],
319
+ cached: true,
320
+ });
321
+ }
322
+ catch {
323
+ // best-effort
324
+ }
142
325
  return {
143
326
  status: cached.status,
144
327
  summary: cached.summary || 'Agent completed (cached).',
145
- agentFindings: cached.findings,
328
+ agentFindings: filteredFindings,
146
329
  findings: [],
330
+ investigationNotes: cached.investigationNotes ?? [],
331
+ coverageGaps: cached.coverageGaps ?? [],
147
332
  stepsUsed: cached.stepsUsed,
333
+ stepsGranted: cached.stepsGranted ?? 40,
334
+ extensionsGranted: cached.extensionsGranted ?? 0,
148
335
  costSpentUsd: 0,
149
336
  transcript: [],
150
337
  cached: true,
151
- provenCount: cached.findings.filter((f) => f.proven === 'PROVEN').length,
338
+ provenCount: filteredFindings.filter((f) => f.proven === 'PROVEN').length,
339
+ reviewQueue: {
340
+ added: [],
341
+ pendingCount: filteredFindings.filter((f) => f.reviewStatus === 'pending').length,
342
+ },
152
343
  };
153
344
  }
154
345
  }
@@ -159,14 +350,38 @@ async function toolAgentScan(ctx, args) {
159
350
  }
160
351
  }
161
352
  // 3. Run the agent loop
162
- // Load workspace memory (false positives + known facts) to inject into target
163
- let workspaceMemory;
164
- try {
165
- const memory = (0, agentMemory_1.loadAgentMemory)(ctx.workspaceRoot);
166
- workspaceMemory = (0, agentMemory_1.formatMemoryForPrompt)(memory) || undefined;
167
- }
168
- catch {
169
- // best-effort — proceed without memory
353
+ // (workspaceMemory was loaded above)
354
+ // 3b. Auto-load a cached architecture context (if present and valid)
355
+ // so the vulnerability investigator starts with project-wide context
356
+ // instead of having to discover "where is auth?", "what's the data
357
+ // layer?" from scratch. The architecture context is produced by
358
+ // `securecode.map action:architecture` and cached in
359
+ // .securecode/architecture-context.json. If it's stale or absent, the
360
+ // agent proceeds without it (no extra cost).
361
+ let architectureContextStr;
362
+ if (filePath) {
363
+ try {
364
+ const map = (0, cache_1.readCache)(ctx.workspaceRoot);
365
+ if (map) {
366
+ // Try each depth; 'standard' is the most common. The cache
367
+ // returns null if the entry is stale or missing.
368
+ for (const d of ['standard', 'deep', 'quick']) {
369
+ const cached = (0, architectureContext_1.getCachedArchitectureContext)(ctx.workspaceRoot, d, map.builtAt, map.version);
370
+ if (cached) {
371
+ architectureContextStr = (0, architectureContext_1.formatArchitectureContextForPrompt)(cached);
372
+ // Append architecture risk tasks specific to this target file
373
+ const riskTasks = (0, architectureContext_1.formatArchitectureRiskTasksForTarget)(cached, filePath);
374
+ if (riskTasks) {
375
+ architectureContextStr = (architectureContextStr || '') + '\n\n' + riskTasks;
376
+ }
377
+ break;
378
+ }
379
+ }
380
+ }
381
+ }
382
+ catch {
383
+ // best-effort — proceed without architecture context
384
+ }
170
385
  }
171
386
  const target = {
172
387
  filePath: filePath || 'inline-code',
@@ -174,6 +389,8 @@ async function toolAgentScan(ctx, args) {
174
389
  fileContent: code,
175
390
  endpointContext,
176
391
  workspaceMemory,
392
+ scope,
393
+ architectureContext: architectureContextStr,
177
394
  };
178
395
  const agentResult = await (0, agentScanLoop_1.runAgentScan)(ctx, target, {
179
396
  signal: args._signal,
@@ -188,6 +405,19 @@ async function toolAgentScan(ctx, args) {
188
405
  // 4. Verify each finding via the verify subagent (replaces sandbox prove + juror)
189
406
  const client = new client_1.ApiClient({ baseUrl: ctx.apiUrl, token: ctx.apiToken });
190
407
  const provenFindings = [];
408
+ // Aggregate verify budget — prevents Phase 2 from spawning unbounded
409
+ // LLM calls (8 rounds × N findings × 2 calls) and unbounded wall-clock.
410
+ const verifyBudget = (0, agentScanProtocol_1.defaultVerifyBudget)();
411
+ const verifyTracker = new agentScanProtocol_1.VerifyBudgetTracker(verifyBudget);
412
+ const abortSignal = args._signal;
413
+ // Track why findings ended up INCONCLUSIVE so the final result can
414
+ // surface one actionable hint instead of N identical per-finding
415
+ // reasons. The most common case on a developer's laptop is
416
+ // `sandbox-unavailable` (no Docker/Deno installed) — that gets a
417
+ // top-level `verifyHint` with install URLs so the user sees it once,
418
+ // at the top of the result, rather than buried in finding.reason.
419
+ let sandboxUnavailableCount = 0;
420
+ let budgetExhaustedCount = 0;
191
421
  const proveable = agentResult.findings.filter(shouldProve);
192
422
  if (proveable.length > 0 && progress) {
193
423
  progress(0, proveable.length, `Verifying ${proveable.length} finding(s)...`);
@@ -198,6 +428,20 @@ async function toolAgentScan(ctx, args) {
198
428
  provenFindings.push({ ...finding, proven: 'SKIPPED', provenReason: 'Low severity — not verified' });
199
429
  continue;
200
430
  }
431
+ // Pre-flight budget check — stop before spending any LLM calls if
432
+ // the aggregate is already exhausted. This is also enforced inside
433
+ // runVerifyLoop, but checking here lets us mark the finding SKIPPED
434
+ // with a clean reason instead of running INCONCLUSIVE.
435
+ if (!verifyTracker.canAttemptFinding()) {
436
+ budgetExhaustedCount++;
437
+ provenFindings.push({
438
+ ...finding,
439
+ proven: 'INCONCLUSIVE',
440
+ provenReason: `Verification budget exhausted (${verifyTracker.findingsAttempted}/${verifyBudget.maxFindings} findings, ${verifyTracker.llmCallsUsed}/${verifyBudget.maxLlmCalls} LLM calls, ${Math.round(verifyTracker.wallClockElapsedMs / 1000)}s/${Math.round(verifyBudget.maxWallClockMs / 1000)}s).`,
441
+ verifySubVerdict: 'budget-exhausted',
442
+ });
443
+ continue;
444
+ }
201
445
  if (progress) {
202
446
  proveIdx++;
203
447
  progress(proveIdx, proveable.length, `Verifying ${finding.type} at line ${finding.line}...`);
@@ -222,15 +466,77 @@ async function toolAgentScan(ctx, args) {
222
466
  workspaceRoot: ctx.workspaceRoot,
223
467
  language,
224
468
  client,
469
+ budgetTracker: verifyTracker,
470
+ signal: abortSignal,
225
471
  onProgress: (round, maxR, msg) => {
226
472
  if (progress)
227
473
  progress(proveIdx, proveable.length, `Verify round ${round}/${maxR}: ${msg}`);
228
474
  },
229
475
  });
476
+ if (result.subVerdict === 'budget-exhausted')
477
+ budgetExhaustedCount++;
478
+ // User cancelled mid-verify: stop verifying further findings and
479
+ // mark the rest as SKIPPED so the report still includes them.
480
+ if (result.subVerdict === 'cancelled') {
481
+ provenFindings.push({
482
+ ...finding,
483
+ proven: result.verdict,
484
+ provenReason: result.reason,
485
+ verifySubVerdict: result.subVerdict,
486
+ });
487
+ for (const remaining of agentResult.findings.slice(agentResult.findings.indexOf(finding) + 1)) {
488
+ provenFindings.push({
489
+ ...remaining,
490
+ proven: 'SKIPPED',
491
+ provenReason: 'Scan cancelled by user — not verified.',
492
+ });
493
+ }
494
+ break;
495
+ }
496
+ // No local sandbox (Docker/Deno) on the user's machine. Fall back to
497
+ // the API-side sandbox on Vultr, which has Docker installed. This
498
+ // gives every user exploit verification without a local install.
499
+ if (result.subVerdict === 'sandbox-unavailable') {
500
+ try {
501
+ const proveResp = await client.postJson('/sandbox/prove', {
502
+ code,
503
+ language,
504
+ vulnerabilityType: finding.type,
505
+ line: finding.line,
506
+ lineEnd: finding.lineEnd,
507
+ evidence: finding.evidence,
508
+ why: finding.why,
509
+ });
510
+ provenFindings.push({
511
+ ...finding,
512
+ proven: proveResp.proven,
513
+ provenReason: proveResp.rationale || proveResp.skipReason || proveResp.sandbox?.reason,
514
+ verifySubVerdict: result.subVerdict,
515
+ verificationLevel: mapVerificationLevel(proveResp.proven, true, finding.verificationLevel),
516
+ proofEvidence: proveResp.proofEvidence,
517
+ proofGateResult: proveResp.proofGateResult,
518
+ });
519
+ }
520
+ catch (proveErr) {
521
+ sandboxUnavailableCount++;
522
+ provenFindings.push({
523
+ ...finding,
524
+ proven: 'INCONCLUSIVE',
525
+ provenReason: result.reason,
526
+ verifySubVerdict: result.subVerdict,
527
+ verificationLevel: mapVerificationLevel('INCONCLUSIVE', false, finding.verificationLevel),
528
+ });
529
+ }
530
+ continue;
531
+ }
230
532
  provenFindings.push({
231
533
  ...finding,
232
534
  proven: result.verdict,
233
535
  provenReason: result.reason,
536
+ verifySubVerdict: result.subVerdict,
537
+ verificationLevel: mapVerificationLevel(result.verdict, false, finding.verificationLevel),
538
+ proofEvidence: result.proofEvidence,
539
+ proofGateResult: result.proofGateResult,
234
540
  });
235
541
  }
236
542
  catch (err) {
@@ -249,6 +555,9 @@ async function toolAgentScan(ctx, args) {
249
555
  ...finding,
250
556
  proven: proveResp.proven,
251
557
  provenReason: proveResp.rationale || proveResp.skipReason || proveResp.sandbox?.reason,
558
+ verificationLevel: mapVerificationLevel(proveResp.proven, true, finding.verificationLevel),
559
+ proofEvidence: proveResp.proofEvidence,
560
+ proofGateResult: proveResp.proofGateResult,
252
561
  });
253
562
  }
254
563
  catch (err2) {
@@ -256,70 +565,297 @@ async function toolAgentScan(ctx, args) {
256
565
  ...finding,
257
566
  proven: 'INCONCLUSIVE',
258
567
  provenReason: `Verify failed: ${err.message}; Sandbox fallback also failed: ${err2.message}`,
568
+ verificationLevel: mapVerificationLevel('INCONCLUSIVE', false, finding.verificationLevel),
259
569
  });
260
570
  }
261
571
  }
262
572
  }
263
- // 4b. Generate fixes for proven/suspected findings
264
- const fixableFindings = provenFindings.filter(f => f.proven === 'PROVEN' || (f.proven !== 'UNPROVEN' && f.confidence >= 60));
265
- if (fixableFindings.length > 0 && progress) {
266
- progress(0, fixableFindings.length, `Generating fixes for ${fixableFindings.length} finding(s)...`);
573
+ // 4a. Capability-based confidence clamping
574
+ // Confidence must reflect what was actually proven, not what the LLM believes.
575
+ // A finding with no structural evidence (no taint trace, no guard check, no
576
+ // verify) cannot be reported at 95% confidence — that's how false positives
577
+ // erode trust in a security tool. The clamp is deterministic and based on:
578
+ // 1. What tools the agent actually used (transcript scan)
579
+ // 2. What tools were available for this language (capability registry)
580
+ // 3. The verify subagent's verdict (PROVEN/UNPROVEN/INCONCLUSIVE)
581
+ clampConfidenceByCapability(provenFindings, agentResult.transcript, language);
582
+ // 4a-ter. Mark findings requiring human review based on proof quality.
583
+ for (const finding of provenFindings) {
584
+ if (finding.proven !== 'PROVEN')
585
+ continue;
586
+ const needsReview = finding.severity === 'critical' ||
587
+ (finding.proofEvidence && finding.proofEvidence.assumptions.length > 0) ||
588
+ !finding.proofEvidence ||
589
+ !finding.proofGateResult ||
590
+ !finding.proofGateResult.eligibleForProven;
591
+ finding.humanReviewRequired = needsReview;
267
592
  }
268
- let fixIdx = 0;
269
- for (const finding of fixableFindings) {
270
- if (progress) {
271
- fixIdx++;
272
- progress(fixIdx, fixableFindings.length, `Fixing ${finding.type} at line ${finding.line}...`);
593
+ // 4a-bis. Queue INCONCLUSIVE findings for non-blocking human review.
594
+ //
595
+ // Findings the verify subagent couldn't prove or disprove are added to a
596
+ // local review queue (.securecode/finding-review-queue.json). The scan
597
+ // does NOT block — the user can later adjudicate each item via the
598
+ // securecode.review-findings and securecode.decide-finding MCP tools.
599
+ // Only findings with a real file path are queued (inline-code scans have
600
+ // no persistent location to review).
601
+ const reviewQueueAdded = [];
602
+ if (filePath) {
603
+ for (const finding of provenFindings) {
604
+ if (!shouldQueueForHumanReview(finding))
605
+ continue;
606
+ try {
607
+ const reviewItem = (0, findingReviewQueue_1.enqueueFindingReview)(ctx.workspaceRoot, {
608
+ workspaceRelativePath: filePath,
609
+ fileContent: code,
610
+ line: finding.line,
611
+ lineEnd: finding.lineEnd,
612
+ findingType: finding.type,
613
+ severity: finding.severity,
614
+ confidence: finding.confidence,
615
+ proven: finding.proven,
616
+ reviewReason: mapToReviewReason(finding),
617
+ verificationReason: finding.provenReason,
618
+ evidence: finding.evidence || '',
619
+ });
620
+ finding.reviewStatus = 'pending';
621
+ finding.reviewId = reviewItem.id;
622
+ reviewQueueAdded.push(reviewItem.id);
623
+ }
624
+ catch (reviewErr) {
625
+ // Review queue persistence failure must not block the scan.
626
+ console.warn(`[Agent Scan] Review queue enqueue failed: ${reviewErr?.message || reviewErr}`);
627
+ }
273
628
  }
629
+ }
630
+ // 4b. Generate fixes for proven/suspected findings (requires approval)
631
+ if (skipFix) {
632
+ console.log('[Agent Scan] Skipping fix generation (_skipFix=true)');
633
+ }
634
+ else {
635
+ const fixableFindings = provenFindings.filter(f => f.proven === 'PROVEN' || (f.proven !== 'UNPROVEN' && f.confidence >= 60));
636
+ if (fixableFindings.length > 0 && progress) {
637
+ progress(0, fixableFindings.length, `Generating fixes for ${fixableFindings.length} finding(s)...`);
638
+ }
639
+ let fixIdx = 0;
640
+ const fixBroker = fixableFindings.length > 0 ? new broker_1.ApprovalBroker() : null;
641
+ if (fixBroker)
642
+ await fixBroker.start();
274
643
  try {
275
- const fixResp = await client.postJson('/fix', {
276
- code,
277
- language,
278
- vulnerability: {
279
- type: finding.type,
280
- line_start: finding.line,
281
- line_end: finding.lineEnd || finding.line,
282
- evidence_snippet: finding.evidence,
283
- },
284
- });
285
- if (fixResp.fixed_code) {
286
- finding.fix = {
287
- fixedCode: fixResp.fixed_code,
288
- replaceRange: { start_line: finding.line, end_line: finding.lineEnd || finding.line },
289
- fixSummary: fixResp.fix_summary || '',
290
- importsNeeded: fixResp.imports_needed,
291
- confidence: fixResp.confidence,
292
- };
644
+ for (const finding of fixableFindings) {
645
+ if (progress) {
646
+ fixIdx++;
647
+ progress(fixIdx, fixableFindings.length, `Fixing ${finding.type} at line ${finding.line}...`);
648
+ }
649
+ const fixSummary = `Generate fix for ${finding.type} at line ${finding.line}${finding.lineEnd ? `-${finding.lineEnd}` : ''}\nSeverity: ${finding.severity} | Confidence: ${finding.confidence}%\nEvidence: ${finding.evidence?.substring(0, 200) || '(none)'}`;
650
+ try {
651
+ const approval = await fixBroker.requestApproval('securecode.agent-scan (fix generation)', fixSummary, [code, language, finding.type, finding.line, finding.lineEnd, finding.evidence, finding.severity, finding.confidence], 60_000, 'paid-generation', ctx.workspaceRoot);
652
+ if (!approval.approved) {
653
+ finding.fixStatus = 'fix-denied';
654
+ finding.fixDeniedReason = approval.reason;
655
+ continue;
656
+ }
657
+ const fixResp = await client.postJson('/fix', {
658
+ code,
659
+ language,
660
+ vulnerability: {
661
+ type: finding.type,
662
+ line_start: finding.line,
663
+ line_end: finding.lineEnd || finding.line,
664
+ evidence_snippet: finding.evidence,
665
+ },
666
+ });
667
+ if (fixResp.fixed_code) {
668
+ finding.fix = {
669
+ fixedCode: fixResp.fixed_code,
670
+ replaceRange: { start_line: finding.line, end_line: finding.lineEnd || finding.line },
671
+ fixSummary: fixResp.fix_summary || '',
672
+ importsNeeded: fixResp.imports_needed,
673
+ confidence: fixResp.confidence,
674
+ };
675
+ finding.fixStatus = 'fix-generated';
676
+ finding.fixApprovalId = approval.requestId;
677
+ // 4c. Re-verify the fix — re-run the exploit against the
678
+ // merged fixed code to prove the fix actually closed the
679
+ // vulnerability. Uses a separate, smaller budget so a
680
+ // single fix verification cannot consume the entire
681
+ // original scan verification budget. No second approval
682
+ // needed: the user already approved fix generation, and
683
+ // this runs inside the existing sandbox without modifying
684
+ // any workspace files.
685
+ try {
686
+ if (progress) {
687
+ progress(fixIdx, fixableFindings.length, `Verifying fix for ${finding.type} at line ${finding.line}...`);
688
+ }
689
+ const fixVerifyBudget = (0, agentScanProtocol_1.defaultFixVerifyBudget)();
690
+ const fixVerifyTracker = new agentScanProtocol_1.VerifyBudgetTracker(fixVerifyBudget);
691
+ const fixVerifyResult = await (0, fixVerifyLoop_1.runFixVerifyLoop)({
692
+ finding: {
693
+ type: finding.type,
694
+ line: finding.line,
695
+ lineEnd: finding.lineEnd,
696
+ evidence: finding.evidence,
697
+ why: finding.why,
698
+ severity: finding.severity,
699
+ },
700
+ originalCode: code,
701
+ fixedCode: fixResp.fixed_code,
702
+ replaceRange: { start_line: finding.line, end_line: finding.lineEnd || finding.line },
703
+ filePath: filePath || '',
704
+ relatedFiles: relatedFiles.map(rf => ({
705
+ filePath: rf.filePath,
706
+ content: rf.content,
707
+ relationship: rf.relationship,
708
+ })),
709
+ workspaceRoot: ctx.workspaceRoot,
710
+ language,
711
+ client,
712
+ originalVerdict: finding.proven,
713
+ budgetTracker: fixVerifyTracker,
714
+ signal: abortSignal,
715
+ onProgress: (round, maxR, msg) => {
716
+ if (progress)
717
+ progress(fixIdx, fixableFindings.length, `Fix verify round ${round}/${maxR}: ${msg}`);
718
+ },
719
+ });
720
+ finding.fixVerification = fixVerifyResult;
721
+ // Map the fix verification status to the fixStatus field.
722
+ switch (fixVerifyResult.status) {
723
+ case 'closed':
724
+ finding.fixStatus = 'fix-verified-closed';
725
+ break;
726
+ case 'still-vulnerable':
727
+ finding.fixStatus = 'fix-still-vulnerable';
728
+ break;
729
+ case 'inconclusive':
730
+ finding.fixStatus = 'fix-verification-inconclusive';
731
+ break;
732
+ case 'syntax-invalid':
733
+ finding.fixStatus = 'fix-syntax-invalid';
734
+ break;
735
+ // sandbox-unavailable and cancelled leave fixStatus as 'fix-generated'
736
+ }
737
+ }
738
+ catch (fixVerifyErr) {
739
+ // Fix verification failure must not block the scan — the
740
+ // fix is still generated, just not re-verified.
741
+ console.warn(`[Agent Scan] Fix verification failed for ${finding.type} at L${finding.line}: ${fixVerifyErr?.message || fixVerifyErr}`);
742
+ }
743
+ }
744
+ }
745
+ catch (err) {
746
+ finding.fixStatus = 'fix-error';
747
+ finding.fixDeniedReason = err.message;
748
+ console.warn(`[Agent Scan] Fix generation failed for ${finding.type} at L${finding.line}: ${err.message}`);
749
+ }
293
750
  }
294
751
  }
295
- catch (err) {
296
- // Best-effort — finding stays without fix
297
- console.warn(`[Agent Scan] Fix generation failed for ${finding.type} at L${finding.line}: ${err.message}`);
752
+ finally {
753
+ if (fixBroker)
754
+ await fixBroker.stop();
298
755
  }
299
- }
756
+ } // end else (skipFix)
300
757
  // 5. Write to cache before returning
301
758
  if (useCache) {
302
759
  try {
303
760
  (0, scanCache_1.writeCachedScan)(ctx.workspaceRoot, filePath, code, {
304
761
  findings: provenFindings,
762
+ investigationNotes: agentResult.investigationNotes,
763
+ coverageGaps: agentResult.coverageGaps,
305
764
  status: agentResult.status,
306
765
  summary: agentResult.summary,
307
766
  stepsUsed: agentResult.stepsUsed,
767
+ stepsGranted: agentResult.stepsGranted,
768
+ extensionsGranted: agentResult.extensionsGranted,
308
769
  costSpentUsd: agentResult.costSpentUsd,
309
- });
770
+ }, memoryFingerprint);
310
771
  }
311
772
  catch (err) {
312
773
  console.warn(`[Agent Scan] Cache write skipped: ${err?.message || err}`);
313
774
  }
314
775
  }
315
776
  // 6. Return result — the verify subagent IS the verifier (no separate Juror call)
777
+ //
778
+ // verifyHint: surfaced once at the top level when one or more findings
779
+ // couldn't be exploit-verified. The local sandbox (Docker/Deno) was
780
+ // unavailable AND the API-side sandbox fallback failed — so the finding
781
+ // got INCONCLUSIVE. The hint tells the user what to install for local
782
+ // verification (faster, no round-trip) and reminds them the API sandbox
783
+ // is the automatic fallback.
784
+ let verifyHint;
785
+ if (sandboxUnavailableCount > 0) {
786
+ verifyHint = `Exploit verification was skipped for ${sandboxUnavailableCount} finding(s). No local sandbox (Docker or Deno) was detected and the API-side sandbox was unavailable. Findings are reported as INCONCLUSIVE with confidence capped at 75%.\n${localTestRunner_1.SANDBOX_UNAVAILABLE_MESSAGE}`;
787
+ }
788
+ else if (budgetExhaustedCount > 0) {
789
+ verifyHint = `Exploit verification was skipped for ${budgetExhaustedCount} finding(s) because the per-scan verification budget (${verifyBudget.maxFindings} findings, ${verifyBudget.maxLlmCalls} LLM calls, ${Math.round(verifyBudget.maxWallClockMs / 1000)}s) was exhausted. Re-run the scan to verify the remaining findings, or raise the budget via the VerifyBudget config.`;
790
+ }
791
+ // 6a. Record metadata-only audit sample (no source code, no evidence strings)
792
+ try {
793
+ const fileHash = require('crypto').createHash('sha256').update(code).digest('hex').substring(0, 16);
794
+ (0, scanAuditLog_1.recordScanAuditSample)(ctx.workspaceRoot, {
795
+ filePath: filePath || 'inline-code',
796
+ fileHash,
797
+ language,
798
+ scanStatus: agentResult.status,
799
+ stepsUsed: agentResult.stepsUsed,
800
+ costSpentUsd: agentResult.costSpentUsd,
801
+ agentFindings: provenFindings,
802
+ transcript: agentResult.transcript,
803
+ cached: false,
804
+ verifyUsage: {
805
+ findingsAttempted: verifyTracker.findingsAttempted,
806
+ roundsUsed: verifyTracker.roundsUsed,
807
+ llmCallsUsed: verifyTracker.llmCallsUsed,
808
+ costSpentUsd: verifyTracker.costSpentUsd,
809
+ wallClockMs: verifyTracker.wallClockElapsedMs,
810
+ },
811
+ scope,
812
+ investigationNotes: agentResult.investigationNotes,
813
+ coverageGaps: agentResult.coverageGaps,
814
+ hasArchitectureContext: !!architectureContextStr,
815
+ stepsGranted: agentResult.stepsGranted,
816
+ extensionsGranted: agentResult.extensionsGranted,
817
+ terminationReason: agentResult.terminationReason,
818
+ });
819
+ }
820
+ catch {
821
+ // best-effort — audit failure must not block scan results
822
+ }
823
+ // 6b. Persist investigation notes and coverage gaps to workspace memory
824
+ // so future scans can use them as context. These are NOT findings —
825
+ // they guide future investigation without suppressing new discoveries.
826
+ try {
827
+ const fileHashes = new Map();
828
+ if (filePath) {
829
+ fileHashes.set(filePath, require('crypto').createHash('sha256').update(code).digest('hex').substring(0, 16));
830
+ }
831
+ if (agentResult.investigationNotes && agentResult.investigationNotes.length > 0) {
832
+ (0, agentMemory_1.saveInvestigationNotes)(ctx.workspaceRoot, {
833
+ notes: agentResult.investigationNotes,
834
+ fileHashes,
835
+ });
836
+ }
837
+ if (agentResult.coverageGaps && agentResult.coverageGaps.length > 0) {
838
+ (0, agentMemory_1.saveCoverageGaps)(ctx.workspaceRoot, {
839
+ gaps: agentResult.coverageGaps,
840
+ fileHashes,
841
+ });
842
+ }
843
+ }
844
+ catch {
845
+ // best-effort — memory persistence failure must not block results
846
+ }
316
847
  return {
317
848
  status: agentResult.status,
318
849
  summary: agentResult.summary,
319
850
  agentFindings: provenFindings,
320
851
  verifiedFindings: provenFindings.filter(f => f.proven === 'PROVEN'),
321
852
  allFindings: [],
853
+ investigationNotes: agentResult.investigationNotes ?? [],
854
+ coverageGaps: agentResult.coverageGaps ?? [],
322
855
  stepsUsed: agentResult.stepsUsed,
856
+ stepsGranted: agentResult.stepsGranted,
857
+ extensionsGranted: agentResult.extensionsGranted,
858
+ terminationReason: agentResult.terminationReason,
323
859
  costSpentUsd: agentResult.costSpentUsd,
324
860
  transcript: agentResult.transcript,
325
861
  provenCount: provenFindings.filter(f => f.proven === 'PROVEN').length,
@@ -327,6 +863,129 @@ async function toolAgentScan(ctx, args) {
327
863
  inconclusiveCount: provenFindings.filter(f => f.proven === 'INCONCLUSIVE').length,
328
864
  notReproducibleCount: provenFindings.filter(f => f.proven === 'NOT_REPRODUCIBLE').length,
329
865
  skippedCount: provenFindings.filter(f => f.proven === 'SKIPPED').length,
866
+ verifyHint,
867
+ verifyUsage: {
868
+ findingsAttempted: verifyTracker.findingsAttempted,
869
+ roundsUsed: verifyTracker.roundsUsed,
870
+ llmCallsUsed: verifyTracker.llmCallsUsed,
871
+ costSpentUsd: verifyTracker.costSpentUsd,
872
+ wallClockMs: verifyTracker.wallClockElapsedMs,
873
+ budget: verifyBudget,
874
+ },
875
+ reviewQueue: {
876
+ added: reviewQueueAdded,
877
+ pendingCount: provenFindings.filter(f => f.reviewStatus === 'pending').length,
878
+ },
330
879
  };
331
880
  }
881
+ function clampConfidenceByCapability(findings, transcript, language) {
882
+ const cap = (0, capabilityRegistry_1.getCapability)(language);
883
+ let usedTaint = false;
884
+ let usedGuard = false;
885
+ let usedPolicy = false;
886
+ for (const step of transcript) {
887
+ const t = step.action.type;
888
+ if (t === 'trace_flow' || t === 'trace_flow_cross_file')
889
+ usedTaint = true;
890
+ if (t === 'check_guard')
891
+ usedGuard = true;
892
+ if (t === 'check_policy')
893
+ usedPolicy = true;
894
+ }
895
+ let evidenceTools = 0;
896
+ if (usedTaint)
897
+ evidenceTools++;
898
+ if (usedGuard)
899
+ evidenceTools++;
900
+ if (usedPolicy)
901
+ evidenceTools++;
902
+ for (const f of findings) {
903
+ const original = f.confidence;
904
+ const ceiling = confidenceCeilingForFinding(f.proven, cap.tier, evidenceTools);
905
+ // The ceiling is the maximum allowed confidence for this verdict +
906
+ // capability combination. PROVEN findings have no ceiling (undefined)
907
+ // — they're already proven, let the LLM's confidence stand (but still
908
+ // floor at 80 because PROVEN should look confident).
909
+ let clamped = original;
910
+ if (ceiling !== undefined) {
911
+ clamped = Math.min(clamped, ceiling);
912
+ }
913
+ // PROVEN floor — a finding verified by exploit test should not
914
+ // display with a wishy-washy 40% confidence; if the agent is unsure
915
+ // about a PROVEN finding, the proof overrules the agent's doubt.
916
+ if (f.proven === 'PROVEN') {
917
+ clamped = Math.max(clamped, 80);
918
+ }
919
+ clamped = Math.round(clamped);
920
+ f.evidenceLevel = (0, capabilityRegistry_1.evidenceLevelTag)(usedTaint, usedGuard, usedPolicy, f.proven);
921
+ if (clamped !== original) {
922
+ f.originalConfidence = original;
923
+ f.confidence = clamped;
924
+ }
925
+ }
926
+ }
927
+ /**
928
+ * Explicit confidence-ceiling policy for a finding, based on:
929
+ * - the verify verdict (PROVEN / UNPROVEN / INCONCLUSIVE /
930
+ * NOT_REPRODUCIBLE / SKIPPED)
931
+ * - the language capability tier (deep / standard / fallback)
932
+ * - how many structural evidence tools the agent actually used
933
+ * (trace_flow, check_guard, check_policy)
934
+ *
935
+ * Returns `undefined` for PROVEN (no ceiling — the proof overrules the
936
+ * LLM's belief). For everything else, returns the maximum allowed
937
+ * confidence. The caller takes `Math.min(confidence, ceiling)` so multiple
938
+ * applicable bounds compose correctly — last-write-wins is impossible.
939
+ *
940
+ * Verdict semantics:
941
+ * - PROVEN: exploit test ran and reproduced the vulnerability. Strong
942
+ * evidence. No ceiling; floor at 80.
943
+ * - UNPROVEN: exploit test ran and did NOT reproduce. The finding is
944
+ * likely a false positive. Hard cap at 25.
945
+ * - NOT_REPRODUCIBLE: exploit test ran, but the test setup couldn't
946
+ * trigger the vulnerability (e.g. the path requires a live DB). This
947
+ * is NOT the same as UNPROVEN (which actively disproved) and NOT the
948
+ * same as INCONCLUSIVE (which couldn't even run a test). Treat it as
949
+ * "we tried and couldn't confirm" — cap at 35, tighter than
950
+ * INCONCLUSIVE but looser than UNPROVEN.
951
+ * - INCONCLUSIVE: no test could be generated or the sandbox was
952
+ * unavailable. The finding is unverified — keep the LLM's confidence
953
+ * but cap it based on structural evidence (no tools → ≤40, ≥2 tools
954
+ * → ≤75). Fallback-tier languages get ≤55 across the board.
955
+ * - SKIPPED: low severity, we didn't even try. Same as INCONCLUSIVE.
956
+ */
957
+ function confidenceCeilingForFinding(verdict, capabilityTier, evidenceTools) {
958
+ switch (verdict) {
959
+ case 'PROVEN':
960
+ return undefined;
961
+ case 'UNPROVEN':
962
+ return 25;
963
+ case 'NOT_REPRODUCIBLE':
964
+ return 35;
965
+ case 'INCONCLUSIVE':
966
+ case 'SKIPPED':
967
+ if (capabilityTier === 'fallback')
968
+ return 55;
969
+ if (evidenceTools === 0)
970
+ return 40;
971
+ if (evidenceTools === 1)
972
+ return 60;
973
+ return 75;
974
+ default:
975
+ return 40;
976
+ }
977
+ }
978
+ /**
979
+ * Apply the confidence clamp to a single finding. Exported for testing
980
+ * so the test imports the real policy, not a copy that can drift.
981
+ */
982
+ function applyConfidenceClamp(original, verdict, capabilityTier, evidenceTools) {
983
+ const ceiling = confidenceCeilingForFinding(verdict, capabilityTier, evidenceTools);
984
+ let clamped = original;
985
+ if (ceiling !== undefined)
986
+ clamped = Math.min(clamped, ceiling);
987
+ if (verdict === 'PROVEN')
988
+ clamped = Math.max(clamped, 80);
989
+ return Math.round(clamped);
990
+ }
332
991
  //# sourceMappingURL=agentScan.js.map