@securecode-ai/mcp 0.5.5 → 0.5.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/api/types.d.ts +7 -0
- package/dist/approval/auditLog.d.ts +1 -1
- package/dist/approval/auditLog.js +11 -3
- package/dist/approval/auditLog.js.map +1 -1
- package/dist/approval/broker.d.ts +7 -2
- package/dist/approval/broker.js +239 -110
- package/dist/approval/broker.js.map +1 -1
- package/dist/approval/policy.d.ts +10 -0
- package/dist/approval/policy.js +104 -0
- package/dist/approval/policy.js.map +1 -0
- package/dist/approval/types.d.ts +11 -2
- package/dist/approval/types.js +10 -1
- package/dist/approval/types.js.map +1 -1
- package/dist/attack/agentScanExecutor.d.ts +35 -1
- package/dist/attack/agentScanExecutor.js +405 -76
- package/dist/attack/agentScanExecutor.js.map +1 -1
- package/dist/attack/agentScanLoop.d.ts +20 -0
- package/dist/attack/agentScanLoop.js +1264 -60
- package/dist/attack/agentScanLoop.js.map +1 -1
- package/dist/attack/agentScanProtocol.d.ts +382 -6
- package/dist/attack/agentScanProtocol.js +110 -4
- package/dist/attack/agentScanProtocol.js.map +1 -1
- package/dist/attack/agentTrace.d.ts +73 -0
- package/dist/attack/agentTrace.js +228 -0
- package/dist/attack/agentTrace.js.map +1 -0
- package/dist/attack/architectureScoutExecutor.d.ts +18 -0
- package/dist/attack/architectureScoutExecutor.js +211 -0
- package/dist/attack/architectureScoutExecutor.js.map +1 -0
- package/dist/attack/architectureScoutLoop.d.ts +36 -0
- package/dist/attack/architectureScoutLoop.js +403 -0
- package/dist/attack/architectureScoutLoop.js.map +1 -0
- package/dist/attack/architectureScoutProtocol.d.ts +209 -0
- package/dist/attack/architectureScoutProtocol.js +47 -0
- package/dist/attack/architectureScoutProtocol.js.map +1 -0
- package/dist/attack/candidateStore.d.ts +95 -0
- package/dist/attack/candidateStore.js +231 -0
- package/dist/attack/candidateStore.js.map +1 -0
- package/dist/attack/evidenceLedger.d.ts +67 -0
- package/dist/attack/evidenceLedger.js +192 -0
- package/dist/attack/evidenceLedger.js.map +1 -0
- package/dist/attack/finishGate.d.ts +62 -0
- package/dist/attack/finishGate.js +209 -0
- package/dist/attack/finishGate.js.map +1 -0
- package/dist/attack/fixCodeMerge.d.ts +27 -0
- package/dist/attack/fixCodeMerge.js +42 -0
- package/dist/attack/fixCodeMerge.js.map +1 -0
- package/dist/attack/fixVerifyLoop.d.ts +55 -0
- package/dist/attack/fixVerifyLoop.js +187 -0
- package/dist/attack/fixVerifyLoop.js.map +1 -0
- package/dist/attack/investigationProfiles.d.ts +36 -0
- package/dist/attack/investigationProfiles.js +144 -0
- package/dist/attack/investigationProfiles.js.map +1 -0
- package/dist/attack/investigationState.d.ts +227 -0
- package/dist/attack/investigationState.js +666 -0
- package/dist/attack/investigationState.js.map +1 -0
- package/dist/attack/mutationOperators.d.ts +22 -0
- package/dist/attack/mutationOperators.js +170 -0
- package/dist/attack/mutationOperators.js.map +1 -0
- package/dist/attack/mutationTest.d.ts +33 -0
- package/dist/attack/mutationTest.js +110 -0
- package/dist/attack/mutationTest.js.map +1 -0
- package/dist/attack/proofGate.d.ts +20 -0
- package/dist/attack/proofGate.js +104 -0
- package/dist/attack/proofGate.js.map +1 -0
- package/dist/attack/proofTypes.d.ts +57 -0
- package/dist/attack/proofTypes.js +32 -0
- package/dist/attack/proofTypes.js.map +1 -0
- package/dist/attack/protocolValidator.d.ts +50 -0
- package/dist/attack/protocolValidator.js +427 -0
- package/dist/attack/protocolValidator.js.map +1 -0
- package/dist/attack/qualityMetrics.d.ts +133 -0
- package/dist/attack/qualityMetrics.js +226 -0
- package/dist/attack/qualityMetrics.js.map +1 -0
- package/dist/attack/scanScheduler.d.ts +52 -0
- package/dist/attack/scanScheduler.js +265 -0
- package/dist/attack/scanScheduler.js.map +1 -0
- package/dist/attack/scanState.d.ts +58 -0
- package/dist/attack/scanState.js +102 -0
- package/dist/attack/scanState.js.map +1 -0
- package/dist/attack/searchIntent.d.ts +16 -0
- package/dist/attack/searchIntent.js +57 -0
- package/dist/attack/searchIntent.js.map +1 -0
- package/dist/attack/verifyLoop.d.ts +43 -0
- package/dist/attack/verifyLoop.js +314 -14
- package/dist/attack/verifyLoop.js.map +1 -1
- package/dist/attack/workItem.d.ts +57 -0
- package/dist/attack/workItem.js +225 -0
- package/dist/attack/workItem.js.map +1 -0
- package/dist/audit/findingReviewQueue.d.ts +109 -0
- package/dist/audit/findingReviewQueue.js +335 -0
- package/dist/audit/findingReviewQueue.js.map +1 -0
- package/dist/audit/scanAuditLog.d.ts +160 -0
- package/dist/audit/scanAuditLog.js +404 -0
- package/dist/audit/scanAuditLog.js.map +1 -0
- package/dist/dependency/dependencyChecker.js +35 -9
- package/dist/dependency/dependencyChecker.js.map +1 -1
- package/dist/dependency/exploitPriority.d.ts +33 -0
- package/dist/dependency/exploitPriority.js +78 -0
- package/dist/dependency/exploitPriority.js.map +1 -0
- package/dist/dependency/finding.d.ts +16 -0
- package/dist/dependency/osvClient.js +2 -0
- package/dist/dependency/osvClient.js.map +1 -1
- package/dist/dependency/types.d.ts +12 -1
- package/dist/mcp/server.js +13 -2
- package/dist/mcp/server.js.map +1 -1
- package/dist/mcp/tools.d.ts +1 -0
- package/dist/mcp/tools.js +127 -6
- package/dist/mcp/tools.js.map +1 -1
- package/dist/project-map/agentMemory.d.ts +83 -1
- package/dist/project-map/agentMemory.js +197 -7
- package/dist/project-map/agentMemory.js.map +1 -1
- package/dist/project-map/architectureContext.d.ts +202 -0
- package/dist/project-map/architectureContext.js +421 -0
- package/dist/project-map/architectureContext.js.map +1 -0
- package/dist/project-map/blastRadius.d.ts +54 -0
- package/dist/project-map/blastRadius.js +196 -0
- package/dist/project-map/blastRadius.js.map +1 -0
- package/dist/project-map/callGraphExtractor.d.ts +13 -0
- package/dist/project-map/callGraphExtractor.js +222 -0
- package/dist/project-map/callGraphExtractor.js.map +1 -0
- package/dist/project-map/capabilityRegistry.d.ts +46 -0
- package/dist/project-map/capabilityRegistry.js +91 -0
- package/dist/project-map/capabilityRegistry.js.map +1 -0
- package/dist/project-map/findTests.d.ts +29 -0
- package/dist/project-map/findTests.js +236 -0
- package/dist/project-map/findTests.js.map +1 -0
- package/dist/project-map/handlerInventory.d.ts +112 -0
- package/dist/project-map/handlerInventory.js +184 -0
- package/dist/project-map/handlerInventory.js.map +1 -0
- package/dist/project-map/implementationResolver.d.ts +43 -0
- package/dist/project-map/implementationResolver.js +217 -0
- package/dist/project-map/implementationResolver.js.map +1 -0
- package/dist/project-map/mapContext.d.ts +6 -1
- package/dist/project-map/mapContext.js +1 -0
- package/dist/project-map/mapContext.js.map +1 -1
- package/dist/project-map/scanCache.d.ts +57 -3
- package/dist/project-map/scanCache.js +61 -2
- package/dist/project-map/scanCache.js.map +1 -1
- package/dist/project-map/symbolIndex.d.ts +39 -0
- package/dist/project-map/symbolIndex.js +385 -0
- package/dist/project-map/symbolIndex.js.map +1 -0
- package/dist/tooling/agentEvalScoring.d.ts +77 -0
- package/dist/tooling/agentEvalScoring.js +140 -0
- package/dist/tooling/agentEvalScoring.js.map +1 -0
- package/dist/tooling/agentRegression.d.ts +53 -0
- package/dist/tooling/agentRegression.js +99 -0
- package/dist/tooling/agentRegression.js.map +1 -0
- package/dist/tools/agentScan.d.ts +69 -0
- package/dist/tools/agentScan.js +702 -43
- package/dist/tools/agentScan.js.map +1 -1
- package/dist/tools/findingReviewTools.d.ts +22 -0
- package/dist/tools/findingReviewTools.js +121 -0
- package/dist/tools/findingReviewTools.js.map +1 -0
- package/dist/tools/fix.js +1 -1
- package/dist/tools/fix.js.map +1 -1
- package/dist/tools/map.js +191 -1
- package/dist/tools/map.js.map +1 -1
- package/dist/tools/runTests.d.ts +2 -0
- package/dist/tools/runTests.js +29 -0
- package/dist/tools/runTests.js.map +1 -0
- package/dist/utils/effectMock.d.ts +1 -0
- package/dist/utils/effectMock.js +162 -0
- package/dist/utils/effectMock.js.map +1 -0
- package/dist/utils/gitContext.d.ts +26 -0
- package/dist/utils/gitContext.js +317 -0
- package/dist/utils/gitContext.js.map +1 -0
- package/dist/utils/localTestRunner.d.ts +57 -2
- package/dist/utils/localTestRunner.js +122 -115
- package/dist/utils/localTestRunner.js.map +1 -1
- package/dist/utils/securityConfig.d.ts +17 -0
- package/dist/utils/securityConfig.js +192 -0
- package/dist/utils/securityConfig.js.map +1 -0
- package/dist/utils/testCommandPolicy.d.ts +36 -0
- package/dist/utils/testCommandPolicy.js +252 -0
- package/dist/utils/testCommandPolicy.js.map +1 -0
- package/dist/utils/testRunner.d.ts +56 -0
- package/dist/utils/testRunner.js +246 -0
- package/dist/utils/testRunner.js.map +1 -0
- package/dist/utils/testSafety.d.ts +40 -2
- package/dist/utils/testSafety.js +205 -26
- package/dist/utils/testSafety.js.map +1 -1
- package/dist/utils/verificationSandbox.d.ts +113 -0
- package/dist/utils/verificationSandbox.js +656 -0
- package/dist/utils/verificationSandbox.js.map +1 -0
- package/package.json +1 -1
package/dist/tools/agentScan.js
CHANGED
|
@@ -51,20 +51,115 @@ var __importStar = (this && this.__importStar) || (function () {
|
|
|
51
51
|
})();
|
|
52
52
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
53
53
|
exports.toolAgentScan = toolAgentScan;
|
|
54
|
+
exports.confidenceCeilingForFinding = confidenceCeilingForFinding;
|
|
55
|
+
exports.applyConfidenceClamp = applyConfidenceClamp;
|
|
54
56
|
const client_1 = require("../api/client");
|
|
55
57
|
const path = __importStar(require("path"));
|
|
56
58
|
const files_1 = require("../utils/files");
|
|
57
59
|
const mapContext_1 = require("../project-map/mapContext");
|
|
58
60
|
const scanCache_1 = require("../project-map/scanCache");
|
|
59
61
|
const agentMemory_1 = require("../project-map/agentMemory");
|
|
62
|
+
const capabilityRegistry_1 = require("../project-map/capabilityRegistry");
|
|
63
|
+
const cache_1 = require("../project-map/cache");
|
|
64
|
+
const architectureContext_1 = require("../project-map/architectureContext");
|
|
60
65
|
const agentScanLoop_1 = require("../attack/agentScanLoop");
|
|
61
66
|
const verifyLoop_1 = require("../attack/verifyLoop");
|
|
67
|
+
const fixVerifyLoop_1 = require("../attack/fixVerifyLoop");
|
|
68
|
+
const agentScanProtocol_1 = require("../attack/agentScanProtocol");
|
|
69
|
+
const localTestRunner_1 = require("../utils/localTestRunner");
|
|
70
|
+
const broker_1 = require("../approval/broker");
|
|
71
|
+
const gitContext_1 = require("../utils/gitContext");
|
|
72
|
+
const blastRadius_1 = require("../project-map/blastRadius");
|
|
73
|
+
const scanAuditLog_1 = require("../audit/scanAuditLog");
|
|
74
|
+
const findingReviewQueue_1 = require("../audit/findingReviewQueue");
|
|
62
75
|
/** Prove high/critical/medium findings — skip only low. */
|
|
63
76
|
function shouldProve(finding) {
|
|
64
77
|
return finding.severity === 'critical' || finding.severity === 'high' || finding.severity === 'medium';
|
|
65
78
|
}
|
|
79
|
+
/**
|
|
80
|
+
* Map a verification verdict to a precision verification level.
|
|
81
|
+
*
|
|
82
|
+
* PROVEN via local sandbox → exploit-confirmed (end-to-end test ran)
|
|
83
|
+
* PROVEN via API sandbox → impact-confirmed (server-side test ran)
|
|
84
|
+
* UNPROVEN → logic-confirmed (test proved behavior is safe)
|
|
85
|
+
* INCONCLUSIVE/SKIPPED → logic-confirmed (couldn't determine)
|
|
86
|
+
*
|
|
87
|
+
* If the finding already has a verificationLevel from the agent, keep the
|
|
88
|
+
* higher of the two (agent's level vs verify-mapped level).
|
|
89
|
+
*/
|
|
90
|
+
function mapVerificationLevel(proven, viaApiSandbox, agentLevel) {
|
|
91
|
+
let mapped;
|
|
92
|
+
if (proven === 'PROVEN') {
|
|
93
|
+
mapped = viaApiSandbox ? 'impact-confirmed' : 'exploit-confirmed';
|
|
94
|
+
}
|
|
95
|
+
else if (proven === 'UNPROVEN') {
|
|
96
|
+
mapped = 'logic-confirmed';
|
|
97
|
+
}
|
|
98
|
+
else {
|
|
99
|
+
mapped = 'logic-confirmed';
|
|
100
|
+
}
|
|
101
|
+
const order = ['logic-confirmed', 'path-confirmed', 'impact-confirmed', 'exploit-confirmed'];
|
|
102
|
+
if (agentLevel) {
|
|
103
|
+
const agentIdx = order.indexOf(agentLevel);
|
|
104
|
+
const mappedIdx = order.indexOf(mapped);
|
|
105
|
+
return agentIdx > mappedIdx ? agentLevel : mapped;
|
|
106
|
+
}
|
|
107
|
+
return mapped;
|
|
108
|
+
}
|
|
109
|
+
/**
|
|
110
|
+
* Map a verify loop sub-verdict to a human-readable review reason.
|
|
111
|
+
* Falls back to 'inconclusive-verification' when no specific reason is known.
|
|
112
|
+
*/
|
|
113
|
+
function mapToReviewReason(finding) {
|
|
114
|
+
const sv = finding.verifySubVerdict;
|
|
115
|
+
switch (sv) {
|
|
116
|
+
case 'sandbox-unavailable': return 'sandbox-unavailable';
|
|
117
|
+
case 'budget-exhausted': return 'verification-budget-exhausted';
|
|
118
|
+
case 'runtime-blocked': return 'runtime-blocked';
|
|
119
|
+
case 'cannot-test': return 'test-generation-failed';
|
|
120
|
+
case 'blocked': return 'runtime-blocked';
|
|
121
|
+
default: return 'inconclusive-verification';
|
|
122
|
+
}
|
|
123
|
+
}
|
|
124
|
+
/**
|
|
125
|
+
* Determine whether a finding should be queued for human review.
|
|
126
|
+
*
|
|
127
|
+
* Queue when:
|
|
128
|
+
* - proven === 'INCONCLUSIVE' (verification couldn't decide)
|
|
129
|
+
* - proven === 'SKIPPED' for medium/high/critical (intentionally skipped but
|
|
130
|
+
* still potentially real)
|
|
131
|
+
* - proven === 'UNPROVEN' but confidence remains high (≥60) — the LLM still
|
|
132
|
+
* believes it's real despite the verify test not reproducing
|
|
133
|
+
*
|
|
134
|
+
* Do NOT queue:
|
|
135
|
+
* - Low-severity SKIPPED findings (intentionally not verified)
|
|
136
|
+
* - PROVEN findings (already confirmed by exploit)
|
|
137
|
+
* - UNPROVEN findings with low confidence (likely false positive)
|
|
138
|
+
* - Cancelled/aborted findings (user ended the scan)
|
|
139
|
+
*/
|
|
140
|
+
function shouldQueueForHumanReview(finding) {
|
|
141
|
+
if (finding.proven === 'PROVEN')
|
|
142
|
+
return false;
|
|
143
|
+
if (finding.proven === 'NOT_REPRODUCIBLE')
|
|
144
|
+
return false;
|
|
145
|
+
// Don't queue aborted/cancelled findings.
|
|
146
|
+
if (finding.verifySubVerdict === 'cancelled' || finding.verifySubVerdict === 'aborted') {
|
|
147
|
+
return false;
|
|
148
|
+
}
|
|
149
|
+
if (finding.proven === 'INCONCLUSIVE')
|
|
150
|
+
return true;
|
|
151
|
+
// SKIPPED: queue medium/high/critical, skip low.
|
|
152
|
+
if (finding.proven === 'SKIPPED') {
|
|
153
|
+
return finding.severity === 'critical' || finding.severity === 'high' || finding.severity === 'medium';
|
|
154
|
+
}
|
|
155
|
+
// UNPROVEN: queue only if high confidence remains after clamping.
|
|
156
|
+
if (finding.proven === 'UNPROVEN' && finding.confidence >= 60)
|
|
157
|
+
return true;
|
|
158
|
+
return false;
|
|
159
|
+
}
|
|
66
160
|
async function toolAgentScan(ctx, args) {
|
|
67
161
|
const progress = args._progress;
|
|
162
|
+
const skipFix = !!args._skipFix;
|
|
68
163
|
// 1. Resolve code + language
|
|
69
164
|
let code;
|
|
70
165
|
let language;
|
|
@@ -130,25 +225,121 @@ async function toolAgentScan(ctx, args) {
|
|
|
130
225
|
// best-effort — proceed without context
|
|
131
226
|
}
|
|
132
227
|
}
|
|
133
|
-
//
|
|
228
|
+
// 2a. Diff-aware blast radius scoping — if baseRef is provided, compute
|
|
229
|
+
// the set of changed files and their blast radius from the project map.
|
|
230
|
+
// The scope is passed to the agent target so the API prompt can guide
|
|
231
|
+
// the agent to focus on changed files and their dependents.
|
|
232
|
+
let scope;
|
|
233
|
+
const baseRef = args.baseRef;
|
|
234
|
+
if (baseRef) {
|
|
235
|
+
try {
|
|
236
|
+
const diffResult = await (0, gitContext_1.getGitChangedFiles)(ctx.workspaceRoot, baseRef, args.headRef);
|
|
237
|
+
if (diffResult.ok && diffResult.files.length > 0) {
|
|
238
|
+
let blastFiles = diffResult.files;
|
|
239
|
+
try {
|
|
240
|
+
const map = await (0, mapContext_1.getMap)(ctx.workspaceRoot);
|
|
241
|
+
if (map && map.files) {
|
|
242
|
+
const blastResult = (0, blastRadius_1.computeBlastRadius)({
|
|
243
|
+
changedFiles: diffResult.files,
|
|
244
|
+
map,
|
|
245
|
+
});
|
|
246
|
+
blastFiles = blastResult.files;
|
|
247
|
+
}
|
|
248
|
+
}
|
|
249
|
+
catch {
|
|
250
|
+
// map not available — use just the changed files
|
|
251
|
+
}
|
|
252
|
+
scope = {
|
|
253
|
+
changedFiles: diffResult.files,
|
|
254
|
+
blastRadius: blastFiles,
|
|
255
|
+
baseRef: diffResult.baseRef,
|
|
256
|
+
headRef: diffResult.headRef,
|
|
257
|
+
};
|
|
258
|
+
}
|
|
259
|
+
}
|
|
260
|
+
catch {
|
|
261
|
+
// best-effort — proceed without scope
|
|
262
|
+
}
|
|
263
|
+
}
|
|
264
|
+
// 2b. Load workspace memory BEFORE the cache check.
|
|
265
|
+
//
|
|
266
|
+
// Memory (dismissed false positives + known facts) is part of the cache
|
|
267
|
+
// key now — a cache hit on an unchanged file must still respect findings
|
|
268
|
+
// the user has dismissed since the cache was written. We load memory
|
|
269
|
+
// first, compute its fingerprint, and:
|
|
270
|
+
// - On cache hit with a matching fingerprint: return as-is.
|
|
271
|
+
// - On cache hit with a different fingerprint (or no fingerprint, for
|
|
272
|
+
// pre-v22 entries): filter the cached findings against current
|
|
273
|
+
// memory before returning.
|
|
274
|
+
// - On cache miss: pass the memory into the agent target as before.
|
|
275
|
+
let workspaceMemory;
|
|
276
|
+
let memoryFingerprint = '';
|
|
277
|
+
let falsePositives = [];
|
|
278
|
+
try {
|
|
279
|
+
const memory = (0, agentMemory_1.loadAgentMemory)(ctx.workspaceRoot);
|
|
280
|
+
workspaceMemory = (0, agentMemory_1.formatMemoryForPrompt)(memory) || undefined;
|
|
281
|
+
falsePositives = memory.falsePositives.map(fp => ({ findingType: fp.findingType, evidenceHash: fp.evidenceHash }));
|
|
282
|
+
memoryFingerprint = (0, scanCache_1.computeMemoryFingerprint)(falsePositives);
|
|
283
|
+
if (filePath) {
|
|
284
|
+
const fileHash = require('crypto').createHash('sha256').update(code).digest('hex').substring(0, 16);
|
|
285
|
+
(0, agentMemory_1.invalidateStaleEntries)(ctx.workspaceRoot, new Map([[filePath, fileHash]]));
|
|
286
|
+
}
|
|
287
|
+
}
|
|
288
|
+
catch {
|
|
289
|
+
// best-effort — proceed without memory
|
|
290
|
+
}
|
|
291
|
+
// 2c. Check scan cache — if the file hasn't changed, return cached results
|
|
134
292
|
const skipCacheRead = !!args._noCache;
|
|
135
293
|
const useCache = !!filePath;
|
|
136
294
|
if (useCache && !skipCacheRead) {
|
|
137
295
|
try {
|
|
138
296
|
const cached = (0, scanCache_1.getCachedScan)(ctx.workspaceRoot, filePath, code);
|
|
139
297
|
if (cached) {
|
|
298
|
+
// Filter against current memory. If the fingerprint matches
|
|
299
|
+
// what was stored at write time, no findings need to be
|
|
300
|
+
// dropped — we can return the cached list as-is. If it
|
|
301
|
+
// differs (or the entry predates memoryHash), filter.
|
|
302
|
+
let filteredFindings = cached.findings;
|
|
303
|
+
if (cached.memoryHash !== memoryFingerprint) {
|
|
304
|
+
filteredFindings = (0, scanCache_1.filterCachedFindingsAgainstMemory)(cached.findings, falsePositives);
|
|
305
|
+
}
|
|
140
306
|
if (progress)
|
|
141
307
|
progress(1, 1, 'Cached result — file unchanged since last scan.');
|
|
308
|
+
try {
|
|
309
|
+
const fileHash = require('crypto').createHash('sha256').update(code).digest('hex').substring(0, 16);
|
|
310
|
+
(0, scanAuditLog_1.recordScanAuditSample)(ctx.workspaceRoot, {
|
|
311
|
+
filePath: filePath,
|
|
312
|
+
fileHash,
|
|
313
|
+
language,
|
|
314
|
+
scanStatus: cached.status,
|
|
315
|
+
stepsUsed: cached.stepsUsed,
|
|
316
|
+
costSpentUsd: 0,
|
|
317
|
+
agentFindings: filteredFindings,
|
|
318
|
+
transcript: [],
|
|
319
|
+
cached: true,
|
|
320
|
+
});
|
|
321
|
+
}
|
|
322
|
+
catch {
|
|
323
|
+
// best-effort
|
|
324
|
+
}
|
|
142
325
|
return {
|
|
143
326
|
status: cached.status,
|
|
144
327
|
summary: cached.summary || 'Agent completed (cached).',
|
|
145
|
-
agentFindings:
|
|
328
|
+
agentFindings: filteredFindings,
|
|
146
329
|
findings: [],
|
|
330
|
+
investigationNotes: cached.investigationNotes ?? [],
|
|
331
|
+
coverageGaps: cached.coverageGaps ?? [],
|
|
147
332
|
stepsUsed: cached.stepsUsed,
|
|
333
|
+
stepsGranted: cached.stepsGranted ?? 40,
|
|
334
|
+
extensionsGranted: cached.extensionsGranted ?? 0,
|
|
148
335
|
costSpentUsd: 0,
|
|
149
336
|
transcript: [],
|
|
150
337
|
cached: true,
|
|
151
|
-
provenCount:
|
|
338
|
+
provenCount: filteredFindings.filter((f) => f.proven === 'PROVEN').length,
|
|
339
|
+
reviewQueue: {
|
|
340
|
+
added: [],
|
|
341
|
+
pendingCount: filteredFindings.filter((f) => f.reviewStatus === 'pending').length,
|
|
342
|
+
},
|
|
152
343
|
};
|
|
153
344
|
}
|
|
154
345
|
}
|
|
@@ -159,14 +350,38 @@ async function toolAgentScan(ctx, args) {
|
|
|
159
350
|
}
|
|
160
351
|
}
|
|
161
352
|
// 3. Run the agent loop
|
|
162
|
-
//
|
|
163
|
-
|
|
164
|
-
|
|
165
|
-
|
|
166
|
-
|
|
167
|
-
|
|
168
|
-
|
|
169
|
-
|
|
353
|
+
// (workspaceMemory was loaded above)
|
|
354
|
+
// 3b. Auto-load a cached architecture context (if present and valid)
|
|
355
|
+
// so the vulnerability investigator starts with project-wide context
|
|
356
|
+
// instead of having to discover "where is auth?", "what's the data
|
|
357
|
+
// layer?" from scratch. The architecture context is produced by
|
|
358
|
+
// `securecode.map action:architecture` and cached in
|
|
359
|
+
// .securecode/architecture-context.json. If it's stale or absent, the
|
|
360
|
+
// agent proceeds without it (no extra cost).
|
|
361
|
+
let architectureContextStr;
|
|
362
|
+
if (filePath) {
|
|
363
|
+
try {
|
|
364
|
+
const map = (0, cache_1.readCache)(ctx.workspaceRoot);
|
|
365
|
+
if (map) {
|
|
366
|
+
// Try each depth; 'standard' is the most common. The cache
|
|
367
|
+
// returns null if the entry is stale or missing.
|
|
368
|
+
for (const d of ['standard', 'deep', 'quick']) {
|
|
369
|
+
const cached = (0, architectureContext_1.getCachedArchitectureContext)(ctx.workspaceRoot, d, map.builtAt, map.version);
|
|
370
|
+
if (cached) {
|
|
371
|
+
architectureContextStr = (0, architectureContext_1.formatArchitectureContextForPrompt)(cached);
|
|
372
|
+
// Append architecture risk tasks specific to this target file
|
|
373
|
+
const riskTasks = (0, architectureContext_1.formatArchitectureRiskTasksForTarget)(cached, filePath);
|
|
374
|
+
if (riskTasks) {
|
|
375
|
+
architectureContextStr = (architectureContextStr || '') + '\n\n' + riskTasks;
|
|
376
|
+
}
|
|
377
|
+
break;
|
|
378
|
+
}
|
|
379
|
+
}
|
|
380
|
+
}
|
|
381
|
+
}
|
|
382
|
+
catch {
|
|
383
|
+
// best-effort — proceed without architecture context
|
|
384
|
+
}
|
|
170
385
|
}
|
|
171
386
|
const target = {
|
|
172
387
|
filePath: filePath || 'inline-code',
|
|
@@ -174,6 +389,8 @@ async function toolAgentScan(ctx, args) {
|
|
|
174
389
|
fileContent: code,
|
|
175
390
|
endpointContext,
|
|
176
391
|
workspaceMemory,
|
|
392
|
+
scope,
|
|
393
|
+
architectureContext: architectureContextStr,
|
|
177
394
|
};
|
|
178
395
|
const agentResult = await (0, agentScanLoop_1.runAgentScan)(ctx, target, {
|
|
179
396
|
signal: args._signal,
|
|
@@ -188,6 +405,19 @@ async function toolAgentScan(ctx, args) {
|
|
|
188
405
|
// 4. Verify each finding via the verify subagent (replaces sandbox prove + juror)
|
|
189
406
|
const client = new client_1.ApiClient({ baseUrl: ctx.apiUrl, token: ctx.apiToken });
|
|
190
407
|
const provenFindings = [];
|
|
408
|
+
// Aggregate verify budget — prevents Phase 2 from spawning unbounded
|
|
409
|
+
// LLM calls (8 rounds × N findings × 2 calls) and unbounded wall-clock.
|
|
410
|
+
const verifyBudget = (0, agentScanProtocol_1.defaultVerifyBudget)();
|
|
411
|
+
const verifyTracker = new agentScanProtocol_1.VerifyBudgetTracker(verifyBudget);
|
|
412
|
+
const abortSignal = args._signal;
|
|
413
|
+
// Track why findings ended up INCONCLUSIVE so the final result can
|
|
414
|
+
// surface one actionable hint instead of N identical per-finding
|
|
415
|
+
// reasons. The most common case on a developer's laptop is
|
|
416
|
+
// `sandbox-unavailable` (no Docker/Deno installed) — that gets a
|
|
417
|
+
// top-level `verifyHint` with install URLs so the user sees it once,
|
|
418
|
+
// at the top of the result, rather than buried in finding.reason.
|
|
419
|
+
let sandboxUnavailableCount = 0;
|
|
420
|
+
let budgetExhaustedCount = 0;
|
|
191
421
|
const proveable = agentResult.findings.filter(shouldProve);
|
|
192
422
|
if (proveable.length > 0 && progress) {
|
|
193
423
|
progress(0, proveable.length, `Verifying ${proveable.length} finding(s)...`);
|
|
@@ -198,6 +428,20 @@ async function toolAgentScan(ctx, args) {
|
|
|
198
428
|
provenFindings.push({ ...finding, proven: 'SKIPPED', provenReason: 'Low severity — not verified' });
|
|
199
429
|
continue;
|
|
200
430
|
}
|
|
431
|
+
// Pre-flight budget check — stop before spending any LLM calls if
|
|
432
|
+
// the aggregate is already exhausted. This is also enforced inside
|
|
433
|
+
// runVerifyLoop, but checking here lets us mark the finding SKIPPED
|
|
434
|
+
// with a clean reason instead of running INCONCLUSIVE.
|
|
435
|
+
if (!verifyTracker.canAttemptFinding()) {
|
|
436
|
+
budgetExhaustedCount++;
|
|
437
|
+
provenFindings.push({
|
|
438
|
+
...finding,
|
|
439
|
+
proven: 'INCONCLUSIVE',
|
|
440
|
+
provenReason: `Verification budget exhausted (${verifyTracker.findingsAttempted}/${verifyBudget.maxFindings} findings, ${verifyTracker.llmCallsUsed}/${verifyBudget.maxLlmCalls} LLM calls, ${Math.round(verifyTracker.wallClockElapsedMs / 1000)}s/${Math.round(verifyBudget.maxWallClockMs / 1000)}s).`,
|
|
441
|
+
verifySubVerdict: 'budget-exhausted',
|
|
442
|
+
});
|
|
443
|
+
continue;
|
|
444
|
+
}
|
|
201
445
|
if (progress) {
|
|
202
446
|
proveIdx++;
|
|
203
447
|
progress(proveIdx, proveable.length, `Verifying ${finding.type} at line ${finding.line}...`);
|
|
@@ -222,15 +466,77 @@ async function toolAgentScan(ctx, args) {
|
|
|
222
466
|
workspaceRoot: ctx.workspaceRoot,
|
|
223
467
|
language,
|
|
224
468
|
client,
|
|
469
|
+
budgetTracker: verifyTracker,
|
|
470
|
+
signal: abortSignal,
|
|
225
471
|
onProgress: (round, maxR, msg) => {
|
|
226
472
|
if (progress)
|
|
227
473
|
progress(proveIdx, proveable.length, `Verify round ${round}/${maxR}: ${msg}`);
|
|
228
474
|
},
|
|
229
475
|
});
|
|
476
|
+
if (result.subVerdict === 'budget-exhausted')
|
|
477
|
+
budgetExhaustedCount++;
|
|
478
|
+
// User cancelled mid-verify: stop verifying further findings and
|
|
479
|
+
// mark the rest as SKIPPED so the report still includes them.
|
|
480
|
+
if (result.subVerdict === 'cancelled') {
|
|
481
|
+
provenFindings.push({
|
|
482
|
+
...finding,
|
|
483
|
+
proven: result.verdict,
|
|
484
|
+
provenReason: result.reason,
|
|
485
|
+
verifySubVerdict: result.subVerdict,
|
|
486
|
+
});
|
|
487
|
+
for (const remaining of agentResult.findings.slice(agentResult.findings.indexOf(finding) + 1)) {
|
|
488
|
+
provenFindings.push({
|
|
489
|
+
...remaining,
|
|
490
|
+
proven: 'SKIPPED',
|
|
491
|
+
provenReason: 'Scan cancelled by user — not verified.',
|
|
492
|
+
});
|
|
493
|
+
}
|
|
494
|
+
break;
|
|
495
|
+
}
|
|
496
|
+
// No local sandbox (Docker/Deno) on the user's machine. Fall back to
|
|
497
|
+
// the API-side sandbox on Vultr, which has Docker installed. This
|
|
498
|
+
// gives every user exploit verification without a local install.
|
|
499
|
+
if (result.subVerdict === 'sandbox-unavailable') {
|
|
500
|
+
try {
|
|
501
|
+
const proveResp = await client.postJson('/sandbox/prove', {
|
|
502
|
+
code,
|
|
503
|
+
language,
|
|
504
|
+
vulnerabilityType: finding.type,
|
|
505
|
+
line: finding.line,
|
|
506
|
+
lineEnd: finding.lineEnd,
|
|
507
|
+
evidence: finding.evidence,
|
|
508
|
+
why: finding.why,
|
|
509
|
+
});
|
|
510
|
+
provenFindings.push({
|
|
511
|
+
...finding,
|
|
512
|
+
proven: proveResp.proven,
|
|
513
|
+
provenReason: proveResp.rationale || proveResp.skipReason || proveResp.sandbox?.reason,
|
|
514
|
+
verifySubVerdict: result.subVerdict,
|
|
515
|
+
verificationLevel: mapVerificationLevel(proveResp.proven, true, finding.verificationLevel),
|
|
516
|
+
proofEvidence: proveResp.proofEvidence,
|
|
517
|
+
proofGateResult: proveResp.proofGateResult,
|
|
518
|
+
});
|
|
519
|
+
}
|
|
520
|
+
catch (proveErr) {
|
|
521
|
+
sandboxUnavailableCount++;
|
|
522
|
+
provenFindings.push({
|
|
523
|
+
...finding,
|
|
524
|
+
proven: 'INCONCLUSIVE',
|
|
525
|
+
provenReason: result.reason,
|
|
526
|
+
verifySubVerdict: result.subVerdict,
|
|
527
|
+
verificationLevel: mapVerificationLevel('INCONCLUSIVE', false, finding.verificationLevel),
|
|
528
|
+
});
|
|
529
|
+
}
|
|
530
|
+
continue;
|
|
531
|
+
}
|
|
230
532
|
provenFindings.push({
|
|
231
533
|
...finding,
|
|
232
534
|
proven: result.verdict,
|
|
233
535
|
provenReason: result.reason,
|
|
536
|
+
verifySubVerdict: result.subVerdict,
|
|
537
|
+
verificationLevel: mapVerificationLevel(result.verdict, false, finding.verificationLevel),
|
|
538
|
+
proofEvidence: result.proofEvidence,
|
|
539
|
+
proofGateResult: result.proofGateResult,
|
|
234
540
|
});
|
|
235
541
|
}
|
|
236
542
|
catch (err) {
|
|
@@ -249,6 +555,9 @@ async function toolAgentScan(ctx, args) {
|
|
|
249
555
|
...finding,
|
|
250
556
|
proven: proveResp.proven,
|
|
251
557
|
provenReason: proveResp.rationale || proveResp.skipReason || proveResp.sandbox?.reason,
|
|
558
|
+
verificationLevel: mapVerificationLevel(proveResp.proven, true, finding.verificationLevel),
|
|
559
|
+
proofEvidence: proveResp.proofEvidence,
|
|
560
|
+
proofGateResult: proveResp.proofGateResult,
|
|
252
561
|
});
|
|
253
562
|
}
|
|
254
563
|
catch (err2) {
|
|
@@ -256,70 +565,297 @@ async function toolAgentScan(ctx, args) {
|
|
|
256
565
|
...finding,
|
|
257
566
|
proven: 'INCONCLUSIVE',
|
|
258
567
|
provenReason: `Verify failed: ${err.message}; Sandbox fallback also failed: ${err2.message}`,
|
|
568
|
+
verificationLevel: mapVerificationLevel('INCONCLUSIVE', false, finding.verificationLevel),
|
|
259
569
|
});
|
|
260
570
|
}
|
|
261
571
|
}
|
|
262
572
|
}
|
|
263
|
-
//
|
|
264
|
-
|
|
265
|
-
|
|
266
|
-
|
|
573
|
+
// 4a. Capability-based confidence clamping
|
|
574
|
+
// Confidence must reflect what was actually proven, not what the LLM believes.
|
|
575
|
+
// A finding with no structural evidence (no taint trace, no guard check, no
|
|
576
|
+
// verify) cannot be reported at 95% confidence — that's how false positives
|
|
577
|
+
// erode trust in a security tool. The clamp is deterministic and based on:
|
|
578
|
+
// 1. What tools the agent actually used (transcript scan)
|
|
579
|
+
// 2. What tools were available for this language (capability registry)
|
|
580
|
+
// 3. The verify subagent's verdict (PROVEN/UNPROVEN/INCONCLUSIVE)
|
|
581
|
+
clampConfidenceByCapability(provenFindings, agentResult.transcript, language);
|
|
582
|
+
// 4a-ter. Mark findings requiring human review based on proof quality.
|
|
583
|
+
for (const finding of provenFindings) {
|
|
584
|
+
if (finding.proven !== 'PROVEN')
|
|
585
|
+
continue;
|
|
586
|
+
const needsReview = finding.severity === 'critical' ||
|
|
587
|
+
(finding.proofEvidence && finding.proofEvidence.assumptions.length > 0) ||
|
|
588
|
+
!finding.proofEvidence ||
|
|
589
|
+
!finding.proofGateResult ||
|
|
590
|
+
!finding.proofGateResult.eligibleForProven;
|
|
591
|
+
finding.humanReviewRequired = needsReview;
|
|
267
592
|
}
|
|
268
|
-
|
|
269
|
-
|
|
270
|
-
|
|
271
|
-
|
|
272
|
-
|
|
593
|
+
// 4a-bis. Queue INCONCLUSIVE findings for non-blocking human review.
|
|
594
|
+
//
|
|
595
|
+
// Findings the verify subagent couldn't prove or disprove are added to a
|
|
596
|
+
// local review queue (.securecode/finding-review-queue.json). The scan
|
|
597
|
+
// does NOT block — the user can later adjudicate each item via the
|
|
598
|
+
// securecode.review-findings and securecode.decide-finding MCP tools.
|
|
599
|
+
// Only findings with a real file path are queued (inline-code scans have
|
|
600
|
+
// no persistent location to review).
|
|
601
|
+
const reviewQueueAdded = [];
|
|
602
|
+
if (filePath) {
|
|
603
|
+
for (const finding of provenFindings) {
|
|
604
|
+
if (!shouldQueueForHumanReview(finding))
|
|
605
|
+
continue;
|
|
606
|
+
try {
|
|
607
|
+
const reviewItem = (0, findingReviewQueue_1.enqueueFindingReview)(ctx.workspaceRoot, {
|
|
608
|
+
workspaceRelativePath: filePath,
|
|
609
|
+
fileContent: code,
|
|
610
|
+
line: finding.line,
|
|
611
|
+
lineEnd: finding.lineEnd,
|
|
612
|
+
findingType: finding.type,
|
|
613
|
+
severity: finding.severity,
|
|
614
|
+
confidence: finding.confidence,
|
|
615
|
+
proven: finding.proven,
|
|
616
|
+
reviewReason: mapToReviewReason(finding),
|
|
617
|
+
verificationReason: finding.provenReason,
|
|
618
|
+
evidence: finding.evidence || '',
|
|
619
|
+
});
|
|
620
|
+
finding.reviewStatus = 'pending';
|
|
621
|
+
finding.reviewId = reviewItem.id;
|
|
622
|
+
reviewQueueAdded.push(reviewItem.id);
|
|
623
|
+
}
|
|
624
|
+
catch (reviewErr) {
|
|
625
|
+
// Review queue persistence failure must not block the scan.
|
|
626
|
+
console.warn(`[Agent Scan] Review queue enqueue failed: ${reviewErr?.message || reviewErr}`);
|
|
627
|
+
}
|
|
273
628
|
}
|
|
629
|
+
}
|
|
630
|
+
// 4b. Generate fixes for proven/suspected findings (requires approval)
|
|
631
|
+
if (skipFix) {
|
|
632
|
+
console.log('[Agent Scan] Skipping fix generation (_skipFix=true)');
|
|
633
|
+
}
|
|
634
|
+
else {
|
|
635
|
+
const fixableFindings = provenFindings.filter(f => f.proven === 'PROVEN' || (f.proven !== 'UNPROVEN' && f.confidence >= 60));
|
|
636
|
+
if (fixableFindings.length > 0 && progress) {
|
|
637
|
+
progress(0, fixableFindings.length, `Generating fixes for ${fixableFindings.length} finding(s)...`);
|
|
638
|
+
}
|
|
639
|
+
let fixIdx = 0;
|
|
640
|
+
const fixBroker = fixableFindings.length > 0 ? new broker_1.ApprovalBroker() : null;
|
|
641
|
+
if (fixBroker)
|
|
642
|
+
await fixBroker.start();
|
|
274
643
|
try {
|
|
275
|
-
const
|
|
276
|
-
|
|
277
|
-
|
|
278
|
-
|
|
279
|
-
|
|
280
|
-
|
|
281
|
-
|
|
282
|
-
|
|
283
|
-
|
|
284
|
-
|
|
285
|
-
|
|
286
|
-
|
|
287
|
-
|
|
288
|
-
|
|
289
|
-
|
|
290
|
-
|
|
291
|
-
|
|
292
|
-
|
|
644
|
+
for (const finding of fixableFindings) {
|
|
645
|
+
if (progress) {
|
|
646
|
+
fixIdx++;
|
|
647
|
+
progress(fixIdx, fixableFindings.length, `Fixing ${finding.type} at line ${finding.line}...`);
|
|
648
|
+
}
|
|
649
|
+
const fixSummary = `Generate fix for ${finding.type} at line ${finding.line}${finding.lineEnd ? `-${finding.lineEnd}` : ''}\nSeverity: ${finding.severity} | Confidence: ${finding.confidence}%\nEvidence: ${finding.evidence?.substring(0, 200) || '(none)'}`;
|
|
650
|
+
try {
|
|
651
|
+
const approval = await fixBroker.requestApproval('securecode.agent-scan (fix generation)', fixSummary, [code, language, finding.type, finding.line, finding.lineEnd, finding.evidence, finding.severity, finding.confidence], 60_000, 'paid-generation', ctx.workspaceRoot);
|
|
652
|
+
if (!approval.approved) {
|
|
653
|
+
finding.fixStatus = 'fix-denied';
|
|
654
|
+
finding.fixDeniedReason = approval.reason;
|
|
655
|
+
continue;
|
|
656
|
+
}
|
|
657
|
+
const fixResp = await client.postJson('/fix', {
|
|
658
|
+
code,
|
|
659
|
+
language,
|
|
660
|
+
vulnerability: {
|
|
661
|
+
type: finding.type,
|
|
662
|
+
line_start: finding.line,
|
|
663
|
+
line_end: finding.lineEnd || finding.line,
|
|
664
|
+
evidence_snippet: finding.evidence,
|
|
665
|
+
},
|
|
666
|
+
});
|
|
667
|
+
if (fixResp.fixed_code) {
|
|
668
|
+
finding.fix = {
|
|
669
|
+
fixedCode: fixResp.fixed_code,
|
|
670
|
+
replaceRange: { start_line: finding.line, end_line: finding.lineEnd || finding.line },
|
|
671
|
+
fixSummary: fixResp.fix_summary || '',
|
|
672
|
+
importsNeeded: fixResp.imports_needed,
|
|
673
|
+
confidence: fixResp.confidence,
|
|
674
|
+
};
|
|
675
|
+
finding.fixStatus = 'fix-generated';
|
|
676
|
+
finding.fixApprovalId = approval.requestId;
|
|
677
|
+
// 4c. Re-verify the fix — re-run the exploit against the
|
|
678
|
+
// merged fixed code to prove the fix actually closed the
|
|
679
|
+
// vulnerability. Uses a separate, smaller budget so a
|
|
680
|
+
// single fix verification cannot consume the entire
|
|
681
|
+
// original scan verification budget. No second approval
|
|
682
|
+
// needed: the user already approved fix generation, and
|
|
683
|
+
// this runs inside the existing sandbox without modifying
|
|
684
|
+
// any workspace files.
|
|
685
|
+
try {
|
|
686
|
+
if (progress) {
|
|
687
|
+
progress(fixIdx, fixableFindings.length, `Verifying fix for ${finding.type} at line ${finding.line}...`);
|
|
688
|
+
}
|
|
689
|
+
const fixVerifyBudget = (0, agentScanProtocol_1.defaultFixVerifyBudget)();
|
|
690
|
+
const fixVerifyTracker = new agentScanProtocol_1.VerifyBudgetTracker(fixVerifyBudget);
|
|
691
|
+
const fixVerifyResult = await (0, fixVerifyLoop_1.runFixVerifyLoop)({
|
|
692
|
+
finding: {
|
|
693
|
+
type: finding.type,
|
|
694
|
+
line: finding.line,
|
|
695
|
+
lineEnd: finding.lineEnd,
|
|
696
|
+
evidence: finding.evidence,
|
|
697
|
+
why: finding.why,
|
|
698
|
+
severity: finding.severity,
|
|
699
|
+
},
|
|
700
|
+
originalCode: code,
|
|
701
|
+
fixedCode: fixResp.fixed_code,
|
|
702
|
+
replaceRange: { start_line: finding.line, end_line: finding.lineEnd || finding.line },
|
|
703
|
+
filePath: filePath || '',
|
|
704
|
+
relatedFiles: relatedFiles.map(rf => ({
|
|
705
|
+
filePath: rf.filePath,
|
|
706
|
+
content: rf.content,
|
|
707
|
+
relationship: rf.relationship,
|
|
708
|
+
})),
|
|
709
|
+
workspaceRoot: ctx.workspaceRoot,
|
|
710
|
+
language,
|
|
711
|
+
client,
|
|
712
|
+
originalVerdict: finding.proven,
|
|
713
|
+
budgetTracker: fixVerifyTracker,
|
|
714
|
+
signal: abortSignal,
|
|
715
|
+
onProgress: (round, maxR, msg) => {
|
|
716
|
+
if (progress)
|
|
717
|
+
progress(fixIdx, fixableFindings.length, `Fix verify round ${round}/${maxR}: ${msg}`);
|
|
718
|
+
},
|
|
719
|
+
});
|
|
720
|
+
finding.fixVerification = fixVerifyResult;
|
|
721
|
+
// Map the fix verification status to the fixStatus field.
|
|
722
|
+
switch (fixVerifyResult.status) {
|
|
723
|
+
case 'closed':
|
|
724
|
+
finding.fixStatus = 'fix-verified-closed';
|
|
725
|
+
break;
|
|
726
|
+
case 'still-vulnerable':
|
|
727
|
+
finding.fixStatus = 'fix-still-vulnerable';
|
|
728
|
+
break;
|
|
729
|
+
case 'inconclusive':
|
|
730
|
+
finding.fixStatus = 'fix-verification-inconclusive';
|
|
731
|
+
break;
|
|
732
|
+
case 'syntax-invalid':
|
|
733
|
+
finding.fixStatus = 'fix-syntax-invalid';
|
|
734
|
+
break;
|
|
735
|
+
// sandbox-unavailable and cancelled leave fixStatus as 'fix-generated'
|
|
736
|
+
}
|
|
737
|
+
}
|
|
738
|
+
catch (fixVerifyErr) {
|
|
739
|
+
// Fix verification failure must not block the scan — the
|
|
740
|
+
// fix is still generated, just not re-verified.
|
|
741
|
+
console.warn(`[Agent Scan] Fix verification failed for ${finding.type} at L${finding.line}: ${fixVerifyErr?.message || fixVerifyErr}`);
|
|
742
|
+
}
|
|
743
|
+
}
|
|
744
|
+
}
|
|
745
|
+
catch (err) {
|
|
746
|
+
finding.fixStatus = 'fix-error';
|
|
747
|
+
finding.fixDeniedReason = err.message;
|
|
748
|
+
console.warn(`[Agent Scan] Fix generation failed for ${finding.type} at L${finding.line}: ${err.message}`);
|
|
749
|
+
}
|
|
293
750
|
}
|
|
294
751
|
}
|
|
295
|
-
|
|
296
|
-
|
|
297
|
-
|
|
752
|
+
finally {
|
|
753
|
+
if (fixBroker)
|
|
754
|
+
await fixBroker.stop();
|
|
298
755
|
}
|
|
299
|
-
}
|
|
756
|
+
} // end else (skipFix)
|
|
300
757
|
// 5. Write to cache before returning
|
|
301
758
|
if (useCache) {
|
|
302
759
|
try {
|
|
303
760
|
(0, scanCache_1.writeCachedScan)(ctx.workspaceRoot, filePath, code, {
|
|
304
761
|
findings: provenFindings,
|
|
762
|
+
investigationNotes: agentResult.investigationNotes,
|
|
763
|
+
coverageGaps: agentResult.coverageGaps,
|
|
305
764
|
status: agentResult.status,
|
|
306
765
|
summary: agentResult.summary,
|
|
307
766
|
stepsUsed: agentResult.stepsUsed,
|
|
767
|
+
stepsGranted: agentResult.stepsGranted,
|
|
768
|
+
extensionsGranted: agentResult.extensionsGranted,
|
|
308
769
|
costSpentUsd: agentResult.costSpentUsd,
|
|
309
|
-
});
|
|
770
|
+
}, memoryFingerprint);
|
|
310
771
|
}
|
|
311
772
|
catch (err) {
|
|
312
773
|
console.warn(`[Agent Scan] Cache write skipped: ${err?.message || err}`);
|
|
313
774
|
}
|
|
314
775
|
}
|
|
315
776
|
// 6. Return result — the verify subagent IS the verifier (no separate Juror call)
|
|
777
|
+
//
|
|
778
|
+
// verifyHint: surfaced once at the top level when one or more findings
|
|
779
|
+
// couldn't be exploit-verified. The local sandbox (Docker/Deno) was
|
|
780
|
+
// unavailable AND the API-side sandbox fallback failed — so the finding
|
|
781
|
+
// got INCONCLUSIVE. The hint tells the user what to install for local
|
|
782
|
+
// verification (faster, no round-trip) and reminds them the API sandbox
|
|
783
|
+
// is the automatic fallback.
|
|
784
|
+
let verifyHint;
|
|
785
|
+
if (sandboxUnavailableCount > 0) {
|
|
786
|
+
verifyHint = `Exploit verification was skipped for ${sandboxUnavailableCount} finding(s). No local sandbox (Docker or Deno) was detected and the API-side sandbox was unavailable. Findings are reported as INCONCLUSIVE with confidence capped at 75%.\n${localTestRunner_1.SANDBOX_UNAVAILABLE_MESSAGE}`;
|
|
787
|
+
}
|
|
788
|
+
else if (budgetExhaustedCount > 0) {
|
|
789
|
+
verifyHint = `Exploit verification was skipped for ${budgetExhaustedCount} finding(s) because the per-scan verification budget (${verifyBudget.maxFindings} findings, ${verifyBudget.maxLlmCalls} LLM calls, ${Math.round(verifyBudget.maxWallClockMs / 1000)}s) was exhausted. Re-run the scan to verify the remaining findings, or raise the budget via the VerifyBudget config.`;
|
|
790
|
+
}
|
|
791
|
+
// 6a. Record metadata-only audit sample (no source code, no evidence strings)
|
|
792
|
+
try {
|
|
793
|
+
const fileHash = require('crypto').createHash('sha256').update(code).digest('hex').substring(0, 16);
|
|
794
|
+
(0, scanAuditLog_1.recordScanAuditSample)(ctx.workspaceRoot, {
|
|
795
|
+
filePath: filePath || 'inline-code',
|
|
796
|
+
fileHash,
|
|
797
|
+
language,
|
|
798
|
+
scanStatus: agentResult.status,
|
|
799
|
+
stepsUsed: agentResult.stepsUsed,
|
|
800
|
+
costSpentUsd: agentResult.costSpentUsd,
|
|
801
|
+
agentFindings: provenFindings,
|
|
802
|
+
transcript: agentResult.transcript,
|
|
803
|
+
cached: false,
|
|
804
|
+
verifyUsage: {
|
|
805
|
+
findingsAttempted: verifyTracker.findingsAttempted,
|
|
806
|
+
roundsUsed: verifyTracker.roundsUsed,
|
|
807
|
+
llmCallsUsed: verifyTracker.llmCallsUsed,
|
|
808
|
+
costSpentUsd: verifyTracker.costSpentUsd,
|
|
809
|
+
wallClockMs: verifyTracker.wallClockElapsedMs,
|
|
810
|
+
},
|
|
811
|
+
scope,
|
|
812
|
+
investigationNotes: agentResult.investigationNotes,
|
|
813
|
+
coverageGaps: agentResult.coverageGaps,
|
|
814
|
+
hasArchitectureContext: !!architectureContextStr,
|
|
815
|
+
stepsGranted: agentResult.stepsGranted,
|
|
816
|
+
extensionsGranted: agentResult.extensionsGranted,
|
|
817
|
+
terminationReason: agentResult.terminationReason,
|
|
818
|
+
});
|
|
819
|
+
}
|
|
820
|
+
catch {
|
|
821
|
+
// best-effort — audit failure must not block scan results
|
|
822
|
+
}
|
|
823
|
+
// 6b. Persist investigation notes and coverage gaps to workspace memory
|
|
824
|
+
// so future scans can use them as context. These are NOT findings —
|
|
825
|
+
// they guide future investigation without suppressing new discoveries.
|
|
826
|
+
try {
|
|
827
|
+
const fileHashes = new Map();
|
|
828
|
+
if (filePath) {
|
|
829
|
+
fileHashes.set(filePath, require('crypto').createHash('sha256').update(code).digest('hex').substring(0, 16));
|
|
830
|
+
}
|
|
831
|
+
if (agentResult.investigationNotes && agentResult.investigationNotes.length > 0) {
|
|
832
|
+
(0, agentMemory_1.saveInvestigationNotes)(ctx.workspaceRoot, {
|
|
833
|
+
notes: agentResult.investigationNotes,
|
|
834
|
+
fileHashes,
|
|
835
|
+
});
|
|
836
|
+
}
|
|
837
|
+
if (agentResult.coverageGaps && agentResult.coverageGaps.length > 0) {
|
|
838
|
+
(0, agentMemory_1.saveCoverageGaps)(ctx.workspaceRoot, {
|
|
839
|
+
gaps: agentResult.coverageGaps,
|
|
840
|
+
fileHashes,
|
|
841
|
+
});
|
|
842
|
+
}
|
|
843
|
+
}
|
|
844
|
+
catch {
|
|
845
|
+
// best-effort — memory persistence failure must not block results
|
|
846
|
+
}
|
|
316
847
|
return {
|
|
317
848
|
status: agentResult.status,
|
|
318
849
|
summary: agentResult.summary,
|
|
319
850
|
agentFindings: provenFindings,
|
|
320
851
|
verifiedFindings: provenFindings.filter(f => f.proven === 'PROVEN'),
|
|
321
852
|
allFindings: [],
|
|
853
|
+
investigationNotes: agentResult.investigationNotes ?? [],
|
|
854
|
+
coverageGaps: agentResult.coverageGaps ?? [],
|
|
322
855
|
stepsUsed: agentResult.stepsUsed,
|
|
856
|
+
stepsGranted: agentResult.stepsGranted,
|
|
857
|
+
extensionsGranted: agentResult.extensionsGranted,
|
|
858
|
+
terminationReason: agentResult.terminationReason,
|
|
323
859
|
costSpentUsd: agentResult.costSpentUsd,
|
|
324
860
|
transcript: agentResult.transcript,
|
|
325
861
|
provenCount: provenFindings.filter(f => f.proven === 'PROVEN').length,
|
|
@@ -327,6 +863,129 @@ async function toolAgentScan(ctx, args) {
|
|
|
327
863
|
inconclusiveCount: provenFindings.filter(f => f.proven === 'INCONCLUSIVE').length,
|
|
328
864
|
notReproducibleCount: provenFindings.filter(f => f.proven === 'NOT_REPRODUCIBLE').length,
|
|
329
865
|
skippedCount: provenFindings.filter(f => f.proven === 'SKIPPED').length,
|
|
866
|
+
verifyHint,
|
|
867
|
+
verifyUsage: {
|
|
868
|
+
findingsAttempted: verifyTracker.findingsAttempted,
|
|
869
|
+
roundsUsed: verifyTracker.roundsUsed,
|
|
870
|
+
llmCallsUsed: verifyTracker.llmCallsUsed,
|
|
871
|
+
costSpentUsd: verifyTracker.costSpentUsd,
|
|
872
|
+
wallClockMs: verifyTracker.wallClockElapsedMs,
|
|
873
|
+
budget: verifyBudget,
|
|
874
|
+
},
|
|
875
|
+
reviewQueue: {
|
|
876
|
+
added: reviewQueueAdded,
|
|
877
|
+
pendingCount: provenFindings.filter(f => f.reviewStatus === 'pending').length,
|
|
878
|
+
},
|
|
330
879
|
};
|
|
331
880
|
}
|
|
881
|
+
function clampConfidenceByCapability(findings, transcript, language) {
|
|
882
|
+
const cap = (0, capabilityRegistry_1.getCapability)(language);
|
|
883
|
+
let usedTaint = false;
|
|
884
|
+
let usedGuard = false;
|
|
885
|
+
let usedPolicy = false;
|
|
886
|
+
for (const step of transcript) {
|
|
887
|
+
const t = step.action.type;
|
|
888
|
+
if (t === 'trace_flow' || t === 'trace_flow_cross_file')
|
|
889
|
+
usedTaint = true;
|
|
890
|
+
if (t === 'check_guard')
|
|
891
|
+
usedGuard = true;
|
|
892
|
+
if (t === 'check_policy')
|
|
893
|
+
usedPolicy = true;
|
|
894
|
+
}
|
|
895
|
+
let evidenceTools = 0;
|
|
896
|
+
if (usedTaint)
|
|
897
|
+
evidenceTools++;
|
|
898
|
+
if (usedGuard)
|
|
899
|
+
evidenceTools++;
|
|
900
|
+
if (usedPolicy)
|
|
901
|
+
evidenceTools++;
|
|
902
|
+
for (const f of findings) {
|
|
903
|
+
const original = f.confidence;
|
|
904
|
+
const ceiling = confidenceCeilingForFinding(f.proven, cap.tier, evidenceTools);
|
|
905
|
+
// The ceiling is the maximum allowed confidence for this verdict +
|
|
906
|
+
// capability combination. PROVEN findings have no ceiling (undefined)
|
|
907
|
+
// — they're already proven, let the LLM's confidence stand (but still
|
|
908
|
+
// floor at 80 because PROVEN should look confident).
|
|
909
|
+
let clamped = original;
|
|
910
|
+
if (ceiling !== undefined) {
|
|
911
|
+
clamped = Math.min(clamped, ceiling);
|
|
912
|
+
}
|
|
913
|
+
// PROVEN floor — a finding verified by exploit test should not
|
|
914
|
+
// display with a wishy-washy 40% confidence; if the agent is unsure
|
|
915
|
+
// about a PROVEN finding, the proof overrules the agent's doubt.
|
|
916
|
+
if (f.proven === 'PROVEN') {
|
|
917
|
+
clamped = Math.max(clamped, 80);
|
|
918
|
+
}
|
|
919
|
+
clamped = Math.round(clamped);
|
|
920
|
+
f.evidenceLevel = (0, capabilityRegistry_1.evidenceLevelTag)(usedTaint, usedGuard, usedPolicy, f.proven);
|
|
921
|
+
if (clamped !== original) {
|
|
922
|
+
f.originalConfidence = original;
|
|
923
|
+
f.confidence = clamped;
|
|
924
|
+
}
|
|
925
|
+
}
|
|
926
|
+
}
|
|
927
|
+
/**
|
|
928
|
+
* Explicit confidence-ceiling policy for a finding, based on:
|
|
929
|
+
* - the verify verdict (PROVEN / UNPROVEN / INCONCLUSIVE /
|
|
930
|
+
* NOT_REPRODUCIBLE / SKIPPED)
|
|
931
|
+
* - the language capability tier (deep / standard / fallback)
|
|
932
|
+
* - how many structural evidence tools the agent actually used
|
|
933
|
+
* (trace_flow, check_guard, check_policy)
|
|
934
|
+
*
|
|
935
|
+
* Returns `undefined` for PROVEN (no ceiling — the proof overrules the
|
|
936
|
+
* LLM's belief). For everything else, returns the maximum allowed
|
|
937
|
+
* confidence. The caller takes `Math.min(confidence, ceiling)` so multiple
|
|
938
|
+
* applicable bounds compose correctly — last-write-wins is impossible.
|
|
939
|
+
*
|
|
940
|
+
* Verdict semantics:
|
|
941
|
+
* - PROVEN: exploit test ran and reproduced the vulnerability. Strong
|
|
942
|
+
* evidence. No ceiling; floor at 80.
|
|
943
|
+
* - UNPROVEN: exploit test ran and did NOT reproduce. The finding is
|
|
944
|
+
* likely a false positive. Hard cap at 25.
|
|
945
|
+
* - NOT_REPRODUCIBLE: exploit test ran, but the test setup couldn't
|
|
946
|
+
* trigger the vulnerability (e.g. the path requires a live DB). This
|
|
947
|
+
* is NOT the same as UNPROVEN (which actively disproved) and NOT the
|
|
948
|
+
* same as INCONCLUSIVE (which couldn't even run a test). Treat it as
|
|
949
|
+
* "we tried and couldn't confirm" — cap at 35, tighter than
|
|
950
|
+
* INCONCLUSIVE but looser than UNPROVEN.
|
|
951
|
+
* - INCONCLUSIVE: no test could be generated or the sandbox was
|
|
952
|
+
* unavailable. The finding is unverified — keep the LLM's confidence
|
|
953
|
+
* but cap it based on structural evidence (no tools → ≤40, ≥2 tools
|
|
954
|
+
* → ≤75). Fallback-tier languages get ≤55 across the board.
|
|
955
|
+
* - SKIPPED: low severity, we didn't even try. Same as INCONCLUSIVE.
|
|
956
|
+
*/
|
|
957
|
+
function confidenceCeilingForFinding(verdict, capabilityTier, evidenceTools) {
|
|
958
|
+
switch (verdict) {
|
|
959
|
+
case 'PROVEN':
|
|
960
|
+
return undefined;
|
|
961
|
+
case 'UNPROVEN':
|
|
962
|
+
return 25;
|
|
963
|
+
case 'NOT_REPRODUCIBLE':
|
|
964
|
+
return 35;
|
|
965
|
+
case 'INCONCLUSIVE':
|
|
966
|
+
case 'SKIPPED':
|
|
967
|
+
if (capabilityTier === 'fallback')
|
|
968
|
+
return 55;
|
|
969
|
+
if (evidenceTools === 0)
|
|
970
|
+
return 40;
|
|
971
|
+
if (evidenceTools === 1)
|
|
972
|
+
return 60;
|
|
973
|
+
return 75;
|
|
974
|
+
default:
|
|
975
|
+
return 40;
|
|
976
|
+
}
|
|
977
|
+
}
|
|
978
|
+
/**
|
|
979
|
+
* Apply the confidence clamp to a single finding. Exported for testing
|
|
980
|
+
* so the test imports the real policy, not a copy that can drift.
|
|
981
|
+
*/
|
|
982
|
+
function applyConfidenceClamp(original, verdict, capabilityTier, evidenceTools) {
|
|
983
|
+
const ceiling = confidenceCeilingForFinding(verdict, capabilityTier, evidenceTools);
|
|
984
|
+
let clamped = original;
|
|
985
|
+
if (ceiling !== undefined)
|
|
986
|
+
clamped = Math.min(clamped, ceiling);
|
|
987
|
+
if (verdict === 'PROVEN')
|
|
988
|
+
clamped = Math.max(clamped, 80);
|
|
989
|
+
return Math.round(clamped);
|
|
990
|
+
}
|
|
332
991
|
//# sourceMappingURL=agentScan.js.map
|