@securecode-ai/mcp 0.5.4 → 0.5.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/api/types.d.ts +7 -0
- package/dist/approval/auditLog.d.ts +1 -1
- package/dist/approval/auditLog.js +11 -3
- package/dist/approval/auditLog.js.map +1 -1
- package/dist/approval/broker.d.ts +7 -2
- package/dist/approval/broker.js +239 -110
- package/dist/approval/broker.js.map +1 -1
- package/dist/approval/policy.d.ts +10 -0
- package/dist/approval/policy.js +104 -0
- package/dist/approval/policy.js.map +1 -0
- package/dist/approval/types.d.ts +11 -2
- package/dist/approval/types.js +10 -1
- package/dist/approval/types.js.map +1 -1
- package/dist/attack/agentScanExecutor.d.ts +35 -1
- package/dist/attack/agentScanExecutor.js +405 -76
- package/dist/attack/agentScanExecutor.js.map +1 -1
- package/dist/attack/agentScanLoop.d.ts +20 -0
- package/dist/attack/agentScanLoop.js +1264 -60
- package/dist/attack/agentScanLoop.js.map +1 -1
- package/dist/attack/agentScanProtocol.d.ts +382 -6
- package/dist/attack/agentScanProtocol.js +110 -4
- package/dist/attack/agentScanProtocol.js.map +1 -1
- package/dist/attack/agentTrace.d.ts +73 -0
- package/dist/attack/agentTrace.js +228 -0
- package/dist/attack/agentTrace.js.map +1 -0
- package/dist/attack/architectureScoutExecutor.d.ts +18 -0
- package/dist/attack/architectureScoutExecutor.js +211 -0
- package/dist/attack/architectureScoutExecutor.js.map +1 -0
- package/dist/attack/architectureScoutLoop.d.ts +36 -0
- package/dist/attack/architectureScoutLoop.js +403 -0
- package/dist/attack/architectureScoutLoop.js.map +1 -0
- package/dist/attack/architectureScoutProtocol.d.ts +209 -0
- package/dist/attack/architectureScoutProtocol.js +47 -0
- package/dist/attack/architectureScoutProtocol.js.map +1 -0
- package/dist/attack/candidateStore.d.ts +95 -0
- package/dist/attack/candidateStore.js +231 -0
- package/dist/attack/candidateStore.js.map +1 -0
- package/dist/attack/evidenceLedger.d.ts +67 -0
- package/dist/attack/evidenceLedger.js +192 -0
- package/dist/attack/evidenceLedger.js.map +1 -0
- package/dist/attack/finishGate.d.ts +62 -0
- package/dist/attack/finishGate.js +209 -0
- package/dist/attack/finishGate.js.map +1 -0
- package/dist/attack/fixCodeMerge.d.ts +27 -0
- package/dist/attack/fixCodeMerge.js +42 -0
- package/dist/attack/fixCodeMerge.js.map +1 -0
- package/dist/attack/fixVerifyLoop.d.ts +55 -0
- package/dist/attack/fixVerifyLoop.js +187 -0
- package/dist/attack/fixVerifyLoop.js.map +1 -0
- package/dist/attack/investigationProfiles.d.ts +36 -0
- package/dist/attack/investigationProfiles.js +144 -0
- package/dist/attack/investigationProfiles.js.map +1 -0
- package/dist/attack/investigationState.d.ts +227 -0
- package/dist/attack/investigationState.js +666 -0
- package/dist/attack/investigationState.js.map +1 -0
- package/dist/attack/mutationOperators.d.ts +22 -0
- package/dist/attack/mutationOperators.js +170 -0
- package/dist/attack/mutationOperators.js.map +1 -0
- package/dist/attack/mutationTest.d.ts +33 -0
- package/dist/attack/mutationTest.js +110 -0
- package/dist/attack/mutationTest.js.map +1 -0
- package/dist/attack/proofGate.d.ts +20 -0
- package/dist/attack/proofGate.js +104 -0
- package/dist/attack/proofGate.js.map +1 -0
- package/dist/attack/proofTypes.d.ts +57 -0
- package/dist/attack/proofTypes.js +32 -0
- package/dist/attack/proofTypes.js.map +1 -0
- package/dist/attack/protocolValidator.d.ts +50 -0
- package/dist/attack/protocolValidator.js +427 -0
- package/dist/attack/protocolValidator.js.map +1 -0
- package/dist/attack/qualityMetrics.d.ts +133 -0
- package/dist/attack/qualityMetrics.js +226 -0
- package/dist/attack/qualityMetrics.js.map +1 -0
- package/dist/attack/scanScheduler.d.ts +52 -0
- package/dist/attack/scanScheduler.js +265 -0
- package/dist/attack/scanScheduler.js.map +1 -0
- package/dist/attack/scanState.d.ts +58 -0
- package/dist/attack/scanState.js +102 -0
- package/dist/attack/scanState.js.map +1 -0
- package/dist/attack/searchIntent.d.ts +16 -0
- package/dist/attack/searchIntent.js +57 -0
- package/dist/attack/searchIntent.js.map +1 -0
- package/dist/attack/verifyLoop.d.ts +43 -0
- package/dist/attack/verifyLoop.js +329 -15
- package/dist/attack/verifyLoop.js.map +1 -1
- package/dist/attack/workItem.d.ts +57 -0
- package/dist/attack/workItem.js +225 -0
- package/dist/attack/workItem.js.map +1 -0
- package/dist/audit/findingReviewQueue.d.ts +109 -0
- package/dist/audit/findingReviewQueue.js +335 -0
- package/dist/audit/findingReviewQueue.js.map +1 -0
- package/dist/audit/scanAuditLog.d.ts +160 -0
- package/dist/audit/scanAuditLog.js +404 -0
- package/dist/audit/scanAuditLog.js.map +1 -0
- package/dist/dependency/dependencyChecker.js +35 -9
- package/dist/dependency/dependencyChecker.js.map +1 -1
- package/dist/dependency/exploitPriority.d.ts +33 -0
- package/dist/dependency/exploitPriority.js +78 -0
- package/dist/dependency/exploitPriority.js.map +1 -0
- package/dist/dependency/finding.d.ts +16 -0
- package/dist/dependency/osvClient.js +2 -0
- package/dist/dependency/osvClient.js.map +1 -1
- package/dist/dependency/types.d.ts +12 -1
- package/dist/mcp/server.js +13 -2
- package/dist/mcp/server.js.map +1 -1
- package/dist/mcp/tools.d.ts +1 -0
- package/dist/mcp/tools.js +127 -6
- package/dist/mcp/tools.js.map +1 -1
- package/dist/project-map/agentMemory.d.ts +83 -1
- package/dist/project-map/agentMemory.js +197 -7
- package/dist/project-map/agentMemory.js.map +1 -1
- package/dist/project-map/architectureContext.d.ts +202 -0
- package/dist/project-map/architectureContext.js +421 -0
- package/dist/project-map/architectureContext.js.map +1 -0
- package/dist/project-map/blastRadius.d.ts +54 -0
- package/dist/project-map/blastRadius.js +196 -0
- package/dist/project-map/blastRadius.js.map +1 -0
- package/dist/project-map/callGraphExtractor.d.ts +13 -0
- package/dist/project-map/callGraphExtractor.js +222 -0
- package/dist/project-map/callGraphExtractor.js.map +1 -0
- package/dist/project-map/capabilityRegistry.d.ts +46 -0
- package/dist/project-map/capabilityRegistry.js +91 -0
- package/dist/project-map/capabilityRegistry.js.map +1 -0
- package/dist/project-map/findTests.d.ts +29 -0
- package/dist/project-map/findTests.js +236 -0
- package/dist/project-map/findTests.js.map +1 -0
- package/dist/project-map/handlerInventory.d.ts +112 -0
- package/dist/project-map/handlerInventory.js +184 -0
- package/dist/project-map/handlerInventory.js.map +1 -0
- package/dist/project-map/implementationResolver.d.ts +43 -0
- package/dist/project-map/implementationResolver.js +217 -0
- package/dist/project-map/implementationResolver.js.map +1 -0
- package/dist/project-map/mapContext.d.ts +6 -1
- package/dist/project-map/mapContext.js +1 -0
- package/dist/project-map/mapContext.js.map +1 -1
- package/dist/project-map/scanCache.d.ts +57 -3
- package/dist/project-map/scanCache.js +61 -2
- package/dist/project-map/scanCache.js.map +1 -1
- package/dist/project-map/symbolIndex.d.ts +39 -0
- package/dist/project-map/symbolIndex.js +385 -0
- package/dist/project-map/symbolIndex.js.map +1 -0
- package/dist/tooling/agentEvalScoring.d.ts +77 -0
- package/dist/tooling/agentEvalScoring.js +140 -0
- package/dist/tooling/agentEvalScoring.js.map +1 -0
- package/dist/tooling/agentRegression.d.ts +53 -0
- package/dist/tooling/agentRegression.js +99 -0
- package/dist/tooling/agentRegression.js.map +1 -0
- package/dist/tools/agentScan.d.ts +69 -0
- package/dist/tools/agentScan.js +702 -43
- package/dist/tools/agentScan.js.map +1 -1
- package/dist/tools/findingReviewTools.d.ts +22 -0
- package/dist/tools/findingReviewTools.js +121 -0
- package/dist/tools/findingReviewTools.js.map +1 -0
- package/dist/tools/fix.js +1 -1
- package/dist/tools/fix.js.map +1 -1
- package/dist/tools/map.js +191 -1
- package/dist/tools/map.js.map +1 -1
- package/dist/tools/runTests.d.ts +2 -0
- package/dist/tools/runTests.js +29 -0
- package/dist/tools/runTests.js.map +1 -0
- package/dist/utils/effectMock.d.ts +1 -0
- package/dist/utils/effectMock.js +162 -0
- package/dist/utils/effectMock.js.map +1 -0
- package/dist/utils/gitContext.d.ts +26 -0
- package/dist/utils/gitContext.js +317 -0
- package/dist/utils/gitContext.js.map +1 -0
- package/dist/utils/localTestRunner.d.ts +57 -2
- package/dist/utils/localTestRunner.js +122 -101
- package/dist/utils/localTestRunner.js.map +1 -1
- package/dist/utils/runtimeDetect.d.ts +12 -2
- package/dist/utils/runtimeDetect.js +247 -10
- package/dist/utils/runtimeDetect.js.map +1 -1
- package/dist/utils/securityConfig.d.ts +17 -0
- package/dist/utils/securityConfig.js +192 -0
- package/dist/utils/securityConfig.js.map +1 -0
- package/dist/utils/testCommandPolicy.d.ts +36 -0
- package/dist/utils/testCommandPolicy.js +252 -0
- package/dist/utils/testCommandPolicy.js.map +1 -0
- package/dist/utils/testRunner.d.ts +56 -0
- package/dist/utils/testRunner.js +246 -0
- package/dist/utils/testRunner.js.map +1 -0
- package/dist/utils/testSafety.d.ts +40 -2
- package/dist/utils/testSafety.js +205 -26
- package/dist/utils/testSafety.js.map +1 -1
- package/dist/utils/verificationSandbox.d.ts +113 -0
- package/dist/utils/verificationSandbox.js +656 -0
- package/dist/utils/verificationSandbox.js.map +1 -0
- package/package.json +1 -1
|
@@ -0,0 +1,102 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
/**
|
|
3
|
+
* Authoritative scan state — the single source of truth for a scan run's
|
|
4
|
+
* phase, budget, recovery, and lifecycle.
|
|
5
|
+
*
|
|
6
|
+
* The MCP owns this state. The model proposes actions; the MCP validates
|
|
7
|
+
* them against this state and owns all transitions.
|
|
8
|
+
*
|
|
9
|
+
* This replaces scattered booleans, counters, and checklist sets with a
|
|
10
|
+
* serializable state object that can be snapshotted for tracing and audit.
|
|
11
|
+
*/
|
|
12
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
13
|
+
exports.createScanRunState = createScanRunState;
|
|
14
|
+
exports.transitionScanPhase = transitionScanPhase;
|
|
15
|
+
exports.terminateScan = terminateScan;
|
|
16
|
+
exports.isScanTerminal = isScanTerminal;
|
|
17
|
+
exports.hasBudgetRemaining = hasBudgetRemaining;
|
|
18
|
+
exports.hasWallClockRemaining = hasWallClockRemaining;
|
|
19
|
+
exports.canExecuteWork = canExecuteWork;
|
|
20
|
+
exports.snapshotState = snapshotState;
|
|
21
|
+
const VALID_TRANSITIONS = {
|
|
22
|
+
planning: ['surveying', 'terminated'],
|
|
23
|
+
surveying: ['tracing', 'terminated'],
|
|
24
|
+
tracing: ['candidate_investigation', 'finish_review', 'terminated'],
|
|
25
|
+
candidate_investigation: ['finish_review', 'tracing', 'terminated'],
|
|
26
|
+
finish_review: ['verifying', 'evidence_recovery', 'surveying', 'tracing', 'terminated'],
|
|
27
|
+
verifying: ['evidence_recovery', 'finalizing', 'terminated'],
|
|
28
|
+
evidence_recovery: ['verifying', 'tracing', 'candidate_investigation', 'terminated'],
|
|
29
|
+
finalizing: ['completed', 'terminated'],
|
|
30
|
+
completed: [],
|
|
31
|
+
terminated: [],
|
|
32
|
+
};
|
|
33
|
+
function createScanRunState(runId, target, budget) {
|
|
34
|
+
return {
|
|
35
|
+
runId,
|
|
36
|
+
phase: 'planning',
|
|
37
|
+
target,
|
|
38
|
+
transcript: [],
|
|
39
|
+
budget: {
|
|
40
|
+
stepsUsed: 0,
|
|
41
|
+
stepsRemaining: budget.stepsRemaining,
|
|
42
|
+
stepsGranted: budget.stepsGranted,
|
|
43
|
+
hardMaxSteps: budget.hardMaxSteps,
|
|
44
|
+
extensionsGranted: budget.extensionsGranted,
|
|
45
|
+
costSpentUsd: 0,
|
|
46
|
+
costCapUsd: budget.costCapUsd,
|
|
47
|
+
startedAt: Date.now(),
|
|
48
|
+
wallClockMs: 720_000,
|
|
49
|
+
},
|
|
50
|
+
recovery: {
|
|
51
|
+
consecutiveBlockedActions: 0,
|
|
52
|
+
totalBlockedActions: 0,
|
|
53
|
+
recoveryAttempts: 0,
|
|
54
|
+
meaningfulProgressSinceRecovery: true,
|
|
55
|
+
},
|
|
56
|
+
finishAttempts: 0,
|
|
57
|
+
verificationCycles: 0,
|
|
58
|
+
};
|
|
59
|
+
}
|
|
60
|
+
function transitionScanPhase(state, next, reason) {
|
|
61
|
+
const allowed = VALID_TRANSITIONS[state.phase];
|
|
62
|
+
if (!allowed.includes(next)) {
|
|
63
|
+
throw new Error(`Invalid scan phase transition: ${state.phase} -> ${next} (${reason}). ` +
|
|
64
|
+
`Valid transitions: ${allowed.join(', ')}`);
|
|
65
|
+
}
|
|
66
|
+
state.phase = next;
|
|
67
|
+
}
|
|
68
|
+
function terminateScan(state, reason, summary) {
|
|
69
|
+
state.termination = {
|
|
70
|
+
reason,
|
|
71
|
+
phase: state.phase,
|
|
72
|
+
timestamp: Date.now(),
|
|
73
|
+
summary,
|
|
74
|
+
};
|
|
75
|
+
state.phase = 'terminated';
|
|
76
|
+
}
|
|
77
|
+
function isScanTerminal(state) {
|
|
78
|
+
return state.phase === 'completed' || state.phase === 'terminated';
|
|
79
|
+
}
|
|
80
|
+
function hasBudgetRemaining(state) {
|
|
81
|
+
return state.budget.stepsRemaining > 0 &&
|
|
82
|
+
state.budget.costSpentUsd < state.budget.costCapUsd;
|
|
83
|
+
}
|
|
84
|
+
function hasWallClockRemaining(state) {
|
|
85
|
+
return Date.now() - state.budget.startedAt < state.budget.wallClockMs;
|
|
86
|
+
}
|
|
87
|
+
function canExecuteWork(state) {
|
|
88
|
+
return hasBudgetRemaining(state) && hasWallClockRemaining(state) && !isScanTerminal(state);
|
|
89
|
+
}
|
|
90
|
+
function snapshotState(state) {
|
|
91
|
+
return JSON.stringify({
|
|
92
|
+
runId: state.runId,
|
|
93
|
+
phase: state.phase,
|
|
94
|
+
stepsUsed: state.budget.stepsUsed,
|
|
95
|
+
stepsRemaining: state.budget.stepsRemaining,
|
|
96
|
+
costSpentUsd: state.budget.costSpentUsd,
|
|
97
|
+
finishAttempts: state.finishAttempts,
|
|
98
|
+
totalBlockedActions: state.recovery.totalBlockedActions,
|
|
99
|
+
termination: state.termination,
|
|
100
|
+
});
|
|
101
|
+
}
|
|
102
|
+
//# sourceMappingURL=scanState.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"scanState.js","sourceRoot":"","sources":["../../src/attack/scanState.ts"],"names":[],"mappings":";AAAA;;;;;;;;;GASG;;AAgFH,gDA8BC;AAED,kDAaC;AAED,sCAYC;AAED,wCAEC;AAED,gDAGC;AAED,sDAEC;AAED,wCAEC;AAED,sCAWC;AAtGD,MAAM,iBAAiB,GAAmC;IACtD,QAAQ,EAAE,CAAC,WAAW,EAAE,YAAY,CAAC;IACrC,SAAS,EAAE,CAAC,SAAS,EAAE,YAAY,CAAC;IACpC,OAAO,EAAE,CAAC,yBAAyB,EAAE,eAAe,EAAE,YAAY,CAAC;IACnE,uBAAuB,EAAE,CAAC,eAAe,EAAE,SAAS,EAAE,YAAY,CAAC;IACnE,aAAa,EAAE,CAAC,WAAW,EAAE,mBAAmB,EAAE,WAAW,EAAE,SAAS,EAAE,YAAY,CAAC;IACvF,SAAS,EAAE,CAAC,mBAAmB,EAAE,YAAY,EAAE,YAAY,CAAC;IAC5D,iBAAiB,EAAE,CAAC,WAAW,EAAE,SAAS,EAAE,yBAAyB,EAAE,YAAY,CAAC;IACpF,UAAU,EAAE,CAAC,WAAW,EAAE,YAAY,CAAC;IACvC,SAAS,EAAE,EAAE;IACb,UAAU,EAAE,EAAE;CACjB,CAAC;AAEF,SAAgB,kBAAkB,CAC9B,KAAa,EACb,MAAuB,EACvB,MAAuB;IAEvB,OAAO;QACH,KAAK;QACL,KAAK,EAAE,UAAU;QACjB,MAAM;QACN,UAAU,EAAE,EAAE;QACd,MAAM,EAAE;YACJ,SAAS,EAAE,CAAC;YACZ,cAAc,EAAE,MAAM,CAAC,cAAc;YACrC,YAAY,EAAE,MAAM,CAAC,YAAY;YACjC,YAAY,EAAE,MAAM,CAAC,YAAY;YACjC,iBAAiB,EAAE,MAAM,CAAC,iBAAiB;YAC3C,YAAY,EAAE,CAAC;YACf,UAAU,EAAE,MAAM,CAAC,UAAU;YAC7B,SAAS,EAAE,IAAI,CAAC,GAAG,EAAE;YACrB,WAAW,EAAE,OAAO;SACvB;QACD,QAAQ,EAAE;YACN,yBAAyB,EAAE,CAAC;YAC5B,mBAAmB,EAAE,CAAC;YACtB,gBAAgB,EAAE,CAAC;YACnB,+BAA+B,EAAE,IAAI;SACxC;QACD,cAAc,EAAE,CAAC;QACjB,kBAAkB,EAAE,CAAC;KACxB,CAAC;AACN,CAAC;AAED,SAAgB,mBAAmB,CAC/B,KAAmB,EACnB,IAAe,EACf,MAAc;IAEd,MAAM,OAAO,GAAG,iBAAiB,CAAC,KAAK,CAAC,KAAK,CAAC,CAAC;IAC/C,IAAI,CAAC,OAAO,CAAC,QAAQ,CAAC,IAAI,CAAC,EAAE,CAAC;QAC1B,MAAM,IAAI,KAAK,CACX,kCAAkC,KAAK,CAAC,KAAK,OAAO,IAAI,KAAK,MAAM,KAAK;YACxE,sBAAsB,OAAO,CAAC,IAAI,CAAC,IAAI,CAAC,EAAE,CAC7C,CAAC;IACN,CAAC;IACD,KAAK,CAAC,KAAK,GAAG,IAAI,CAAC;AACvB,CAAC;AAED,SAAgB,aAAa,CACzB,KAAmB,EACnB,MAA6B,EAC7B,OAAgB;IAEhB,KAAK,CAAC,WAAW,GAAG;QAChB,MAAM;QACN,KAAK,EAAE,KAAK,CAAC,KAAK;QAClB,SAAS,EAAE,IAAI,CAAC,GAAG,EAAE;QACrB,OAAO;KACV,CAAC;IACF,KAAK,CAAC,KAAK,GAAG,YAAY,CAAC;AAC/B,CAAC;AAED,SAAgB,cAAc,CAAC,KAAmB;IAC9C,OAAO,KAAK,CAAC,KAAK,KAAK,WAAW,IAAI,KAAK,CAAC,KAAK,KAAK,YAAY,CAAC;AACvE,CAAC;AAED,SAAgB,kBAAkB,CAAC,KAAmB;IAClD,OAAO,KAAK,CAAC,MAAM,CAAC,cAAc,GAAG,CAAC;QAC/B,KAAK,CAAC,MAAM,CAAC,YAAY,GAAG,KAAK,CAAC,MAAM,CAAC,UAAU,CAAC;AAC/D,CAAC;AAED,SAAgB,qBAAqB,CAAC,KAAmB;IACrD,OAAO,IAAI,CAAC,GAAG,EAAE,GAAG,KAAK,CAAC,MAAM,CAAC,SAAS,GAAG,KAAK,CAAC,MAAM,CAAC,WAAW,CAAC;AAC1E,CAAC;AAED,SAAgB,cAAc,CAAC,KAAmB;IAC9C,OAAO,kBAAkB,CAAC,KAAK,CAAC,IAAI,qBAAqB,CAAC,KAAK,CAAC,IAAI,CAAC,cAAc,CAAC,KAAK,CAAC,CAAC;AAC/F,CAAC;AAED,SAAgB,aAAa,CAAC,KAAmB;IAC7C,OAAO,IAAI,CAAC,SAAS,CAAC;QAClB,KAAK,EAAE,KAAK,CAAC,KAAK;QAClB,KAAK,EAAE,KAAK,CAAC,KAAK;QAClB,SAAS,EAAE,KAAK,CAAC,MAAM,CAAC,SAAS;QACjC,cAAc,EAAE,KAAK,CAAC,MAAM,CAAC,cAAc;QAC3C,YAAY,EAAE,KAAK,CAAC,MAAM,CAAC,YAAY;QACvC,cAAc,EAAE,KAAK,CAAC,cAAc;QACpC,mBAAmB,EAAE,KAAK,CAAC,QAAQ,CAAC,mBAAmB;QACvD,WAAW,EAAE,KAAK,CAAC,WAAW;KACjC,CAAC,CAAC;AACP,CAAC"}
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Search intent normalization — detects when two search_code calls are
|
|
3
|
+
* semantically equivalent even if their patterns differ in ordering or
|
|
4
|
+
* formatting.
|
|
5
|
+
*
|
|
6
|
+
* Without this, the agent can call:
|
|
7
|
+
* search_code("requireOwner|CurrentWsSessionRole")
|
|
8
|
+
* search_code("CurrentWsSessionRole|requireOwner")
|
|
9
|
+
* search_code("requireOwner|CurrentWsSessionRole|isLoopbackHost")
|
|
10
|
+
* ...and each is treated as a "new" search because the raw strings differ.
|
|
11
|
+
*
|
|
12
|
+
* This module normalizes patterns into a canonical form and provides
|
|
13
|
+
* overlap detection so the loop can block equivalent searches.
|
|
14
|
+
*/
|
|
15
|
+
export declare function normalizeSearchPattern(pattern: string): string;
|
|
16
|
+
export declare function isEquivalentSearchIntent(previous: string, current: string): boolean;
|
|
@@ -0,0 +1,57 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
/**
|
|
3
|
+
* Search intent normalization — detects when two search_code calls are
|
|
4
|
+
* semantically equivalent even if their patterns differ in ordering or
|
|
5
|
+
* formatting.
|
|
6
|
+
*
|
|
7
|
+
* Without this, the agent can call:
|
|
8
|
+
* search_code("requireOwner|CurrentWsSessionRole")
|
|
9
|
+
* search_code("CurrentWsSessionRole|requireOwner")
|
|
10
|
+
* search_code("requireOwner|CurrentWsSessionRole|isLoopbackHost")
|
|
11
|
+
* ...and each is treated as a "new" search because the raw strings differ.
|
|
12
|
+
*
|
|
13
|
+
* This module normalizes patterns into a canonical form and provides
|
|
14
|
+
* overlap detection so the loop can block equivalent searches.
|
|
15
|
+
*/
|
|
16
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
17
|
+
exports.normalizeSearchPattern = normalizeSearchPattern;
|
|
18
|
+
exports.isEquivalentSearchIntent = isEquivalentSearchIntent;
|
|
19
|
+
function normalizeSearchPattern(pattern) {
|
|
20
|
+
let p = pattern.trim();
|
|
21
|
+
// Remove common regex wrappers that don't change intent
|
|
22
|
+
p = p.replace(/^\(\?:|\)\$$/g, '');
|
|
23
|
+
// Split on | (alternation) into alternatives
|
|
24
|
+
const alternatives = p
|
|
25
|
+
.split('|')
|
|
26
|
+
.map(a => a.trim())
|
|
27
|
+
.filter(a => a.length > 0);
|
|
28
|
+
if (alternatives.length <= 1) {
|
|
29
|
+
return p.toLowerCase();
|
|
30
|
+
}
|
|
31
|
+
// Sort alternatives alphabetically, deduplicate, lowercase
|
|
32
|
+
const unique = [...new Set(alternatives.map(a => a.toLowerCase()))];
|
|
33
|
+
unique.sort();
|
|
34
|
+
return unique.join('|');
|
|
35
|
+
}
|
|
36
|
+
function isEquivalentSearchIntent(previous, current) {
|
|
37
|
+
const prevNorm = normalizeSearchPattern(previous);
|
|
38
|
+
const currNorm = normalizeSearchPattern(current);
|
|
39
|
+
if (prevNorm === currNorm)
|
|
40
|
+
return true;
|
|
41
|
+
// Check if one is a subset of the other (all alternatives of the
|
|
42
|
+
// smaller pattern are present in the larger one)
|
|
43
|
+
const prevTerms = prevNorm.split('|').filter(t => t.length > 0);
|
|
44
|
+
const currTerms = currNorm.split('|').filter(t => t.length > 0);
|
|
45
|
+
const prevSet = new Set(prevTerms);
|
|
46
|
+
const currSet = new Set(currTerms);
|
|
47
|
+
// If current is a subset of previous, it's equivalent (already searched)
|
|
48
|
+
const currSubsetOfPrev = [...currSet].every(t => prevSet.has(t));
|
|
49
|
+
if (currSubsetOfPrev)
|
|
50
|
+
return true;
|
|
51
|
+
// If previous is a subset of current, the current adds nothing new
|
|
52
|
+
const prevSubsetOfCurr = [...prevSet].every(t => currSet.has(t));
|
|
53
|
+
if (prevSubsetOfCurr)
|
|
54
|
+
return true;
|
|
55
|
+
return false;
|
|
56
|
+
}
|
|
57
|
+
//# sourceMappingURL=searchIntent.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"searchIntent.js","sourceRoot":"","sources":["../../src/attack/searchIntent.ts"],"names":[],"mappings":";AAAA;;;;;;;;;;;;;GAaG;;AAEH,wDAoBC;AAED,4DAuBC;AA7CD,SAAgB,sBAAsB,CAAC,OAAe;IAClD,IAAI,CAAC,GAAG,OAAO,CAAC,IAAI,EAAE,CAAC;IAEvB,wDAAwD;IACxD,CAAC,GAAG,CAAC,CAAC,OAAO,CAAC,eAAe,EAAE,EAAE,CAAC,CAAC;IAEnC,6CAA6C;IAC7C,MAAM,YAAY,GAAG,CAAC;SACjB,KAAK,CAAC,GAAG,CAAC;SACV,GAAG,CAAC,CAAC,CAAC,EAAE,CAAC,CAAC,CAAC,IAAI,EAAE,CAAC;SAClB,MAAM,CAAC,CAAC,CAAC,EAAE,CAAC,CAAC,CAAC,MAAM,GAAG,CAAC,CAAC,CAAC;IAE/B,IAAI,YAAY,CAAC,MAAM,IAAI,CAAC,EAAE,CAAC;QAC3B,OAAO,CAAC,CAAC,WAAW,EAAE,CAAC;IAC3B,CAAC;IAED,2DAA2D;IAC3D,MAAM,MAAM,GAAG,CAAC,GAAG,IAAI,GAAG,CAAC,YAAY,CAAC,GAAG,CAAC,CAAC,CAAC,EAAE,CAAC,CAAC,CAAC,WAAW,EAAE,CAAC,CAAC,CAAC,CAAC;IACpE,MAAM,CAAC,IAAI,EAAE,CAAC;IACd,OAAO,MAAM,CAAC,IAAI,CAAC,GAAG,CAAC,CAAC;AAC5B,CAAC;AAED,SAAgB,wBAAwB,CAAC,QAAgB,EAAE,OAAe;IACtE,MAAM,QAAQ,GAAG,sBAAsB,CAAC,QAAQ,CAAC,CAAC;IAClD,MAAM,QAAQ,GAAG,sBAAsB,CAAC,OAAO,CAAC,CAAC;IAEjD,IAAI,QAAQ,KAAK,QAAQ;QAAE,OAAO,IAAI,CAAC;IAEvC,iEAAiE;IACjE,iDAAiD;IACjD,MAAM,SAAS,GAAG,QAAQ,CAAC,KAAK,CAAC,GAAG,CAAC,CAAC,MAAM,CAAC,CAAC,CAAC,EAAE,CAAC,CAAC,CAAC,MAAM,GAAG,CAAC,CAAC,CAAC;IAChE,MAAM,SAAS,GAAG,QAAQ,CAAC,KAAK,CAAC,GAAG,CAAC,CAAC,MAAM,CAAC,CAAC,CAAC,EAAE,CAAC,CAAC,CAAC,MAAM,GAAG,CAAC,CAAC,CAAC;IAEhE,MAAM,OAAO,GAAG,IAAI,GAAG,CAAC,SAAS,CAAC,CAAC;IACnC,MAAM,OAAO,GAAG,IAAI,GAAG,CAAC,SAAS,CAAC,CAAC;IAEnC,yEAAyE;IACzE,MAAM,gBAAgB,GAAG,CAAC,GAAG,OAAO,CAAC,CAAC,KAAK,CAAC,CAAC,CAAC,EAAE,CAAC,OAAO,CAAC,GAAG,CAAC,CAAC,CAAC,CAAC,CAAC;IACjE,IAAI,gBAAgB;QAAE,OAAO,IAAI,CAAC;IAElC,mEAAmE;IACnE,MAAM,gBAAgB,GAAG,CAAC,GAAG,OAAO,CAAC,CAAC,KAAK,CAAC,CAAC,CAAC,EAAE,CAAC,OAAO,CAAC,GAAG,CAAC,CAAC,CAAC,CAAC,CAAC;IACjE,IAAI,gBAAgB;QAAE,OAAO,IAAI,CAAC;IAElC,OAAO,KAAK,CAAC;AACjB,CAAC"}
|
|
@@ -1,4 +1,5 @@
|
|
|
1
1
|
import type { ApiClient } from '../api/client';
|
|
2
|
+
import { VerifyBudgetTracker, defaultVerifyBudget, type VerifyBudget } from './agentScanProtocol';
|
|
2
3
|
export interface VerifyFinding {
|
|
3
4
|
type: string;
|
|
4
5
|
line: number;
|
|
@@ -20,6 +21,17 @@ export interface VerifyLoopOptions {
|
|
|
20
21
|
language: string;
|
|
21
22
|
client: ApiClient;
|
|
22
23
|
onProgress?: (round: number, maxRounds: number, message: string) => void;
|
|
24
|
+
/** Aggregate budget tracker shared across all findings in one scan. */
|
|
25
|
+
budgetTracker?: VerifyBudgetTracker;
|
|
26
|
+
/** AbortSignal — aborts the loop and any running test. */
|
|
27
|
+
signal?: AbortSignal;
|
|
28
|
+
/**
|
|
29
|
+
* Distinguishes initial vulnerability verification from fix re-verification.
|
|
30
|
+
* - 'original' (default): verifying the original code for a vulnerability.
|
|
31
|
+
* - 'fix': re-verifying the proposed fixed code to see if the exploit still works.
|
|
32
|
+
* Passed to the API so the verifier prompt can be adjusted accordingly.
|
|
33
|
+
*/
|
|
34
|
+
verificationPhase?: 'original' | 'fix';
|
|
23
35
|
}
|
|
24
36
|
export interface VerifyLoopResult {
|
|
25
37
|
verdict: 'PROVEN' | 'UNPROVEN' | 'INCONCLUSIVE';
|
|
@@ -27,5 +39,36 @@ export interface VerifyLoopResult {
|
|
|
27
39
|
roundsUsed: number;
|
|
28
40
|
testScript: string;
|
|
29
41
|
testOutput: string;
|
|
42
|
+
/** Why the loop stopped early, when verdict is INCONCLUSIVE due to budget. */
|
|
43
|
+
budgetExhaustedReason?: string;
|
|
44
|
+
/** Backend that executed the test (docker, deno, etc.). Empty when none ran. */
|
|
45
|
+
backend?: string;
|
|
46
|
+
/** Structured proof evidence from the proof marker, if present. */
|
|
47
|
+
proofEvidence?: import('./proofTypes').ProofEvidence;
|
|
48
|
+
/** Result of the deterministic proof gate evaluation. */
|
|
49
|
+
proofGateResult?: import('./proofTypes').ProofGateResult;
|
|
50
|
+
/** Proof-specific sub-verdict for strict gate failures. */
|
|
51
|
+
proofSubVerdict?: import('./proofTypes').ProofSubVerdict;
|
|
52
|
+
/**
|
|
53
|
+
* Machine-readable sub-verdict so callers (toolAgentScan) can detect the
|
|
54
|
+
* INCONCLUSIVE reason without string-matching `reason`. Distinct from
|
|
55
|
+
* `verdict` because all of these collapse to INCONCLUSIVE at the top
|
|
56
|
+
* level — the difference is what the user should do about it.
|
|
57
|
+
*
|
|
58
|
+
* - 'analyzed' : the analyze LLM ran and returned INCONCLUSIVE
|
|
59
|
+
* (couldn't decide after exhausting rounds).
|
|
60
|
+
* - 'cannot-test' : /verify/generate returned canTest:false — the
|
|
61
|
+
* vuln type/framework can't be tested locally.
|
|
62
|
+
* - 'sandbox-unavailable': no Docker/Deno on the user's machine. The
|
|
63
|
+
* caller should surface an install hint.
|
|
64
|
+
* - 'blocked' : the static safety check rejected the script.
|
|
65
|
+
* - 'budget-exhausted' : the aggregate VerifyBudget ran out.
|
|
66
|
+
* - 'aborted' : the AbortSignal fired.
|
|
67
|
+
* - 'cancelled' : the user cancelled the scan mid-test.
|
|
68
|
+
* - 'runtime-blocked' : detectRuntime said canRunLocally:false.
|
|
69
|
+
* - undefined : PROVEN or UNPROVEN (no sub-verdict needed).
|
|
70
|
+
*/
|
|
71
|
+
subVerdict?: 'analyzed' | 'cannot-test' | 'sandbox-unavailable' | 'blocked' | 'budget-exhausted' | 'aborted' | 'cancelled' | 'runtime-blocked';
|
|
30
72
|
}
|
|
31
73
|
export declare function runVerifyLoop(opts: VerifyLoopOptions): Promise<VerifyLoopResult>;
|
|
74
|
+
export { VerifyBudgetTracker, defaultVerifyBudget, type VerifyBudget };
|
|
@@ -33,24 +33,88 @@ var __importStar = (this && this.__importStar) || (function () {
|
|
|
33
33
|
};
|
|
34
34
|
})();
|
|
35
35
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
36
|
+
exports.defaultVerifyBudget = exports.VerifyBudgetTracker = void 0;
|
|
36
37
|
exports.runVerifyLoop = runVerifyLoop;
|
|
37
38
|
const path = __importStar(require("path"));
|
|
38
39
|
const localTestRunner_1 = require("../utils/localTestRunner");
|
|
39
40
|
const runtimeDetect_1 = require("../utils/runtimeDetect");
|
|
40
|
-
const
|
|
41
|
+
const agentScanProtocol_1 = require("./agentScanProtocol");
|
|
42
|
+
Object.defineProperty(exports, "VerifyBudgetTracker", { enumerable: true, get: function () { return agentScanProtocol_1.VerifyBudgetTracker; } });
|
|
43
|
+
Object.defineProperty(exports, "defaultVerifyBudget", { enumerable: true, get: function () { return agentScanProtocol_1.defaultVerifyBudget; } });
|
|
44
|
+
const protocolValidator_1 = require("./protocolValidator");
|
|
45
|
+
const proofTypes_1 = require("./proofTypes");
|
|
46
|
+
const proofGate_1 = require("./proofGate");
|
|
47
|
+
const mutationTest_1 = require("./mutationTest");
|
|
48
|
+
const PER_FINDING_MAX_ROUNDS = 12;
|
|
41
49
|
async function runVerifyLoop(opts) {
|
|
42
50
|
const { finding, filePath, code, relatedFiles, workspaceRoot, language, client, onProgress } = opts;
|
|
51
|
+
const tracker = opts.budgetTracker;
|
|
52
|
+
const verificationPhase = opts.verificationPhase ?? 'original';
|
|
43
53
|
const previousErrors = [];
|
|
44
54
|
let lastTestScript = '';
|
|
45
55
|
let lastTestOutput = '';
|
|
46
|
-
|
|
56
|
+
// Per-finding round cap = min(per-finding default, budget's per-finding cap).
|
|
57
|
+
const maxRounds = tracker
|
|
58
|
+
? Math.min(PER_FINDING_MAX_ROUNDS, tracker.budget.maxRoundsPerFinding)
|
|
59
|
+
: PER_FINDING_MAX_ROUNDS;
|
|
60
|
+
const runtimeInfo = (0, runtimeDetect_1.detectRuntime)(workspaceRoot, filePath || undefined);
|
|
47
61
|
const testFileDir = path.join(workspaceRoot, '.securecode');
|
|
48
62
|
const relativeImportPath = filePath
|
|
49
63
|
? (0, runtimeDetect_1.computeRelativeImportPath)(testFileDir, path.isAbsolute(filePath) ? filePath : path.join(workspaceRoot, filePath))
|
|
50
64
|
: '';
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
65
|
+
if (!runtimeInfo.canRunLocally) {
|
|
66
|
+
return {
|
|
67
|
+
verdict: 'INCONCLUSIVE',
|
|
68
|
+
reason: runtimeInfo.skipReason || 'Detected framework cannot run a local exploit test.',
|
|
69
|
+
roundsUsed: 0,
|
|
70
|
+
testScript: '',
|
|
71
|
+
testOutput: '',
|
|
72
|
+
subVerdict: 'runtime-blocked',
|
|
73
|
+
};
|
|
74
|
+
}
|
|
75
|
+
// Pre-flight: if the aggregate budget is already exhausted at finding entry,
|
|
76
|
+
// skip without spending any LLM calls. The caller (toolAgentScan) should
|
|
77
|
+
// also check this before calling, but we double-enforce here.
|
|
78
|
+
if (tracker && !tracker.canAttemptFinding()) {
|
|
79
|
+
return {
|
|
80
|
+
verdict: 'INCONCLUSIVE',
|
|
81
|
+
reason: `Verification budget exhausted before this finding (findings=${tracker.findingsAttempted}/${tracker.budget.maxFindings}, llmCalls=${tracker.llmCallsUsed}/${tracker.budget.maxLlmCalls}, wallClock=${tracker.wallClockElapsedMs}ms/${tracker.budget.maxWallClockMs}ms).`,
|
|
82
|
+
roundsUsed: 0,
|
|
83
|
+
testScript: '',
|
|
84
|
+
testOutput: '',
|
|
85
|
+
budgetExhaustedReason: 'aggregate',
|
|
86
|
+
subVerdict: 'budget-exhausted',
|
|
87
|
+
};
|
|
88
|
+
}
|
|
89
|
+
if (tracker)
|
|
90
|
+
tracker.findingsAttempted += 1;
|
|
91
|
+
for (let round = 1; round <= maxRounds; round++) {
|
|
92
|
+
// Per-round budget check (need at least generate + analyze = 2 calls).
|
|
93
|
+
if (tracker && !tracker.canAttemptRound()) {
|
|
94
|
+
return {
|
|
95
|
+
verdict: 'INCONCLUSIVE',
|
|
96
|
+
reason: `Verification budget exhausted mid-finding (llmCalls=${tracker.llmCallsUsed}/${tracker.budget.maxLlmCalls}, wallClock=${tracker.wallClockElapsedMs}ms/${tracker.budget.maxWallClockMs}ms).`,
|
|
97
|
+
roundsUsed: round - 1,
|
|
98
|
+
testScript: lastTestScript,
|
|
99
|
+
testOutput: lastTestOutput,
|
|
100
|
+
budgetExhaustedReason: 'aggregate',
|
|
101
|
+
subVerdict: 'budget-exhausted',
|
|
102
|
+
};
|
|
103
|
+
}
|
|
104
|
+
if (tracker)
|
|
105
|
+
tracker.roundsUsed += 1;
|
|
106
|
+
if (opts.signal?.aborted) {
|
|
107
|
+
return {
|
|
108
|
+
verdict: 'INCONCLUSIVE',
|
|
109
|
+
reason: 'Cancelled by user.',
|
|
110
|
+
roundsUsed: round - 1,
|
|
111
|
+
testScript: lastTestScript,
|
|
112
|
+
testOutput: lastTestOutput,
|
|
113
|
+
subVerdict: 'aborted',
|
|
114
|
+
};
|
|
115
|
+
}
|
|
116
|
+
onProgress?.(round, maxRounds, `Round ${round}: generating test...`);
|
|
117
|
+
const genRespRaw = await client.postJson('/verify/generate', {
|
|
54
118
|
code,
|
|
55
119
|
language,
|
|
56
120
|
vulnerabilityType: finding.type,
|
|
@@ -63,9 +127,29 @@ async function runVerifyLoop(opts) {
|
|
|
63
127
|
previousErrors: previousErrors.length > 0 ? previousErrors : undefined,
|
|
64
128
|
projectRuntime: runtimeInfo.runtime,
|
|
65
129
|
suggestedRunner: runtimeInfo.runner,
|
|
130
|
+
framework: runtimeInfo.framework,
|
|
131
|
+
frameworkVersion: runtimeInfo.frameworkVersion,
|
|
132
|
+
testabilityTier: runtimeInfo.testabilityTier,
|
|
133
|
+
packageManager: runtimeInfo.packageManager,
|
|
66
134
|
testFileDir: '.securecode',
|
|
67
135
|
relativeImportPath,
|
|
136
|
+
depsInstalled: runtimeInfo.depsInstalled,
|
|
137
|
+
verificationPhase,
|
|
68
138
|
});
|
|
139
|
+
if (tracker)
|
|
140
|
+
tracker.recordLlmCall(genRespRaw.costUsd ?? 0);
|
|
141
|
+
const genValidation = (0, protocolValidator_1.validateVerifyGenerateResponse)(genRespRaw);
|
|
142
|
+
if (!genValidation.ok) {
|
|
143
|
+
return {
|
|
144
|
+
verdict: 'INCONCLUSIVE',
|
|
145
|
+
reason: `API returned a malformed verify/generate response: ${genValidation.error}`,
|
|
146
|
+
roundsUsed: round,
|
|
147
|
+
testScript: '',
|
|
148
|
+
testOutput: '',
|
|
149
|
+
subVerdict: 'analyzed',
|
|
150
|
+
};
|
|
151
|
+
}
|
|
152
|
+
const genResp = genValidation.value;
|
|
69
153
|
if (!genResp.canTest || !genResp.testScript) {
|
|
70
154
|
return {
|
|
71
155
|
verdict: 'INCONCLUSIVE',
|
|
@@ -73,15 +157,80 @@ async function runVerifyLoop(opts) {
|
|
|
73
157
|
roundsUsed: round,
|
|
74
158
|
testScript: '',
|
|
75
159
|
testOutput: '',
|
|
160
|
+
subVerdict: 'cannot-test',
|
|
76
161
|
};
|
|
77
162
|
}
|
|
78
163
|
lastTestScript = genResp.testScript;
|
|
79
164
|
const runner = genResp.runner || runtimeInfo.runner || (language === 'typescript' ? 'tsx' : 'node');
|
|
80
|
-
onProgress?.(round,
|
|
81
|
-
|
|
165
|
+
onProgress?.(round, maxRounds, `Round ${round}: running test...`);
|
|
166
|
+
// Cap the local test timeout at the remaining verify-budget wall-clock.
|
|
167
|
+
const testTimeoutMs = tracker
|
|
168
|
+
? Math.min(30_000, tracker.remainingWallClockMs())
|
|
169
|
+
: 30_000;
|
|
170
|
+
const testResult = await (0, localTestRunner_1.runLocalTest)(genResp.testScript, runner, workspaceRoot, {
|
|
171
|
+
setupScript: genResp.setupScript || null,
|
|
172
|
+
timeoutMs: testTimeoutMs,
|
|
173
|
+
signal: opts.signal,
|
|
174
|
+
});
|
|
82
175
|
lastTestOutput = testResult.output;
|
|
83
|
-
|
|
84
|
-
|
|
176
|
+
// Handle non-LLM-judged verdicts that should short-circuit the loop.
|
|
177
|
+
if (testResult.verdict === 'sandbox-unavailable') {
|
|
178
|
+
// No isolation backend on the user's machine. We must NOT run
|
|
179
|
+
// the script with host privileges — return INCONCLUSIVE so the
|
|
180
|
+
// finding keeps its non-PROVEN status but isn't buried as UNPROVEN.
|
|
181
|
+
return {
|
|
182
|
+
verdict: 'INCONCLUSIVE',
|
|
183
|
+
reason: testResult.output,
|
|
184
|
+
roundsUsed: round,
|
|
185
|
+
testScript: lastTestScript,
|
|
186
|
+
testOutput: lastTestOutput,
|
|
187
|
+
backend: testResult.backend || '',
|
|
188
|
+
subVerdict: 'sandbox-unavailable',
|
|
189
|
+
};
|
|
190
|
+
}
|
|
191
|
+
if (testResult.verdict === 'blocked') {
|
|
192
|
+
// Static safety check rejected the script. Don't retry — the
|
|
193
|
+
// LLM would have to emit a less dangerous script, but anything
|
|
194
|
+
// that trips the blocklist is almost certainly trying to do
|
|
195
|
+
// something we don't want.
|
|
196
|
+
return {
|
|
197
|
+
verdict: 'INCONCLUSIVE',
|
|
198
|
+
reason: `Test script rejected by safety check: ${testResult.output}`,
|
|
199
|
+
roundsUsed: round,
|
|
200
|
+
testScript: lastTestScript,
|
|
201
|
+
testOutput: lastTestOutput,
|
|
202
|
+
subVerdict: 'blocked',
|
|
203
|
+
};
|
|
204
|
+
}
|
|
205
|
+
if (testResult.verdict === 'cancelled') {
|
|
206
|
+
// User aborted the scan. Do NOT retry — return immediately so the
|
|
207
|
+
// cancellation propagates up to toolAgentScan, which stops the loop.
|
|
208
|
+
return {
|
|
209
|
+
verdict: 'INCONCLUSIVE',
|
|
210
|
+
reason: `Verification cancelled by user at round ${round}.`,
|
|
211
|
+
roundsUsed: round,
|
|
212
|
+
testScript: lastTestScript,
|
|
213
|
+
testOutput: lastTestOutput,
|
|
214
|
+
subVerdict: 'cancelled',
|
|
215
|
+
};
|
|
216
|
+
}
|
|
217
|
+
if (testResult.verdict === 'timeout') {
|
|
218
|
+
// Treat as a retryable error — feed it back to the LLM.
|
|
219
|
+
previousErrors.push(`Round ${round}: timed out — ${testResult.output.slice(0, 500)}`);
|
|
220
|
+
if (round >= maxRounds) {
|
|
221
|
+
return {
|
|
222
|
+
verdict: 'INCONCLUSIVE',
|
|
223
|
+
reason: `Test timed out after ${round} round(s).`,
|
|
224
|
+
roundsUsed: round,
|
|
225
|
+
testScript: lastTestScript,
|
|
226
|
+
testOutput: lastTestOutput,
|
|
227
|
+
subVerdict: 'analyzed',
|
|
228
|
+
};
|
|
229
|
+
}
|
|
230
|
+
continue;
|
|
231
|
+
}
|
|
232
|
+
onProgress?.(round, maxRounds, `Round ${round}: analyzing result...`);
|
|
233
|
+
const analyzeRespRaw = await client.postJson('/verify/analyze', {
|
|
85
234
|
vulnerabilityType: finding.type,
|
|
86
235
|
line: finding.line,
|
|
87
236
|
evidence: finding.evidence,
|
|
@@ -91,19 +240,184 @@ async function runVerifyLoop(opts) {
|
|
|
91
240
|
stderr: '',
|
|
92
241
|
exitCode: testResult.exitCode,
|
|
93
242
|
round,
|
|
94
|
-
maxRounds
|
|
243
|
+
maxRounds,
|
|
244
|
+
verificationPhase,
|
|
95
245
|
});
|
|
246
|
+
if (tracker)
|
|
247
|
+
tracker.recordLlmCall(analyzeRespRaw.costUsd ?? 0);
|
|
248
|
+
const analyzeValidation = (0, protocolValidator_1.validateVerifyAnalyzeResponse)(analyzeRespRaw);
|
|
249
|
+
if (!analyzeValidation.ok) {
|
|
250
|
+
return {
|
|
251
|
+
verdict: 'INCONCLUSIVE',
|
|
252
|
+
reason: `API returned a malformed verify/analyze response: ${analyzeValidation.error}`,
|
|
253
|
+
roundsUsed: round,
|
|
254
|
+
testScript: lastTestScript,
|
|
255
|
+
testOutput: lastTestOutput,
|
|
256
|
+
backend: testResult.backend || '',
|
|
257
|
+
subVerdict: 'analyzed',
|
|
258
|
+
};
|
|
259
|
+
}
|
|
260
|
+
const analyzeResp = analyzeValidation.value;
|
|
261
|
+
const proofMarker = (0, proofTypes_1.parseProofMarker)(lastTestOutput);
|
|
262
|
+
const gateResult = (0, proofGate_1.evaluateProofGate)(proofMarker, {
|
|
263
|
+
sandboxBackend: testResult.backend || 'unknown',
|
|
264
|
+
targetFile: filePath || '',
|
|
265
|
+
targetLine: finding.line,
|
|
266
|
+
repeatedRuns: 1,
|
|
267
|
+
repeatPasses: proofMarker.found && proofMarker.exploit === 'pass' ? 1 : 0,
|
|
268
|
+
llmVerdict: analyzeResp.verdict,
|
|
269
|
+
sourceMode: proofMarker.sourceMode,
|
|
270
|
+
minimumRepeatRuns: 1,
|
|
271
|
+
});
|
|
272
|
+
const proofEvidence = proofMarker.found
|
|
273
|
+
? (0, proofGate_1.buildProofEvidence)(proofMarker, {
|
|
274
|
+
sandboxBackend: testResult.backend || 'unknown',
|
|
275
|
+
targetFile: filePath || '',
|
|
276
|
+
targetLine: finding.line,
|
|
277
|
+
repeatedRuns: 1,
|
|
278
|
+
repeatPasses: proofMarker.found && proofMarker.exploit === 'pass' ? 1 : 0,
|
|
279
|
+
})
|
|
280
|
+
: undefined;
|
|
96
281
|
if (analyzeResp.verdict === 'PROVEN') {
|
|
97
|
-
|
|
282
|
+
if (gateResult.eligibleForProven) {
|
|
283
|
+
const REPEAT_RUNS = 3;
|
|
284
|
+
const runResults = [];
|
|
285
|
+
// First run already done — validate it independently
|
|
286
|
+
runResults.push({
|
|
287
|
+
runIndex: 1,
|
|
288
|
+
verdict: testResult.verdict,
|
|
289
|
+
markerValid: proofMarker.found,
|
|
290
|
+
targetReached: proofMarker.targetReached === true,
|
|
291
|
+
impactObserved: proofMarker.impact === 'observed',
|
|
292
|
+
baselineResult: proofMarker.baseline,
|
|
293
|
+
exploitResult: proofMarker.exploit,
|
|
294
|
+
});
|
|
295
|
+
let lastRepOutput = lastTestOutput;
|
|
296
|
+
let lastRepMarker = proofMarker;
|
|
297
|
+
for (let rep = 2; rep <= REPEAT_RUNS; rep++) {
|
|
298
|
+
if (opts.signal?.aborted)
|
|
299
|
+
break;
|
|
300
|
+
if (tracker && !tracker.canAttemptRound())
|
|
301
|
+
break;
|
|
302
|
+
onProgress?.(round, maxRounds, `Round ${round}: repeatability run ${rep}/${REPEAT_RUNS}...`);
|
|
303
|
+
const repResult = await (0, localTestRunner_1.runLocalTest)(genResp.testScript, runner, opts.workspaceRoot, { setupScript: genResp.setupScript || null, timeoutMs: tracker ? Math.min(30_000, tracker.remainingWallClockMs()) : 30_000, signal: opts.signal });
|
|
304
|
+
lastRepOutput = repResult.output;
|
|
305
|
+
const repMarker = (0, proofTypes_1.parseProofMarker)(repResult.output);
|
|
306
|
+
lastRepMarker = repMarker;
|
|
307
|
+
runResults.push({
|
|
308
|
+
runIndex: rep,
|
|
309
|
+
verdict: repResult.verdict,
|
|
310
|
+
markerValid: repMarker.found,
|
|
311
|
+
targetReached: repMarker.targetReached === true,
|
|
312
|
+
impactObserved: repMarker.impact === 'observed',
|
|
313
|
+
baselineResult: repMarker.baseline,
|
|
314
|
+
exploitResult: repMarker.exploit,
|
|
315
|
+
});
|
|
316
|
+
}
|
|
317
|
+
// A run only counts as "passed" if ALL proof properties are valid.
|
|
318
|
+
// Do not count a run as passed solely because verdict === 'pass'.
|
|
319
|
+
const totalPasses = runResults.filter(r => r.verdict === 'pass' &&
|
|
320
|
+
r.markerValid &&
|
|
321
|
+
r.targetReached &&
|
|
322
|
+
r.impactObserved).length;
|
|
323
|
+
const repGateResult = (0, proofGate_1.evaluateProofGate)(lastRepMarker, {
|
|
324
|
+
sandboxBackend: testResult.backend || 'unknown',
|
|
325
|
+
targetFile: filePath || '',
|
|
326
|
+
targetLine: finding.line,
|
|
327
|
+
repeatedRuns: runResults.length,
|
|
328
|
+
repeatPasses: totalPasses,
|
|
329
|
+
llmVerdict: 'PROVEN',
|
|
330
|
+
sourceMode: lastRepMarker.sourceMode,
|
|
331
|
+
});
|
|
332
|
+
if (!repGateResult.eligibleForProven) {
|
|
333
|
+
console.warn(`[Verify Loop] Repeatability check failed: ${totalPasses}/${runResults.length} runs passed`);
|
|
334
|
+
const repEvidence = lastRepMarker.found
|
|
335
|
+
? (0, proofGate_1.buildProofEvidence)(lastRepMarker, {
|
|
336
|
+
sandboxBackend: testResult.backend || 'unknown',
|
|
337
|
+
targetFile: filePath || '',
|
|
338
|
+
targetLine: finding.line,
|
|
339
|
+
repeatedRuns: runResults.length,
|
|
340
|
+
repeatPasses: totalPasses,
|
|
341
|
+
})
|
|
342
|
+
: proofEvidence;
|
|
343
|
+
return {
|
|
344
|
+
verdict: 'INCONCLUSIVE',
|
|
345
|
+
reason: `Repeatability check failed: ${totalPasses}/${runResults.length} runs passed. ${repGateResult.warnings.join(' ')}`,
|
|
346
|
+
roundsUsed: round,
|
|
347
|
+
testScript: lastTestScript,
|
|
348
|
+
testOutput: lastTestOutput,
|
|
349
|
+
backend: testResult.backend || '',
|
|
350
|
+
proofEvidence: repEvidence,
|
|
351
|
+
proofGateResult: repGateResult,
|
|
352
|
+
proofSubVerdict: 'gate-rejected',
|
|
353
|
+
subVerdict: 'analyzed',
|
|
354
|
+
};
|
|
355
|
+
}
|
|
356
|
+
if (proofEvidence) {
|
|
357
|
+
proofEvidence.repeatedRuns = REPEAT_RUNS;
|
|
358
|
+
proofEvidence.repeatPasses = totalPasses;
|
|
359
|
+
}
|
|
360
|
+
const mutationResult = await (0, mutationTest_1.runMutationTest)({
|
|
361
|
+
vulnerableCode: code,
|
|
362
|
+
testScript: lastTestScript,
|
|
363
|
+
runner: genResp.runner || runtimeInfo.runner || 'node',
|
|
364
|
+
workspaceRoot: opts.workspaceRoot,
|
|
365
|
+
filePath: filePath || '',
|
|
366
|
+
vulnerabilityType: finding.type,
|
|
367
|
+
line: finding.line,
|
|
368
|
+
});
|
|
369
|
+
if (!mutationResult.discriminating) {
|
|
370
|
+
console.warn(`[Verify Loop] Mutation test non-discriminating: ${mutationResult.reason}`);
|
|
371
|
+
if (proofEvidence) {
|
|
372
|
+
proofEvidence.assumptions.push(`mutation-test: ${mutationResult.reason}`);
|
|
373
|
+
}
|
|
374
|
+
return {
|
|
375
|
+
verdict: 'INCONCLUSIVE',
|
|
376
|
+
reason: `Proof gate passed but mutation test was non-discriminating: ${mutationResult.reason}`,
|
|
377
|
+
roundsUsed: round,
|
|
378
|
+
testScript: lastTestScript,
|
|
379
|
+
testOutput: lastTestOutput,
|
|
380
|
+
backend: testResult.backend || '',
|
|
381
|
+
proofEvidence,
|
|
382
|
+
proofGateResult: { ...gateResult, failedGates: [...gateResult.failedGates, 'mutation-non-discriminating'] },
|
|
383
|
+
proofSubVerdict: 'gate-rejected',
|
|
384
|
+
subVerdict: 'analyzed',
|
|
385
|
+
};
|
|
386
|
+
}
|
|
387
|
+
return {
|
|
388
|
+
verdict: 'PROVEN',
|
|
389
|
+
reason: `${analyzeResp.reason} [mutation: ${mutationResult.reason}]`,
|
|
390
|
+
roundsUsed: round,
|
|
391
|
+
testScript: lastTestScript,
|
|
392
|
+
testOutput: lastTestOutput,
|
|
393
|
+
backend: testResult.backend || '',
|
|
394
|
+
proofEvidence,
|
|
395
|
+
proofGateResult: gateResult,
|
|
396
|
+
proofSubVerdict: 'gate-passed',
|
|
397
|
+
};
|
|
398
|
+
}
|
|
399
|
+
console.warn(`[Verify Loop] LLM said PROVEN but proof gate rejected: ${gateResult.failedGates.join(', ')}`);
|
|
400
|
+
return {
|
|
401
|
+
verdict: gateResult.downgradedVerdict === 'UNPROVEN' ? 'UNPROVEN' : 'INCONCLUSIVE',
|
|
402
|
+
reason: `Proof gate rejected: ${gateResult.failedGates.join(', ')}. ${gateResult.warnings.join(' ')}. LLM reason: ${analyzeResp.reason}`,
|
|
403
|
+
roundsUsed: round,
|
|
404
|
+
testScript: lastTestScript,
|
|
405
|
+
testOutput: lastTestOutput,
|
|
406
|
+
backend: testResult.backend || '',
|
|
407
|
+
proofEvidence,
|
|
408
|
+
proofGateResult: gateResult,
|
|
409
|
+
proofSubVerdict: 'gate-rejected',
|
|
410
|
+
subVerdict: gateResult.downgradedVerdict === 'UNPROVEN' ? undefined : 'analyzed',
|
|
411
|
+
};
|
|
98
412
|
}
|
|
99
413
|
if (analyzeResp.verdict === 'UNPROVEN') {
|
|
100
|
-
return { verdict: 'UNPROVEN', reason: analyzeResp.reason, roundsUsed: round, testScript: lastTestScript, testOutput: lastTestOutput };
|
|
414
|
+
return { verdict: 'UNPROVEN', reason: analyzeResp.reason, roundsUsed: round, testScript: lastTestScript, testOutput: lastTestOutput, backend: testResult.backend || '', proofEvidence, proofGateResult: gateResult };
|
|
101
415
|
}
|
|
102
|
-
if (!analyzeResp.shouldRetry || round >=
|
|
103
|
-
return { verdict: 'INCONCLUSIVE', reason: analyzeResp.reason || `Could not verify after ${round} round(s).`, roundsUsed: round, testScript: lastTestScript, testOutput: lastTestOutput };
|
|
416
|
+
if (!analyzeResp.shouldRetry || round >= maxRounds) {
|
|
417
|
+
return { verdict: 'INCONCLUSIVE', reason: analyzeResp.reason || `Could not verify after ${round} round(s).`, roundsUsed: round, testScript: lastTestScript, testOutput: lastTestOutput, backend: testResult.backend || '', subVerdict: 'analyzed', proofEvidence, proofGateResult: gateResult };
|
|
104
418
|
}
|
|
105
419
|
previousErrors.push(`Round ${round}: ${testResult.verdict} — ${testResult.output.slice(0, 500)}`);
|
|
106
420
|
}
|
|
107
|
-
return { verdict: 'INCONCLUSIVE', reason: `Exhausted all ${
|
|
421
|
+
return { verdict: 'INCONCLUSIVE', reason: `Exhausted all ${maxRounds} rounds.`, roundsUsed: maxRounds, testScript: lastTestScript, testOutput: lastTestOutput, subVerdict: 'analyzed' };
|
|
108
422
|
}
|
|
109
423
|
//# sourceMappingURL=verifyLoop.js.map
|