@securecode-ai/mcp 0.5.5 → 0.5.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +52 -12
- package/dist/api/types.d.ts +7 -0
- package/dist/approval/auditLog.d.ts +1 -1
- package/dist/approval/auditLog.js +11 -3
- package/dist/approval/auditLog.js.map +1 -1
- package/dist/approval/broker.d.ts +7 -2
- package/dist/approval/broker.js +239 -110
- package/dist/approval/broker.js.map +1 -1
- package/dist/approval/policy.d.ts +10 -0
- package/dist/approval/policy.js +104 -0
- package/dist/approval/policy.js.map +1 -0
- package/dist/approval/types.d.ts +11 -2
- package/dist/approval/types.js +10 -1
- package/dist/approval/types.js.map +1 -1
- package/dist/attack/agentScanExecutor.d.ts +35 -1
- package/dist/attack/agentScanExecutor.js +405 -76
- package/dist/attack/agentScanExecutor.js.map +1 -1
- package/dist/attack/agentScanLoop.d.ts +20 -0
- package/dist/attack/agentScanLoop.js +1264 -60
- package/dist/attack/agentScanLoop.js.map +1 -1
- package/dist/attack/agentScanProtocol.d.ts +382 -6
- package/dist/attack/agentScanProtocol.js +110 -4
- package/dist/attack/agentScanProtocol.js.map +1 -1
- package/dist/attack/agentTrace.d.ts +73 -0
- package/dist/attack/agentTrace.js +228 -0
- package/dist/attack/agentTrace.js.map +1 -0
- package/dist/attack/architectureScoutExecutor.d.ts +18 -0
- package/dist/attack/architectureScoutExecutor.js +211 -0
- package/dist/attack/architectureScoutExecutor.js.map +1 -0
- package/dist/attack/architectureScoutLoop.d.ts +36 -0
- package/dist/attack/architectureScoutLoop.js +403 -0
- package/dist/attack/architectureScoutLoop.js.map +1 -0
- package/dist/attack/architectureScoutProtocol.d.ts +209 -0
- package/dist/attack/architectureScoutProtocol.js +47 -0
- package/dist/attack/architectureScoutProtocol.js.map +1 -0
- package/dist/attack/candidateStore.d.ts +95 -0
- package/dist/attack/candidateStore.js +231 -0
- package/dist/attack/candidateStore.js.map +1 -0
- package/dist/attack/evidenceLedger.d.ts +67 -0
- package/dist/attack/evidenceLedger.js +192 -0
- package/dist/attack/evidenceLedger.js.map +1 -0
- package/dist/attack/finishGate.d.ts +62 -0
- package/dist/attack/finishGate.js +209 -0
- package/dist/attack/finishGate.js.map +1 -0
- package/dist/attack/fixCodeMerge.d.ts +27 -0
- package/dist/attack/fixCodeMerge.js +42 -0
- package/dist/attack/fixCodeMerge.js.map +1 -0
- package/dist/attack/fixVerifyLoop.d.ts +55 -0
- package/dist/attack/fixVerifyLoop.js +187 -0
- package/dist/attack/fixVerifyLoop.js.map +1 -0
- package/dist/attack/investigationProfiles.d.ts +36 -0
- package/dist/attack/investigationProfiles.js +144 -0
- package/dist/attack/investigationProfiles.js.map +1 -0
- package/dist/attack/investigationState.d.ts +227 -0
- package/dist/attack/investigationState.js +666 -0
- package/dist/attack/investigationState.js.map +1 -0
- package/dist/attack/mutationOperators.d.ts +22 -0
- package/dist/attack/mutationOperators.js +170 -0
- package/dist/attack/mutationOperators.js.map +1 -0
- package/dist/attack/mutationTest.d.ts +33 -0
- package/dist/attack/mutationTest.js +110 -0
- package/dist/attack/mutationTest.js.map +1 -0
- package/dist/attack/proofGate.d.ts +20 -0
- package/dist/attack/proofGate.js +104 -0
- package/dist/attack/proofGate.js.map +1 -0
- package/dist/attack/proofTypes.d.ts +57 -0
- package/dist/attack/proofTypes.js +32 -0
- package/dist/attack/proofTypes.js.map +1 -0
- package/dist/attack/protocolValidator.d.ts +50 -0
- package/dist/attack/protocolValidator.js +427 -0
- package/dist/attack/protocolValidator.js.map +1 -0
- package/dist/attack/qualityMetrics.d.ts +133 -0
- package/dist/attack/qualityMetrics.js +226 -0
- package/dist/attack/qualityMetrics.js.map +1 -0
- package/dist/attack/scanScheduler.d.ts +52 -0
- package/dist/attack/scanScheduler.js +265 -0
- package/dist/attack/scanScheduler.js.map +1 -0
- package/dist/attack/scanState.d.ts +58 -0
- package/dist/attack/scanState.js +102 -0
- package/dist/attack/scanState.js.map +1 -0
- package/dist/attack/searchIntent.d.ts +16 -0
- package/dist/attack/searchIntent.js +57 -0
- package/dist/attack/searchIntent.js.map +1 -0
- package/dist/attack/verifyLoop.d.ts +43 -0
- package/dist/attack/verifyLoop.js +314 -14
- package/dist/attack/verifyLoop.js.map +1 -1
- package/dist/attack/workItem.d.ts +57 -0
- package/dist/attack/workItem.js +225 -0
- package/dist/attack/workItem.js.map +1 -0
- package/dist/audit/findingReviewQueue.d.ts +109 -0
- package/dist/audit/findingReviewQueue.js +335 -0
- package/dist/audit/findingReviewQueue.js.map +1 -0
- package/dist/audit/scanAuditLog.d.ts +160 -0
- package/dist/audit/scanAuditLog.js +404 -0
- package/dist/audit/scanAuditLog.js.map +1 -0
- package/dist/dependency/dependencyChecker.js +35 -9
- package/dist/dependency/dependencyChecker.js.map +1 -1
- package/dist/dependency/exploitPriority.d.ts +33 -0
- package/dist/dependency/exploitPriority.js +78 -0
- package/dist/dependency/exploitPriority.js.map +1 -0
- package/dist/dependency/finding.d.ts +16 -0
- package/dist/dependency/osvClient.js +2 -0
- package/dist/dependency/osvClient.js.map +1 -1
- package/dist/dependency/types.d.ts +12 -1
- package/dist/mcp/server.js +13 -2
- package/dist/mcp/server.js.map +1 -1
- package/dist/mcp/tools.d.ts +1 -0
- package/dist/mcp/tools.js +127 -6
- package/dist/mcp/tools.js.map +1 -1
- package/dist/project-map/agentMemory.d.ts +83 -1
- package/dist/project-map/agentMemory.js +197 -7
- package/dist/project-map/agentMemory.js.map +1 -1
- package/dist/project-map/architectureContext.d.ts +202 -0
- package/dist/project-map/architectureContext.js +421 -0
- package/dist/project-map/architectureContext.js.map +1 -0
- package/dist/project-map/blastRadius.d.ts +54 -0
- package/dist/project-map/blastRadius.js +196 -0
- package/dist/project-map/blastRadius.js.map +1 -0
- package/dist/project-map/callGraphExtractor.d.ts +13 -0
- package/dist/project-map/callGraphExtractor.js +222 -0
- package/dist/project-map/callGraphExtractor.js.map +1 -0
- package/dist/project-map/capabilityRegistry.d.ts +46 -0
- package/dist/project-map/capabilityRegistry.js +91 -0
- package/dist/project-map/capabilityRegistry.js.map +1 -0
- package/dist/project-map/findTests.d.ts +29 -0
- package/dist/project-map/findTests.js +236 -0
- package/dist/project-map/findTests.js.map +1 -0
- package/dist/project-map/handlerInventory.d.ts +112 -0
- package/dist/project-map/handlerInventory.js +184 -0
- package/dist/project-map/handlerInventory.js.map +1 -0
- package/dist/project-map/implementationResolver.d.ts +43 -0
- package/dist/project-map/implementationResolver.js +217 -0
- package/dist/project-map/implementationResolver.js.map +1 -0
- package/dist/project-map/mapContext.d.ts +6 -1
- package/dist/project-map/mapContext.js +1 -0
- package/dist/project-map/mapContext.js.map +1 -1
- package/dist/project-map/scanCache.d.ts +57 -3
- package/dist/project-map/scanCache.js +61 -2
- package/dist/project-map/scanCache.js.map +1 -1
- package/dist/project-map/symbolIndex.d.ts +39 -0
- package/dist/project-map/symbolIndex.js +385 -0
- package/dist/project-map/symbolIndex.js.map +1 -0
- package/dist/tooling/agentEvalScoring.d.ts +77 -0
- package/dist/tooling/agentEvalScoring.js +140 -0
- package/dist/tooling/agentEvalScoring.js.map +1 -0
- package/dist/tooling/agentRegression.d.ts +53 -0
- package/dist/tooling/agentRegression.js +99 -0
- package/dist/tooling/agentRegression.js.map +1 -0
- package/dist/tools/agentScan.d.ts +69 -0
- package/dist/tools/agentScan.js +702 -43
- package/dist/tools/agentScan.js.map +1 -1
- package/dist/tools/findingReviewTools.d.ts +22 -0
- package/dist/tools/findingReviewTools.js +121 -0
- package/dist/tools/findingReviewTools.js.map +1 -0
- package/dist/tools/fix.js +1 -1
- package/dist/tools/fix.js.map +1 -1
- package/dist/tools/map.js +191 -1
- package/dist/tools/map.js.map +1 -1
- package/dist/tools/runTests.d.ts +2 -0
- package/dist/tools/runTests.js +29 -0
- package/dist/tools/runTests.js.map +1 -0
- package/dist/utils/effectMock.d.ts +1 -0
- package/dist/utils/effectMock.js +162 -0
- package/dist/utils/effectMock.js.map +1 -0
- package/dist/utils/gitContext.d.ts +26 -0
- package/dist/utils/gitContext.js +317 -0
- package/dist/utils/gitContext.js.map +1 -0
- package/dist/utils/localTestRunner.d.ts +57 -2
- package/dist/utils/localTestRunner.js +122 -115
- package/dist/utils/localTestRunner.js.map +1 -1
- package/dist/utils/securityConfig.d.ts +17 -0
- package/dist/utils/securityConfig.js +192 -0
- package/dist/utils/securityConfig.js.map +1 -0
- package/dist/utils/testCommandPolicy.d.ts +36 -0
- package/dist/utils/testCommandPolicy.js +252 -0
- package/dist/utils/testCommandPolicy.js.map +1 -0
- package/dist/utils/testRunner.d.ts +56 -0
- package/dist/utils/testRunner.js +246 -0
- package/dist/utils/testRunner.js.map +1 -0
- package/dist/utils/testSafety.d.ts +40 -2
- package/dist/utils/testSafety.js +205 -26
- package/dist/utils/testSafety.js.map +1 -1
- package/dist/utils/verificationSandbox.d.ts +113 -0
- package/dist/utils/verificationSandbox.js +656 -0
- package/dist/utils/verificationSandbox.js.map +1 -0
- package/package.json +1 -1
|
@@ -19,93 +19,374 @@
|
|
|
19
19
|
*/
|
|
20
20
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
21
21
|
exports.runAgentScan = runAgentScan;
|
|
22
|
+
exports.isCoherentText = isCoherentText;
|
|
23
|
+
exports.estimateTranscriptSize = estimateTranscriptSize;
|
|
24
|
+
exports.compactTranscript = compactTranscript;
|
|
25
|
+
exports.compactTranscriptAggressive = compactTranscriptAggressive;
|
|
26
|
+
exports.sanitizeFindings = sanitizeFindings;
|
|
22
27
|
const client_1 = require("../api/client");
|
|
23
28
|
const agentScanExecutor_1 = require("./agentScanExecutor");
|
|
29
|
+
const agentTrace_1 = require("./agentTrace");
|
|
30
|
+
const investigationState_1 = require("./investigationState");
|
|
31
|
+
const investigationProfiles_1 = require("./investigationProfiles");
|
|
32
|
+
const architectureContext_1 = require("../project-map/architectureContext");
|
|
33
|
+
const endpointDiscovery_1 = require("../project-map/endpointDiscovery");
|
|
34
|
+
const implementationResolver_1 = require("../project-map/implementationResolver");
|
|
35
|
+
const searchIntent_1 = require("./searchIntent");
|
|
36
|
+
const scanState_1 = require("./scanState");
|
|
37
|
+
const evidenceLedger_1 = require("./evidenceLedger");
|
|
38
|
+
const workItem_1 = require("./workItem");
|
|
39
|
+
const handlerInventory_1 = require("../project-map/handlerInventory");
|
|
40
|
+
const candidateStore_1 = require("./candidateStore");
|
|
41
|
+
const scanScheduler_1 = require("./scanScheduler");
|
|
42
|
+
const finishGate_1 = require("./finishGate");
|
|
43
|
+
const qualityMetrics_1 = require("./qualityMetrics");
|
|
44
|
+
const agentScanExecutor_2 = require("./agentScanExecutor");
|
|
45
|
+
const protocolValidator_1 = require("./protocolValidator");
|
|
24
46
|
const agentScanProtocol_1 = require("./agentScanProtocol");
|
|
25
47
|
async function runAgentScan(ctx, target, options = {}) {
|
|
26
48
|
const client = new client_1.ApiClient({ baseUrl: ctx.apiUrl, token: ctx.apiToken });
|
|
27
49
|
const budget = {
|
|
28
|
-
stepsRemaining: options.budget?.stepsRemaining ?? agentScanProtocol_1.AGENT_SCAN_DEFAULTS.
|
|
50
|
+
stepsRemaining: options.budget?.stepsRemaining ?? agentScanProtocol_1.AGENT_SCAN_DEFAULTS.initialSteps,
|
|
29
51
|
costSpentUsd: options.budget?.costSpentUsd ?? 0,
|
|
30
52
|
costCapUsd: options.budget?.costCapUsd ?? agentScanProtocol_1.AGENT_SCAN_DEFAULTS.costCapUsd,
|
|
53
|
+
stepsGranted: options.budget?.stepsGranted ?? agentScanProtocol_1.AGENT_SCAN_DEFAULTS.initialSteps,
|
|
54
|
+
hardMaxSteps: options.budget?.hardMaxSteps ?? agentScanProtocol_1.AGENT_SCAN_DEFAULTS.hardMaxSteps,
|
|
55
|
+
extensionsGranted: options.budget?.extensionsGranted ?? 0,
|
|
31
56
|
};
|
|
32
57
|
const startTime = Date.now();
|
|
33
58
|
const wallClockMs = agentScanProtocol_1.AGENT_SCAN_DEFAULTS.wallClockMs;
|
|
34
59
|
let stepsTaken = 0;
|
|
35
60
|
let costSpentUsd = 0;
|
|
61
|
+
let stepsGranted = budget.stepsGranted;
|
|
62
|
+
let extensionsGranted = budget.extensionsGranted;
|
|
63
|
+
let meaningfulProgressSinceLastExtension = false;
|
|
36
64
|
try {
|
|
37
|
-
const
|
|
65
|
+
const startRespRaw = await client.postJson('/agent/scan/start', {}, options.signal);
|
|
66
|
+
const startValidation = (0, protocolValidator_1.validateStartResponse)(startRespRaw);
|
|
67
|
+
if (!startValidation.ok) {
|
|
68
|
+
return {
|
|
69
|
+
status: 'spawn_failed',
|
|
70
|
+
findings: [],
|
|
71
|
+
transcript: [],
|
|
72
|
+
stepsUsed: 0,
|
|
73
|
+
stepsGranted: agentScanProtocol_1.AGENT_SCAN_DEFAULTS.initialSteps,
|
|
74
|
+
extensionsGranted: 0,
|
|
75
|
+
costSpentUsd: 0,
|
|
76
|
+
terminationReason: 'api_error',
|
|
77
|
+
error: `API returned an invalid start response: ${startValidation.error}`,
|
|
78
|
+
investigationNotes: [],
|
|
79
|
+
coverageGaps: [],
|
|
80
|
+
};
|
|
81
|
+
}
|
|
82
|
+
const startResp = startValidation.value;
|
|
83
|
+
const trace = new agentTrace_1.AgentTraceLogger(ctx.workspaceRoot, startResp.runId);
|
|
84
|
+
trace.logRunStarted();
|
|
85
|
+
// Authoritative scan state — the single source of truth for phase,
|
|
86
|
+
// budget, recovery, and lifecycle. Local counters below are kept in
|
|
87
|
+
// sync with this state until they are fully replaced.
|
|
88
|
+
const scanState = (0, scanState_1.createScanRunState)(startResp.runId, target, budget);
|
|
89
|
+
scanState.budget.wallClockMs = wallClockMs;
|
|
90
|
+
const qualityTracker = new qualityMetrics_1.QualityMetricsTracker();
|
|
38
91
|
const transcript = [];
|
|
92
|
+
const investigationState = new investigationState_1.InvestigationState();
|
|
93
|
+
// Select a target-specific investigation profile to set the required
|
|
94
|
+
// checklist steps. This prevents the agent from wasting steps on
|
|
95
|
+
// irrelevant tools (e.g., get_endpoints on a utility file) and ensures
|
|
96
|
+
// required steps for the target type are not skipped.
|
|
97
|
+
const profile = (0, investigationProfiles_1.selectInvestigationProfile)({
|
|
98
|
+
filePath: target.filePath,
|
|
99
|
+
architectureContext: target.architectureContext,
|
|
100
|
+
endpointContext: target.endpointContext,
|
|
101
|
+
});
|
|
102
|
+
investigationState.setRequiredSteps(profile.requiredSteps);
|
|
103
|
+
scanState.profileId = profile.name;
|
|
104
|
+
// Implementation resolution: if the target file looks like a
|
|
105
|
+
// contract-only file (interface, type declaration, abstract class),
|
|
106
|
+
// resolve the actual implementation so the investigation can follow
|
|
107
|
+
// the real code, not just signatures.
|
|
108
|
+
let implementationResolution = null;
|
|
109
|
+
try {
|
|
110
|
+
const fs = require('fs');
|
|
111
|
+
const absPath = require('path').resolve(ctx.workspaceRoot, target.filePath);
|
|
112
|
+
const content = fs.existsSync(absPath) ? fs.readFileSync(absPath, 'utf8') : '';
|
|
113
|
+
if (content && (0, implementationResolver_1.looksLikeContractOnly)(content)) {
|
|
114
|
+
const symbol = target.filePath.replace(/\.ts$/, '').replace(/.*\//, '');
|
|
115
|
+
const resolution = await (0, implementationResolver_1.resolveImplementation)(ctx.workspaceRoot, target.filePath, symbol);
|
|
116
|
+
if (!resolution.unresolved && resolution.implementationLocations.length > 0) {
|
|
117
|
+
implementationResolution = (0, implementationResolver_1.formatImplementationResolution)(resolution);
|
|
118
|
+
target.implementationHint = implementationResolution;
|
|
119
|
+
}
|
|
120
|
+
}
|
|
121
|
+
}
|
|
122
|
+
catch { /* best-effort */ }
|
|
123
|
+
// Control plane infrastructure — evidence ledger, work items, handler
|
|
124
|
+
// inventory, candidate store, and scheduler work together to ensure
|
|
125
|
+
// the investigation is complete before finish is accepted.
|
|
126
|
+
const evidenceLedger = new evidenceLedger_1.EvidenceLedger();
|
|
127
|
+
evidenceLedger.addRequirements(profile.requirements);
|
|
128
|
+
const workItemQueue = new workItem_1.WorkItemQueue();
|
|
129
|
+
// Note: profile requirements are tracked by the evidence ledger and
|
|
130
|
+
// checklist, not as work items. Work items are for architecture-risk
|
|
131
|
+
// tasks, handler reviews, and implementation reviews — structured
|
|
132
|
+
// work that the scheduler can prioritize and attempt-count.
|
|
133
|
+
const handlerInventory = new handlerInventory_1.HandlerInventory();
|
|
134
|
+
const candidateStore = new candidateStore_1.CandidateStore();
|
|
135
|
+
// Convert architecture risks relevant to this target into tracked
|
|
136
|
+
// investigation tasks. Unresolved tasks will appear as coverage gaps.
|
|
137
|
+
const archTasks = (0, architectureContext_1.createInvestigationTasksFromRisks)(target.architectureContext, target.filePath);
|
|
138
|
+
investigationState.addInvestigationTasks(archTasks);
|
|
139
|
+
// Create work items for architecture-risk tasks so the scheduler
|
|
140
|
+
// can prioritize them and suggest deterministic actions.
|
|
141
|
+
for (const task of archTasks) {
|
|
142
|
+
const requirements = [];
|
|
143
|
+
if (task.requiredProofDimensions?.includes('source') || task.requiredProofDimensions?.includes('reachability')) {
|
|
144
|
+
requirements.push({
|
|
145
|
+
id: `${task.id}-req-source`,
|
|
146
|
+
description: 'Trace the code path from input to the sensitive operation',
|
|
147
|
+
acceptedKinds: ['source-range', 'cross-file-flow'],
|
|
148
|
+
targetFiles: task.targetFiles,
|
|
149
|
+
requiredTools: ['read_file', 'trace_flow_cross_file', 'trace_flow'],
|
|
150
|
+
minimumCount: 1,
|
|
151
|
+
});
|
|
152
|
+
}
|
|
153
|
+
if (task.requiredProofDimensions?.includes('control')) {
|
|
154
|
+
requirements.push({
|
|
155
|
+
id: `${task.id}-req-control`,
|
|
156
|
+
description: 'Determine whether the control is present, bypassed, or missing',
|
|
157
|
+
acceptedKinds: ['guard-result', 'policy-result', 'config-result'],
|
|
158
|
+
targetFiles: task.targetFiles,
|
|
159
|
+
requiredTools: ['check_guard', 'check_policy', 'read_config'],
|
|
160
|
+
minimumCount: 1,
|
|
161
|
+
acceptsNegative: true,
|
|
162
|
+
});
|
|
163
|
+
}
|
|
164
|
+
if (task.requiredProofDimensions?.includes('threat-model')) {
|
|
165
|
+
requirements.push({
|
|
166
|
+
id: `${task.id}-req-threat-model`,
|
|
167
|
+
description: 'Establish the threat model (is the attacker in scope?)',
|
|
168
|
+
acceptedKinds: ['threat-model-result', 'config-result', 'source-range'],
|
|
169
|
+
targetFiles: task.targetFiles,
|
|
170
|
+
requiredTools: ['read_config', 'read_file', 'search_code'],
|
|
171
|
+
minimumCount: 1,
|
|
172
|
+
acceptsNegative: true,
|
|
173
|
+
});
|
|
174
|
+
}
|
|
175
|
+
if (task.requiredProofDimensions?.includes('impact')) {
|
|
176
|
+
requirements.push({
|
|
177
|
+
id: `${task.id}-req-impact`,
|
|
178
|
+
description: 'Verify the impact: trace to the actual sensitive sink and confirm reachability',
|
|
179
|
+
acceptedKinds: ['cross-file-flow', 'capability-result', 'source-range'],
|
|
180
|
+
targetFiles: task.targetFiles,
|
|
181
|
+
requiredTools: ['trace_flow_cross_file', 'trace_flow'],
|
|
182
|
+
minimumCount: 1,
|
|
183
|
+
});
|
|
184
|
+
}
|
|
185
|
+
if (task.requiredProofDimensions?.includes('verification')) {
|
|
186
|
+
requirements.push({
|
|
187
|
+
id: `${task.id}-req-verification`,
|
|
188
|
+
description: 'Verify the vulnerability with a test or exploit proof',
|
|
189
|
+
acceptedKinds: ['test-location', 'test-result', 'proof-result'],
|
|
190
|
+
targetFiles: task.targetFiles,
|
|
191
|
+
requiredTools: ['find_tests', 'run_tests'],
|
|
192
|
+
minimumCount: 1,
|
|
193
|
+
acceptsNegative: true,
|
|
194
|
+
});
|
|
195
|
+
}
|
|
196
|
+
if (requirements.length === 0) {
|
|
197
|
+
requirements.push({
|
|
198
|
+
id: `${task.id}-req-0`,
|
|
199
|
+
description: task.requiredEvidence[0] || 'Investigate the risk',
|
|
200
|
+
acceptedKinds: ['source-range', 'cross-file-flow', 'policy-result'],
|
|
201
|
+
minimumCount: 1,
|
|
202
|
+
});
|
|
203
|
+
}
|
|
204
|
+
workItemQueue.add((0, workItem_1.createArchitectureRiskWorkItem)(task.claim, task.targetFiles, requirements));
|
|
205
|
+
}
|
|
39
206
|
const readFiles = new Set();
|
|
40
207
|
const readFileCounts = new Map();
|
|
208
|
+
// Pre-compute function boundaries for chunk alignment
|
|
209
|
+
let functionBoundaries;
|
|
210
|
+
try {
|
|
211
|
+
const fb = await (0, agentScanExecutor_2.extractFunctionBoundaries)(target.fileContent, target.filePath);
|
|
212
|
+
if (fb)
|
|
213
|
+
functionBoundaries = fb;
|
|
214
|
+
}
|
|
215
|
+
catch { /* best-effort */ }
|
|
41
216
|
// Dynamic read cap based on file size:
|
|
42
217
|
// < 200 lines → 5 reads (small file, 5 chunks is enough)
|
|
43
|
-
// < 1000 lines →
|
|
44
|
-
// < 5000 lines →
|
|
45
|
-
// >= 5000 lines →
|
|
218
|
+
// < 1000 lines → 8 reads (medium file)
|
|
219
|
+
// < 5000 lines → 12 reads (large file — use search_code/trace_flow, not brute reading)
|
|
220
|
+
// >= 5000 lines → 15 reads (very large — still capped; the agent must use
|
|
221
|
+
// search_code and trace_flow for pattern discovery, not
|
|
222
|
+
// read the entire file section by section)
|
|
46
223
|
function maxReadsForFile(filePath) {
|
|
47
224
|
try {
|
|
48
225
|
const fs = require('fs');
|
|
49
226
|
const abs = require('path').resolve(ctx.workspaceRoot, filePath);
|
|
50
227
|
const stat = fs.statSync(abs);
|
|
51
228
|
if (stat.size > 200_000)
|
|
52
|
-
return
|
|
229
|
+
return 15;
|
|
53
230
|
const content = fs.readFileSync(abs, 'utf8');
|
|
54
231
|
const lines = content.split('\n').length;
|
|
55
232
|
if (lines < 200)
|
|
56
233
|
return 5;
|
|
57
234
|
if (lines < 1000)
|
|
58
|
-
return
|
|
235
|
+
return 8;
|
|
59
236
|
if (lines < 5000)
|
|
60
|
-
return
|
|
61
|
-
return
|
|
237
|
+
return 12;
|
|
238
|
+
return 15;
|
|
62
239
|
}
|
|
63
240
|
catch {
|
|
64
|
-
return
|
|
241
|
+
return 8;
|
|
65
242
|
}
|
|
66
243
|
}
|
|
67
244
|
// Track non-read tool calls to prevent the agent from looping on the
|
|
68
245
|
// same search_code/trace_flow call repeatedly. Keyed by (type + args).
|
|
69
246
|
const toolCallCounts = new Map();
|
|
70
247
|
const MAX_SAME_TOOL_CALL = 2;
|
|
248
|
+
// Track normalized search patterns to detect equivalent searches
|
|
249
|
+
// (same terms, different order) that the raw toolKey dedup misses.
|
|
250
|
+
const searchedPatterns = new Set();
|
|
251
|
+
let equivalentSearchCount = 0;
|
|
71
252
|
let consecutiveErrors = 0;
|
|
72
253
|
let consecutiveBlockedReads = 0;
|
|
254
|
+
let meaningfulProgressSinceRecovery = true;
|
|
255
|
+
let aggressiveCompaction = false;
|
|
73
256
|
while (true) {
|
|
74
257
|
// Wall clock check
|
|
75
258
|
if (Date.now() - startTime > wallClockMs) {
|
|
259
|
+
(0, scanState_1.terminateScan)(scanState, 'wall_clock', `Wall clock limit (${wallClockMs}ms) exceeded.`);
|
|
76
260
|
return {
|
|
77
261
|
status: 'capped',
|
|
78
262
|
findings: [],
|
|
79
263
|
transcript,
|
|
80
264
|
stepsUsed: stepsTaken,
|
|
265
|
+
stepsGranted,
|
|
266
|
+
extensionsGranted,
|
|
81
267
|
costSpentUsd,
|
|
268
|
+
terminationReason: 'wall_clock',
|
|
82
269
|
summary: `Wall clock limit (${wallClockMs}ms) exceeded.`,
|
|
270
|
+
investigationNotes: [],
|
|
271
|
+
coverageGaps: [],
|
|
83
272
|
};
|
|
84
273
|
}
|
|
85
274
|
// Abort check
|
|
86
275
|
if (options.signal?.aborted) {
|
|
276
|
+
(0, scanState_1.terminateScan)(scanState, 'cancelled', 'Cancelled by user.');
|
|
87
277
|
return {
|
|
88
278
|
status: 'cancelled',
|
|
89
279
|
findings: [],
|
|
90
280
|
transcript,
|
|
91
281
|
stepsUsed: stepsTaken,
|
|
282
|
+
stepsGranted,
|
|
283
|
+
extensionsGranted,
|
|
92
284
|
costSpentUsd,
|
|
285
|
+
terminationReason: 'cancelled',
|
|
93
286
|
summary: 'Cancelled by user.',
|
|
287
|
+
investigationNotes: [],
|
|
288
|
+
coverageGaps: [],
|
|
94
289
|
};
|
|
95
290
|
}
|
|
291
|
+
// Build action constraint for blocked-read recovery.
|
|
292
|
+
// 0-1 blocked reads: normal (no constraint)
|
|
293
|
+
// 2 blocked reads: recovery mode — if unread ranges exist,
|
|
294
|
+
// require the next unread range; if no unread ranges remain,
|
|
295
|
+
// forbid read_file entirely and require an analysis tool.
|
|
296
|
+
// 3 blocked reads: MCP selects a deterministic recovery action (skip API)
|
|
297
|
+
let actionConstraint;
|
|
298
|
+
if (consecutiveBlockedReads >= 2 && consecutiveBlockedReads < 3) {
|
|
299
|
+
const nextRange = target.fileContent
|
|
300
|
+
? investigationState.getPrioritizedUnreadRange(target.filePath, target.fileContent)
|
|
301
|
+
: investigationState.getNextUnreadRange(target.filePath);
|
|
302
|
+
if (nextRange) {
|
|
303
|
+
actionConstraint = {
|
|
304
|
+
mode: 'recovery',
|
|
305
|
+
requiredAction: {
|
|
306
|
+
type: 'read_file',
|
|
307
|
+
path: target.filePath,
|
|
308
|
+
startLine: nextRange.start,
|
|
309
|
+
endLine: nextRange.end,
|
|
310
|
+
rationale: `Read next unread range: lines ${nextRange.start}-${nextRange.end}`,
|
|
311
|
+
},
|
|
312
|
+
reason: `You have been blocked by duplicate/overlapping reads. Read the next unread range: lines ${nextRange.start}-${nextRange.end}, or use an analysis tool (trace_flow, check_guard, check_policy).`,
|
|
313
|
+
};
|
|
314
|
+
}
|
|
315
|
+
else {
|
|
316
|
+
const recoveryAction = investigationState.getRecommendedRecoveryAction();
|
|
317
|
+
actionConstraint = {
|
|
318
|
+
mode: 'recovery',
|
|
319
|
+
forbiddenActions: ['read_file'],
|
|
320
|
+
requiredAction: recoveryAction || undefined,
|
|
321
|
+
reason: 'You have been blocked by duplicate/overlapping reads and no unread ranges remain. Switch to an analysis tool.',
|
|
322
|
+
};
|
|
323
|
+
}
|
|
324
|
+
}
|
|
325
|
+
const compactedTranscript = aggressiveCompaction
|
|
326
|
+
? compactTranscriptAggressive(transcript)
|
|
327
|
+
: compactTranscript(transcript);
|
|
96
328
|
const stepReq = {
|
|
97
329
|
runId: startResp.runId,
|
|
98
330
|
target,
|
|
99
|
-
transcript,
|
|
331
|
+
transcript: compactedTranscript,
|
|
100
332
|
budget: {
|
|
101
333
|
stepsRemaining: budget.stepsRemaining,
|
|
102
334
|
costSpentUsd,
|
|
103
335
|
costCapUsd: budget.costCapUsd,
|
|
336
|
+
stepsGranted,
|
|
337
|
+
hardMaxSteps: budget.hardMaxSteps,
|
|
338
|
+
extensionsGranted,
|
|
339
|
+
},
|
|
340
|
+
clientCapabilities: (0, agentScanProtocol_1.defaultClientCapabilities)(),
|
|
341
|
+
investigationProgress: {
|
|
342
|
+
completedSteps: investigationState.getCompletedSteps(),
|
|
343
|
+
incompleteSteps: investigationState.getIncompleteSteps(),
|
|
344
|
+
consecutiveBlockedReads,
|
|
345
|
+
meaningfulProgressSinceLastExtension,
|
|
104
346
|
},
|
|
347
|
+
actionConstraint,
|
|
105
348
|
};
|
|
106
349
|
let stepResp;
|
|
107
350
|
try {
|
|
108
|
-
|
|
351
|
+
const stepRespRaw = await client.postJson('/agent/scan/step', stepReq, options.signal);
|
|
352
|
+
const stepValidation = (0, protocolValidator_1.validateStepResponse)(stepRespRaw);
|
|
353
|
+
if (!stepValidation.ok) {
|
|
354
|
+
// Wire-level malformed response. Don't execute the
|
|
355
|
+
// potentially-dangerous `next` action; treat it as a
|
|
356
|
+
// controlled error so the agent can retry against a
|
|
357
|
+
// well-formed response on the next call.
|
|
358
|
+
const vErr = stepValidation.error;
|
|
359
|
+
console.warn(`[Agent Scan Loop] Step ${stepsTaken + 1} returned a malformed response: ${vErr}`);
|
|
360
|
+
consecutiveErrors++;
|
|
361
|
+
stepsTaken++;
|
|
362
|
+
budget.stepsRemaining--;
|
|
363
|
+
const errMsg = `API returned a malformed step response: ${vErr}`;
|
|
364
|
+
transcript.push({
|
|
365
|
+
action: {
|
|
366
|
+
type: 'system_event',
|
|
367
|
+
eventType: 'error',
|
|
368
|
+
message: errMsg,
|
|
369
|
+
},
|
|
370
|
+
observation: errMsg,
|
|
371
|
+
});
|
|
372
|
+
if (consecutiveErrors >= 3 || budget.stepsRemaining <= 0) {
|
|
373
|
+
return {
|
|
374
|
+
status: 'capped',
|
|
375
|
+
findings: [],
|
|
376
|
+
transcript,
|
|
377
|
+
stepsUsed: stepsTaken,
|
|
378
|
+
stepsGranted,
|
|
379
|
+
extensionsGranted,
|
|
380
|
+
costSpentUsd,
|
|
381
|
+
terminationReason: 'api_error',
|
|
382
|
+
summary: `Step budget exhausted after ${consecutiveErrors} consecutive malformed API responses. Last error: ${vErr}`,
|
|
383
|
+
investigationNotes: [],
|
|
384
|
+
coverageGaps: [],
|
|
385
|
+
};
|
|
386
|
+
}
|
|
387
|
+
continue;
|
|
388
|
+
}
|
|
389
|
+
stepResp = stepValidation.value;
|
|
109
390
|
}
|
|
110
391
|
catch (stepErr) {
|
|
111
392
|
const errMsg = stepErr?.message || String(stepErr);
|
|
@@ -119,8 +400,13 @@ async function runAgentScan(ctx, target, options = {}) {
|
|
|
119
400
|
findings: [],
|
|
120
401
|
transcript,
|
|
121
402
|
stepsUsed: stepsTaken,
|
|
403
|
+
stepsGranted,
|
|
404
|
+
extensionsGranted,
|
|
122
405
|
costSpentUsd,
|
|
406
|
+
terminationReason: 'api_error',
|
|
123
407
|
error: 'API server restarted mid-scan — the run was lost. Please retry the scan.',
|
|
408
|
+
investigationNotes: [],
|
|
409
|
+
coverageGaps: [],
|
|
124
410
|
};
|
|
125
411
|
}
|
|
126
412
|
// Detect abort — the user cancelled. Don't treat as error.
|
|
@@ -130,26 +416,38 @@ async function runAgentScan(ctx, target, options = {}) {
|
|
|
130
416
|
findings: [],
|
|
131
417
|
transcript,
|
|
132
418
|
stepsUsed: stepsTaken,
|
|
419
|
+
stepsGranted,
|
|
420
|
+
extensionsGranted,
|
|
133
421
|
costSpentUsd,
|
|
422
|
+
terminationReason: 'cancelled',
|
|
134
423
|
summary: 'Cancelled by user.',
|
|
424
|
+
investigationNotes: [],
|
|
425
|
+
coverageGaps: [],
|
|
135
426
|
};
|
|
136
427
|
}
|
|
428
|
+
// Detect timeout / network errors — retry once with aggressive compaction
|
|
429
|
+
if (!aggressiveCompaction && /timeout|ETIMEDOUT|ESOCKETTIMEDOUT|socket hang up|ECONNRESET|fetch failed/i.test(errMsg)) {
|
|
430
|
+
console.warn(`[Agent Scan Loop] Step ${stepsTaken + 1} timeout/network error — retrying with aggressive compaction: ${errMsg}`);
|
|
431
|
+
aggressiveCompaction = true;
|
|
432
|
+
continue;
|
|
433
|
+
}
|
|
137
434
|
// The API rejected the action (malformed, missing field, etc).
|
|
138
|
-
// Add
|
|
139
|
-
// next step and can correct itself.
|
|
140
|
-
//
|
|
435
|
+
// Add a first-class system_event to the transcript so the LLM
|
|
436
|
+
// sees the error on the next step and can correct itself.
|
|
437
|
+
// Previously this piggybacked on `read_file('__ERROR__')`,
|
|
438
|
+
// which the executor would have tried to open and failed.
|
|
141
439
|
console.warn(`[Agent Scan Loop] Step ${stepsTaken + 1} error: ${errMsg}`);
|
|
142
440
|
consecutiveErrors++;
|
|
143
441
|
stepsTaken++;
|
|
144
442
|
budget.stepsRemaining--;
|
|
145
|
-
|
|
443
|
+
const errorMsg = `ERROR: Your previous action was rejected: ${errMsg}. Please try a DIFFERENT action with ALL required fields. Set unused fields to null.`;
|
|
146
444
|
transcript.push({
|
|
147
445
|
action: {
|
|
148
|
-
type: '
|
|
149
|
-
|
|
150
|
-
|
|
446
|
+
type: 'system_event',
|
|
447
|
+
eventType: 'error',
|
|
448
|
+
message: errorMsg,
|
|
151
449
|
},
|
|
152
|
-
observation:
|
|
450
|
+
observation: errorMsg,
|
|
153
451
|
});
|
|
154
452
|
if (consecutiveErrors >= 3 || budget.stepsRemaining <= 0) {
|
|
155
453
|
return {
|
|
@@ -157,74 +455,367 @@ async function runAgentScan(ctx, target, options = {}) {
|
|
|
157
455
|
findings: [],
|
|
158
456
|
transcript,
|
|
159
457
|
stepsUsed: stepsTaken,
|
|
458
|
+
stepsGranted,
|
|
459
|
+
extensionsGranted,
|
|
160
460
|
costSpentUsd,
|
|
461
|
+
terminationReason: 'api_error',
|
|
161
462
|
summary: `Step budget exhausted after ${consecutiveErrors} consecutive API errors. Last error: ${errMsg}`,
|
|
463
|
+
investigationNotes: [],
|
|
464
|
+
coverageGaps: [],
|
|
162
465
|
};
|
|
163
466
|
}
|
|
164
467
|
continue;
|
|
165
468
|
}
|
|
166
469
|
costSpentUsd += stepResp.costUsd || 0;
|
|
470
|
+
scanState.budget.costSpentUsd = costSpentUsd;
|
|
167
471
|
consecutiveErrors = 0;
|
|
472
|
+
trace.logStepRequested(stepResp.model, stepResp.tokens, stepResp.costUsd, stepResp.latencyMs);
|
|
473
|
+
// Transition from planning to surveying on first successful step
|
|
474
|
+
if (scanState.phase === 'planning') {
|
|
475
|
+
(0, scanState_1.transitionScanPhase)(scanState, 'surveying', 'first step received');
|
|
476
|
+
}
|
|
477
|
+
// Budget extension — the API may grant additional steps when the
|
|
478
|
+
// agent demonstrates meaningful progress. Update our local tracking.
|
|
479
|
+
if (stepResp.budgetExtension) {
|
|
480
|
+
const ext = stepResp.budgetExtension;
|
|
481
|
+
stepsGranted = ext.totalGranted;
|
|
482
|
+
extensionsGranted = ext.granted > 0 ? extensionsGranted + 1 : extensionsGranted;
|
|
483
|
+
budget.stepsRemaining = stepResp.stepsRemaining;
|
|
484
|
+
budget.stepsGranted = ext.totalGranted;
|
|
485
|
+
budget.extensionsGranted = extensionsGranted;
|
|
486
|
+
scanState.budget.stepsGranted = ext.totalGranted;
|
|
487
|
+
scanState.budget.extensionsGranted = extensionsGranted;
|
|
488
|
+
scanState.budget.stepsRemaining = stepResp.stepsRemaining;
|
|
489
|
+
meaningfulProgressSinceLastExtension = false;
|
|
490
|
+
if (options.onProgress) {
|
|
491
|
+
options.onProgress(stepsTaken, ext.hardMaxSteps, `Budget extended +${ext.granted} steps (${ext.totalGranted}/${ext.hardMaxSteps}): ${ext.reason}`);
|
|
492
|
+
}
|
|
493
|
+
}
|
|
494
|
+
// System event (e.g., critique from the senior reviewer) — append
|
|
495
|
+
// to the transcript WITHOUT executing it, then continue the loop
|
|
496
|
+
// so the next step's prompt sees the event and re-plans. This
|
|
497
|
+
// replaces the previous `read_file('__CRITIQUE__')` pattern that
|
|
498
|
+
// lost the critique content over the wire.
|
|
499
|
+
if (stepResp.systemEvent) {
|
|
500
|
+
const ev = stepResp.systemEvent;
|
|
501
|
+
if (ev.eventType === 'critique')
|
|
502
|
+
qualityTracker.recordCritique();
|
|
503
|
+
if (ev.eventType === 'finish_gate')
|
|
504
|
+
qualityTracker.recordFinishGate();
|
|
505
|
+
transcript.push({
|
|
506
|
+
action: ev,
|
|
507
|
+
observation: ev.message,
|
|
508
|
+
});
|
|
509
|
+
stepsTaken++;
|
|
510
|
+
budget.stepsRemaining = stepResp.stepsRemaining;
|
|
511
|
+
if (options.onProgress) {
|
|
512
|
+
options.onProgress(stepsTaken, stepsGranted, `Step ${stepsTaken}: system_event(${ev.eventType})`);
|
|
513
|
+
}
|
|
514
|
+
// Re-loop: ask the API for the next action now that the
|
|
515
|
+
// critique is in the transcript. The agent will see the
|
|
516
|
+
// critique in the next step's prompt and either re-investigate
|
|
517
|
+
// or call finish again with an updated selfCritique.
|
|
518
|
+
continue;
|
|
519
|
+
}
|
|
168
520
|
// Null next = done (cost capped or steps exhausted)
|
|
169
521
|
if (!stepResp.next) {
|
|
170
522
|
const status = stepResp.costCapped
|
|
171
523
|
? 'capped'
|
|
172
524
|
: (stepResp.degraded ? 'degraded' : 'completed');
|
|
525
|
+
trace.logRunCompleted(status);
|
|
173
526
|
return {
|
|
174
527
|
status,
|
|
175
528
|
findings: [],
|
|
176
529
|
transcript,
|
|
177
530
|
stepsUsed: stepsTaken,
|
|
531
|
+
stepsGranted,
|
|
532
|
+
extensionsGranted,
|
|
178
533
|
costSpentUsd,
|
|
534
|
+
terminationReason: stepResp.costCapped ? 'cost_cap' : 'budget_exhausted',
|
|
179
535
|
summary: stepResp.costCapped
|
|
180
536
|
? `Cost cap ($${budget.costCapUsd.toFixed(2)}) reached.`
|
|
181
537
|
: 'Agent completed without explicit finish.',
|
|
538
|
+
investigationNotes: [],
|
|
539
|
+
coverageGaps: [],
|
|
182
540
|
};
|
|
183
541
|
}
|
|
184
542
|
const action = stepResp.next;
|
|
185
543
|
stepsTaken++;
|
|
186
544
|
budget.stepsRemaining = stepResp.stepsRemaining;
|
|
545
|
+
scanState.budget.stepsUsed = stepsTaken;
|
|
546
|
+
scanState.budget.stepsRemaining = stepResp.stepsRemaining;
|
|
547
|
+
trace.nextStep();
|
|
548
|
+
trace.logActionSelected(action.type, stepResp.model, stepResp.degraded || stepResp.fallbackFired);
|
|
187
549
|
// Progress callback
|
|
188
550
|
if (options.onProgress) {
|
|
189
551
|
const actionDesc = describeAction(action);
|
|
190
|
-
options.onProgress(stepsTaken,
|
|
552
|
+
options.onProgress(stepsTaken, stepsGranted, `Step ${stepsTaken}: ${actionDesc}`);
|
|
191
553
|
}
|
|
192
554
|
// Finish = done with findings
|
|
193
555
|
if (action.type === 'finish') {
|
|
194
|
-
|
|
195
|
-
|
|
196
|
-
|
|
197
|
-
|
|
198
|
-
|
|
199
|
-
|
|
200
|
-
|
|
201
|
-
|
|
556
|
+
scanState.finishAttempts++;
|
|
557
|
+
qualityTracker.recordFinishAttempt();
|
|
558
|
+
// Register candidates from findings — each finding becomes a
|
|
559
|
+
// tracked candidate. If the candidate is new (discovered), the
|
|
560
|
+
// finish gate will reject and the agent must gather evidence.
|
|
561
|
+
for (const finding of action.findings) {
|
|
562
|
+
const rootCauseId = finding.rootCause?.rootCauseId || `${finding.type}:${finding.line}`;
|
|
563
|
+
const existing = candidateStore.getCandidatesByRootCause(rootCauseId);
|
|
564
|
+
if (existing.length === 0) {
|
|
565
|
+
const candidateId = candidateStore.register({
|
|
566
|
+
rootCauseId,
|
|
567
|
+
type: finding.type,
|
|
568
|
+
severity: finding.severity,
|
|
569
|
+
locations: [{ filePath: target.filePath, line: finding.line }],
|
|
570
|
+
claim: finding.why,
|
|
571
|
+
requiredEvidence: [
|
|
572
|
+
{
|
|
573
|
+
id: `${rootCauseId}-flow`,
|
|
574
|
+
description: 'Verify data flow to the vulnerable location',
|
|
575
|
+
acceptedKinds: ['cross-file-flow'],
|
|
576
|
+
targetFiles: [target.filePath],
|
|
577
|
+
requiredTools: ['trace_flow_cross_file', 'trace_flow'],
|
|
578
|
+
minimumCount: 1,
|
|
579
|
+
},
|
|
580
|
+
{
|
|
581
|
+
id: `${rootCauseId}-guard`,
|
|
582
|
+
description: 'Check for guards/controls on the vulnerable location',
|
|
583
|
+
acceptedKinds: ['guard-result', 'policy-result'],
|
|
584
|
+
targetFiles: [target.filePath],
|
|
585
|
+
requiredTools: ['check_guard', 'check_policy'],
|
|
586
|
+
minimumCount: 1,
|
|
587
|
+
},
|
|
588
|
+
],
|
|
589
|
+
});
|
|
590
|
+
// Backfill evidence: scan the transcript for prior
|
|
591
|
+
// verification actions on this candidate's location.
|
|
592
|
+
// If the agent already ran trace_flow/check_guard on
|
|
593
|
+
// the file before calling finish, credit that evidence.
|
|
594
|
+
const targetFileNorm = target.filePath.replace(/\\/g, '/').toLowerCase();
|
|
595
|
+
const verificationActions = new Set([
|
|
596
|
+
'trace_flow', 'trace_flow_cross_file', 'check_guard', 'check_policy',
|
|
597
|
+
]);
|
|
598
|
+
for (const step of transcript) {
|
|
599
|
+
if (verificationActions.has(step.action.type)) {
|
|
600
|
+
const stepFile = step.action.filePath || step.action.path || target.filePath;
|
|
601
|
+
const stepFileNorm = String(stepFile).replace(/\\/g, '/').toLowerCase();
|
|
602
|
+
if (stepFileNorm === targetFileNorm) {
|
|
603
|
+
const cat = step.action.type === 'trace_flow' || step.action.type === 'trace_flow_cross_file' ? 'flow'
|
|
604
|
+
: step.action.type === 'check_guard' ? 'guard'
|
|
605
|
+
: step.action.type === 'check_policy' ? 'policy'
|
|
606
|
+
: undefined;
|
|
607
|
+
const dim = step.action.type === 'trace_flow' || step.action.type === 'trace_flow_cross_file' ? 'reachability'
|
|
608
|
+
: step.action.type === 'check_guard' || step.action.type === 'check_policy' ? 'control'
|
|
609
|
+
: undefined;
|
|
610
|
+
candidateStore.addEvidence(candidateId, `backfill:${step.action.type}:${stepFileNorm}`, cat, dim);
|
|
611
|
+
}
|
|
612
|
+
}
|
|
613
|
+
if (step.action.type === 'read_file') {
|
|
614
|
+
const stepFile = step.action.path || target.filePath;
|
|
615
|
+
const stepFileNorm = String(stepFile).replace(/\\/g, '/').toLowerCase();
|
|
616
|
+
if (stepFileNorm === targetFileNorm) {
|
|
617
|
+
candidateStore.addEvidence(candidateId, `backfill:source:${stepFileNorm}`, 'source', 'source');
|
|
618
|
+
}
|
|
619
|
+
}
|
|
620
|
+
if (step.action.type === 'read_config') {
|
|
621
|
+
const stepFile = target.filePath;
|
|
622
|
+
const stepFileNorm = stepFile.replace(/\\/g, '/').toLowerCase();
|
|
623
|
+
candidateStore.addEvidence(candidateId, `backfill:config:${stepFileNorm}`, undefined, 'threat-model');
|
|
624
|
+
}
|
|
625
|
+
}
|
|
626
|
+
}
|
|
627
|
+
}
|
|
628
|
+
// Evaluate the finish proposal through the hard finish gate
|
|
629
|
+
const schedDecision = (0, scanScheduler_1.schedule)({
|
|
630
|
+
state: scanState,
|
|
631
|
+
evidence: evidenceLedger,
|
|
632
|
+
workItems: workItemQueue,
|
|
633
|
+
handlers: handlerInventory,
|
|
634
|
+
candidates: candidateStore,
|
|
635
|
+
investigation: investigationState,
|
|
636
|
+
target,
|
|
637
|
+
functionBoundaries,
|
|
638
|
+
});
|
|
639
|
+
const gateResult = (0, finishGate_1.evaluateFinishGate)({
|
|
640
|
+
proposal: action,
|
|
641
|
+
state: scanState,
|
|
642
|
+
evidence: evidenceLedger,
|
|
643
|
+
workItems: workItemQueue,
|
|
644
|
+
handlers: handlerInventory,
|
|
645
|
+
candidates: candidateStore,
|
|
646
|
+
investigation: investigationState,
|
|
647
|
+
scheduler: schedDecision,
|
|
648
|
+
target: { filePath: target.filePath, fileContent: target.fileContent },
|
|
649
|
+
});
|
|
650
|
+
if (gateResult.accepted) {
|
|
651
|
+
trace.logRunCompleted('completed');
|
|
652
|
+
if (gateResult.mode === 'forced-incomplete') {
|
|
653
|
+
(0, scanState_1.terminateScan)(scanState, 'forced_incomplete', 'Budget exhausted with incomplete investigation');
|
|
654
|
+
qualityTracker.recordForcedTermination('forced_incomplete');
|
|
655
|
+
}
|
|
656
|
+
else {
|
|
657
|
+
(0, scanState_1.terminateScan)(scanState, 'agent_finish', gateResult.normalizedFinish?.summary);
|
|
658
|
+
qualityTracker.recordTermination('agent_finish');
|
|
659
|
+
}
|
|
660
|
+
// Merge gate-generated coverage gaps with model-provided gaps
|
|
661
|
+
let coverageGaps = action.coverageGaps ?? [];
|
|
662
|
+
if (gateResult.coverageGaps) {
|
|
663
|
+
coverageGaps = [...coverageGaps, ...gateResult.coverageGaps];
|
|
664
|
+
}
|
|
665
|
+
const covSummary = investigationState.getCoverageSummary(target.filePath);
|
|
666
|
+
qualityTracker.recordCoverage(covSummary.totalLines, covSummary.coveredLines, covSummary.uncoveredRangeCount, covSummary.largestUncoveredRange);
|
|
667
|
+
qualityTracker.recordBudget(stepsTaken, stepsGranted, extensionsGranted, costSpentUsd, budget.costCapUsd, wallClockMs, Date.now() - startTime);
|
|
668
|
+
const investigationNotes = [...(action.investigationNotes ?? [])];
|
|
669
|
+
for (const candidate of candidateStore.getActive()) {
|
|
670
|
+
if (candidate.status === 'discovered' || candidate.status === 'investigating' || candidate.status === 'supported') {
|
|
671
|
+
const missingDims = candidate.requiredProofDimensions?.filter(d => !candidate.satisfiedDimensions?.includes(d)) || [];
|
|
672
|
+
investigationNotes.push({
|
|
673
|
+
title: candidate.claim,
|
|
674
|
+
detail: `Candidate was not fully verified. Status: ${candidate.status}. Missing proof dimensions: ${missingDims.join(', ') || 'none'}. Evidence refs: ${candidate.evidenceRefs.length}.`,
|
|
675
|
+
file: candidate.locations[0]?.filePath || target.filePath,
|
|
676
|
+
line: candidate.locations[0]?.line,
|
|
677
|
+
verificationLevel: 'logic-confirmed',
|
|
678
|
+
rootCauseId: candidate.rootCauseId,
|
|
679
|
+
requiredEvidence: missingDims.map(d => `Satisfy the ${d} proof dimension`),
|
|
680
|
+
priority: candidate.severity === 'critical' || candidate.severity === 'high' ? 'high' : 'medium',
|
|
681
|
+
});
|
|
682
|
+
candidateStore.setUnproven(candidate.id, `Converted to note: missing proof dimensions ${missingDims.join(', ')}`);
|
|
683
|
+
}
|
|
684
|
+
}
|
|
685
|
+
return {
|
|
686
|
+
status: 'completed',
|
|
687
|
+
findings: sanitizeFindings(action.findings),
|
|
688
|
+
investigationNotes,
|
|
689
|
+
coverageGaps,
|
|
690
|
+
transcript,
|
|
691
|
+
stepsUsed: stepsTaken,
|
|
692
|
+
stepsGranted,
|
|
693
|
+
extensionsGranted,
|
|
694
|
+
costSpentUsd,
|
|
695
|
+
terminationReason: gateResult.mode === 'forced-incomplete' ? 'forced_incomplete' : 'agent_finish',
|
|
696
|
+
summary: action.summary,
|
|
697
|
+
qualityMetrics: qualityTracker.getMetrics(),
|
|
698
|
+
};
|
|
699
|
+
}
|
|
700
|
+
// Finish rejected — continue investigation
|
|
701
|
+
trace.logToolBlocked('finish', `Finish rejected: ${gateResult.reasons.map(r => r.description).join('; ')}`);
|
|
702
|
+
qualityTracker.recordFinishRejection();
|
|
703
|
+
// Add a system event so the model sees the rejection
|
|
704
|
+
const rejectionMessage = `FINISH REJECTED — your investigation is incomplete:\n${gateResult.reasons.map(r => ` - ${r.description}`).join('\n')}\n\nYou must complete the remaining investigation steps before calling finish.${gateResult.recoveryAction ? `\n\nYour next action MUST be: ${gateResult.recoveryAction.type}` : ''}`;
|
|
705
|
+
transcript.push({
|
|
706
|
+
action: {
|
|
707
|
+
type: 'system_event',
|
|
708
|
+
eventType: 'finish_rejected',
|
|
709
|
+
message: rejectionMessage,
|
|
710
|
+
},
|
|
711
|
+
observation: rejectionMessage,
|
|
712
|
+
});
|
|
713
|
+
// Execute the recovery action if the scheduler has one
|
|
714
|
+
if (gateResult.recoveryAction) {
|
|
715
|
+
const recoveryAction = gateResult.recoveryAction;
|
|
716
|
+
stepsTaken++;
|
|
717
|
+
budget.stepsRemaining--;
|
|
718
|
+
scanState.budget.stepsUsed = stepsTaken;
|
|
719
|
+
scanState.budget.stepsRemaining = budget.stepsRemaining;
|
|
720
|
+
trace.nextStep();
|
|
721
|
+
trace.logActionSelected(recoveryAction.type, 'finish-gate-recovery', false);
|
|
722
|
+
if (options.onProgress) {
|
|
723
|
+
options.onProgress(stepsTaken, stepsGranted, `Step ${stepsTaken}: finish-gate recovery: ${describeAction(recoveryAction)}`);
|
|
724
|
+
}
|
|
725
|
+
let recoveryObservation;
|
|
726
|
+
if (recoveryAction.type === 'read_file') {
|
|
727
|
+
const readResult = await (0, agentScanExecutor_1.executeReadFileAction)(recoveryAction, ctx);
|
|
728
|
+
recoveryObservation = readResult.observation;
|
|
729
|
+
if (readResult.totalLines > 0) {
|
|
730
|
+
investigationState.recordActualRead(recoveryAction.path, readResult.actualStart, readResult.actualEnd, readResult.totalLines, readResult.truncated);
|
|
731
|
+
}
|
|
732
|
+
}
|
|
733
|
+
else {
|
|
734
|
+
recoveryObservation = await (0, agentScanExecutor_1.executeAction)(recoveryAction, ctx, startResp.runId, client, target);
|
|
735
|
+
}
|
|
736
|
+
qualityTracker.recordToolUse(recoveryAction.type);
|
|
737
|
+
investigationState.recordToolUse(recoveryAction.type);
|
|
738
|
+
trace.logToolCompleted(recoveryAction.type, recoveryObservation);
|
|
739
|
+
transcript.push({ action: recoveryAction, observation: `[FINISH GATE RECOVERY] ${recoveryObservation}` });
|
|
740
|
+
}
|
|
741
|
+
continue;
|
|
742
|
+
}
|
|
743
|
+
// Sync candidates-verified: when all candidates are ready for the
|
|
744
|
+
// Juror (supported or terminal) or there are none, the
|
|
745
|
+
// candidates-verified step is complete.
|
|
746
|
+
if (candidateStore.allReadyForJuror()) {
|
|
747
|
+
investigationState.markCandidatesVerified();
|
|
202
748
|
}
|
|
203
749
|
// Execute the action locally
|
|
204
750
|
let observation;
|
|
205
751
|
let wasBlocked = false;
|
|
752
|
+
let flowResult = null;
|
|
753
|
+
// Reset progress flag — only set to true when actual progress is made
|
|
754
|
+
meaningfulProgressSinceRecovery = false;
|
|
206
755
|
if (action.type === 'read_file') {
|
|
207
756
|
const normalizedPath = action.path.replace(/\\/g, '/').toLowerCase();
|
|
208
|
-
|
|
209
|
-
|
|
210
|
-
|
|
211
|
-
|
|
757
|
+
let totalLines;
|
|
758
|
+
try {
|
|
759
|
+
const fs = require('fs');
|
|
760
|
+
const abs = require('path').resolve(ctx.workspaceRoot, action.path);
|
|
761
|
+
const content = fs.readFileSync(abs, 'utf8');
|
|
762
|
+
totalLines = content.split('\n').length;
|
|
763
|
+
}
|
|
764
|
+
catch { /* best-effort */ }
|
|
765
|
+
const readValue = investigationState.classifyRead(action.path, action.startLine, action.endLine, totalLines);
|
|
766
|
+
const checklist = investigationState.formatChecklistForPrompt();
|
|
767
|
+
const fileMax = maxReadsForFile(action.path);
|
|
768
|
+
const count = investigationState.getReadCount(action.path);
|
|
769
|
+
if (readValue.classification === 'duplicate') {
|
|
770
|
+
qualityTracker.recordRead('duplicate', false);
|
|
771
|
+
investigationState.recordBlockedRead(action.path);
|
|
772
|
+
const nextHint = readValue.nextUnreadRange
|
|
773
|
+
? `\nNext unread range: lines ${readValue.nextUnreadRange.start}-${readValue.nextUnreadRange.end}. Use read_file with startLine=${readValue.nextUnreadRange.start} and endLine=${readValue.nextUnreadRange.end}.`
|
|
774
|
+
: '';
|
|
775
|
+
observation = `BLOCKED: Lines ${action.startLine || 'all'}-${action.endLine || 'all'} of "${action.path}" were already read. The content is in the transcript above.${nextHint}\n\n${checklist}`;
|
|
776
|
+
wasBlocked = true;
|
|
777
|
+
}
|
|
778
|
+
else if (readValue.classification === 'high-overlap') {
|
|
779
|
+
qualityTracker.recordRead('high-overlap', false);
|
|
780
|
+
investigationState.recordBlockedRead(action.path);
|
|
781
|
+
const nextHint = readValue.nextUnreadRange
|
|
782
|
+
? `\nNext unread range: lines ${readValue.nextUnreadRange.start}-${readValue.nextUnreadRange.end}. Use read_file with startLine=${readValue.nextUnreadRange.start} and endLine=${readValue.nextUnreadRange.end}.`
|
|
783
|
+
: '';
|
|
784
|
+
observation = `BLOCKED: Lines ${action.startLine || 1}-${action.endLine || totalLines || '?'} of "${action.path}" substantially overlap already-read ranges (${readValue.newLines} new lines of ${(action.endLine || 0) - (action.startLine || 1) + 1} requested). The content is in the transcript above. Read a DIFFERENT section or use trace_flow/check_guard/check_policy to analyze what you've already read.${nextHint}\n\n${checklist}`;
|
|
785
|
+
wasBlocked = true;
|
|
786
|
+
}
|
|
787
|
+
else if (readValue.classification === 'invalid') {
|
|
788
|
+
qualityTracker.recordRead('invalid', false);
|
|
789
|
+
investigationState.recordBlockedRead(action.path);
|
|
790
|
+
observation = `BLOCKED: Invalid range for "${action.path}" (startLine=${action.startLine}, endLine=${action.endLine}). The requested range is inverted or out of bounds. Use a valid line range.\n\n${checklist}`;
|
|
212
791
|
wasBlocked = true;
|
|
213
792
|
}
|
|
214
793
|
else {
|
|
215
|
-
|
|
216
|
-
|
|
217
|
-
|
|
218
|
-
|
|
219
|
-
|
|
220
|
-
|
|
221
|
-
|
|
222
|
-
|
|
794
|
+
qualityTracker.recordRead(readValue.classification, false);
|
|
795
|
+
const readResult = await (0, agentScanExecutor_1.executeReadFileAction)(action, ctx);
|
|
796
|
+
observation = readResult.observation;
|
|
797
|
+
qualityTracker.recordRead(readValue.classification, readResult.truncated);
|
|
798
|
+
if (readResult.totalLines > 0 && !readResult.truncated) {
|
|
799
|
+
investigationState.recordActualRead(action.path, readResult.actualStart, readResult.actualEnd, readResult.totalLines, readResult.truncated);
|
|
800
|
+
const rangeKey = `${normalizedPath}:${readResult.actualStart || 0}:${readResult.actualEnd || 0}`;
|
|
801
|
+
readFiles.add(rangeKey);
|
|
802
|
+
qualityTracker.recordToolUse('read_file');
|
|
803
|
+
investigationState.recordToolUse('read_file');
|
|
804
|
+
meaningfulProgressSinceLastExtension = true;
|
|
805
|
+
meaningfulProgressSinceRecovery = true;
|
|
806
|
+
if (count >= fileMax) {
|
|
807
|
+
observation += `\n\nNOTE: You have read "${action.path}" ${count + 1} times. Consider using search_code, trace_flow, check_guard, or check_policy to analyze the code you've read. If you have enough evidence, call finish to report your findings.\n\n${checklist}`;
|
|
808
|
+
}
|
|
809
|
+
}
|
|
810
|
+
else if (readResult.truncated) {
|
|
811
|
+
investigationState.recordActualRead(action.path, readResult.actualStart, readResult.actualEnd, readResult.totalLines, readResult.truncated);
|
|
812
|
+
qualityTracker.recordToolUse('read_file');
|
|
813
|
+
investigationState.recordToolUse('read_file');
|
|
814
|
+
meaningfulProgressSinceLastExtension = false;
|
|
815
|
+
meaningfulProgressSinceRecovery = false;
|
|
223
816
|
}
|
|
224
817
|
else {
|
|
225
|
-
|
|
226
|
-
readFiles.add(rangeKey);
|
|
227
|
-
observation = await (0, agentScanExecutor_1.executeAction)(action, ctx, startResp.runId, client, target);
|
|
818
|
+
investigationState.recordBlockedRead(action.path);
|
|
228
819
|
}
|
|
229
820
|
}
|
|
230
821
|
}
|
|
@@ -245,7 +836,73 @@ async function runAgentScan(ctx, target, options = {}) {
|
|
|
245
836
|
else if (action.type === 'check_policy') {
|
|
246
837
|
toolKey = `check_policy:${action.filePath || ''}`;
|
|
247
838
|
}
|
|
839
|
+
else if (action.type === 'call_graph') {
|
|
840
|
+
toolKey = `call_graph:${action.filePath || ''}:${action.functionName || ''}`;
|
|
841
|
+
}
|
|
842
|
+
else if (action.type === 'git_blame') {
|
|
843
|
+
toolKey = `git_blame:${action.filePath || ''}:${action.startLine || 0}:${action.endLine || 0}`;
|
|
844
|
+
}
|
|
845
|
+
else if (action.type === 'git_history') {
|
|
846
|
+
toolKey = `git_history:${action.filePath || ''}:${action.functionName || ''}`;
|
|
847
|
+
}
|
|
848
|
+
else if (action.type === 'git_diff') {
|
|
849
|
+
toolKey = `git_diff:${action.baseRef || ''}:${action.headRef || 'HEAD'}`;
|
|
850
|
+
}
|
|
851
|
+
else if (action.type === 'check_dependencies') {
|
|
852
|
+
toolKey = `check_dependencies`;
|
|
853
|
+
}
|
|
854
|
+
else if (action.type === 'read_config') {
|
|
855
|
+
toolKey = `read_config:${action.configKind || 'all'}`;
|
|
856
|
+
}
|
|
857
|
+
else if (action.type === 'find_definition') {
|
|
858
|
+
toolKey = `find_definition:${action.filePath || ''}:${action.symbol || ''}`;
|
|
859
|
+
}
|
|
860
|
+
else if (action.type === 'find_references') {
|
|
861
|
+
toolKey = `find_references:${action.filePath || ''}:${action.symbol || ''}`;
|
|
862
|
+
}
|
|
863
|
+
else if (action.type === 'find_tests') {
|
|
864
|
+
toolKey = `find_tests:${action.filePath || ''}:${action.symbol || ''}`;
|
|
865
|
+
}
|
|
866
|
+
else if (action.type === 'run_tests') {
|
|
867
|
+
const a = action;
|
|
868
|
+
if (a.mode === 'existing') {
|
|
869
|
+
toolKey = `run_tests:existing:${(a.testFiles || []).join(',')}:${a.testPattern || ''}:${a.packageManager || ''}`;
|
|
870
|
+
}
|
|
871
|
+
else {
|
|
872
|
+
const crypto = require('crypto');
|
|
873
|
+
const scriptHash = a.script ? crypto.createHash('sha256').update(a.script).digest('hex').substring(0, 16) : '';
|
|
874
|
+
toolKey = `run_tests:generated:${a.runner || ''}:${scriptHash}`;
|
|
875
|
+
}
|
|
876
|
+
}
|
|
248
877
|
if (toolKey) {
|
|
878
|
+
// Check for equivalent search intent (same terms, different
|
|
879
|
+
// order) before the raw toolKey dedup.
|
|
880
|
+
if (action.type === 'search_code') {
|
|
881
|
+
const pattern = action.pattern || '';
|
|
882
|
+
const normalized = (0, searchIntent_1.normalizeSearchPattern)(pattern);
|
|
883
|
+
let isEquivalent = false;
|
|
884
|
+
for (const prev of searchedPatterns) {
|
|
885
|
+
if ((0, searchIntent_1.isEquivalentSearchIntent)(prev, pattern)) {
|
|
886
|
+
isEquivalent = true;
|
|
887
|
+
break;
|
|
888
|
+
}
|
|
889
|
+
}
|
|
890
|
+
if (isEquivalent) {
|
|
891
|
+
equivalentSearchCount++;
|
|
892
|
+
if (equivalentSearchCount >= 2) {
|
|
893
|
+
observation = `BLOCKED: This search is equivalent to a previous search (same terms, different order). The results are in the transcript above. Use find_definition, find_references, call_graph, a targeted read_file with specific line numbers, or trace_flow_cross_file instead.`;
|
|
894
|
+
wasBlocked = true;
|
|
895
|
+
const checklist = investigationState.formatChecklistForPrompt();
|
|
896
|
+
observation += `\n\n${checklist}`;
|
|
897
|
+
// Skip execution — jump to the blocked handling
|
|
898
|
+
transcript.push({ action, observation });
|
|
899
|
+
trace.logToolBlocked(action.type, observation.slice(0, 200));
|
|
900
|
+
consecutiveBlockedReads++;
|
|
901
|
+
continue;
|
|
902
|
+
}
|
|
903
|
+
}
|
|
904
|
+
searchedPatterns.add(normalized);
|
|
905
|
+
}
|
|
249
906
|
const count = (toolCallCounts.get(toolKey) || 0) + 1;
|
|
250
907
|
toolCallCounts.set(toolKey, count);
|
|
251
908
|
if (count > MAX_SAME_TOOL_CALL) {
|
|
@@ -253,34 +910,374 @@ async function runAgentScan(ctx, target, options = {}) {
|
|
|
253
910
|
wasBlocked = true;
|
|
254
911
|
}
|
|
255
912
|
else {
|
|
256
|
-
|
|
913
|
+
if (action.type === 'trace_flow' || action.type === 'trace_flow_cross_file') {
|
|
914
|
+
const flowAction = await (0, agentScanExecutor_1.executeFlowAction)(action, ctx, startResp.runId, client, target);
|
|
915
|
+
observation = flowAction.observation;
|
|
916
|
+
flowResult = flowAction.flowResult;
|
|
917
|
+
}
|
|
918
|
+
else {
|
|
919
|
+
observation = await (0, agentScanExecutor_1.executeAction)(action, ctx, startResp.runId, client, target);
|
|
920
|
+
}
|
|
921
|
+
qualityTracker.recordToolUse(action.type);
|
|
922
|
+
investigationState.recordToolUse(action.type);
|
|
923
|
+
if (action.type === 'search_code') {
|
|
924
|
+
investigationState.recordSymbolSearch(action.pattern || '');
|
|
925
|
+
}
|
|
926
|
+
// First call with these args = meaningful progress.
|
|
927
|
+
// Repeated call (count > 1) does NOT reset recovery state.
|
|
928
|
+
const isFirstCall = count === 1;
|
|
929
|
+
meaningfulProgressSinceLastExtension = isFirstCall;
|
|
930
|
+
if (isFirstCall) {
|
|
931
|
+
meaningfulProgressSinceRecovery = true;
|
|
932
|
+
}
|
|
257
933
|
}
|
|
258
934
|
}
|
|
259
935
|
else {
|
|
260
|
-
|
|
936
|
+
if (action.type === 'trace_flow' || action.type === 'trace_flow_cross_file') {
|
|
937
|
+
const flowAction = await (0, agentScanExecutor_1.executeFlowAction)(action, ctx, startResp.runId, client, target);
|
|
938
|
+
observation = flowAction.observation;
|
|
939
|
+
flowResult = flowAction.flowResult;
|
|
940
|
+
}
|
|
941
|
+
else {
|
|
942
|
+
observation = await (0, agentScanExecutor_1.executeAction)(action, ctx, startResp.runId, client, target);
|
|
943
|
+
}
|
|
944
|
+
qualityTracker.recordToolUse(action.type);
|
|
945
|
+
investigationState.recordToolUse(action.type);
|
|
946
|
+
meaningfulProgressSinceLastExtension = true;
|
|
947
|
+
meaningfulProgressSinceRecovery = true;
|
|
261
948
|
}
|
|
262
949
|
}
|
|
263
|
-
// Track consecutive blocked reads —
|
|
264
|
-
//
|
|
950
|
+
// Track consecutive blocked reads — quality-first: don't
|
|
951
|
+
// force-finish while unread ranges remain and budget is available.
|
|
952
|
+
// Instead, force a deterministic recovery action.
|
|
265
953
|
if (wasBlocked) {
|
|
954
|
+
trace.logToolBlocked(action.type, observation.slice(0, 200));
|
|
266
955
|
consecutiveBlockedReads++;
|
|
267
|
-
|
|
268
|
-
|
|
269
|
-
|
|
270
|
-
|
|
271
|
-
|
|
272
|
-
|
|
273
|
-
|
|
274
|
-
|
|
275
|
-
|
|
276
|
-
|
|
277
|
-
|
|
956
|
+
scanState.recovery.consecutiveBlockedActions = consecutiveBlockedReads;
|
|
957
|
+
scanState.recovery.totalBlockedActions++;
|
|
958
|
+
scanState.recovery.lastBlockedAction = action.type;
|
|
959
|
+
scanState.recovery.meaningfulProgressSinceRecovery = false;
|
|
960
|
+
const recoveryLimit = agentScanProtocol_1.AGENT_SCAN_DEFAULTS.blockedReadRecoveryLimit;
|
|
961
|
+
if (consecutiveBlockedReads >= 2 && consecutiveBlockedReads < 3) {
|
|
962
|
+
const nextRange = target.fileContent
|
|
963
|
+
? investigationState.getPrioritizedUnreadRange(target.filePath, target.fileContent)
|
|
964
|
+
: investigationState.getNextUnreadRange(target.filePath);
|
|
965
|
+
if (nextRange) {
|
|
966
|
+
observation += `\n\nRECOVERY REQUIRED: You have been blocked ${consecutiveBlockedReads} time(s). Read the next unread range: lines ${nextRange.start}-${nextRange.end}, or use an analysis tool (trace_flow, check_guard, check_policy).`;
|
|
967
|
+
}
|
|
968
|
+
else {
|
|
969
|
+
const recommendedTool = investigationState.getRecommendedRecoveryAction();
|
|
970
|
+
if (recommendedTool) {
|
|
971
|
+
observation += `\n\nRECOVERY REQUIRED: You have been blocked ${consecutiveBlockedReads} time(s). No unread ranges remain. Your next action MUST be: ${recommendedTool}. If you have enough evidence, call finish.`;
|
|
972
|
+
}
|
|
973
|
+
}
|
|
974
|
+
}
|
|
975
|
+
if (consecutiveBlockedReads >= recoveryLimit) {
|
|
976
|
+
const nextRange = target.fileContent
|
|
977
|
+
? investigationState.getPrioritizedUnreadRange(target.filePath, target.fileContent)
|
|
978
|
+
: investigationState.getNextUnreadRange(target.filePath);
|
|
979
|
+
const hasBudget = budget.stepsRemaining > 0 && costSpentUsd < budget.costCapUsd;
|
|
980
|
+
const hasWallClock = Date.now() - startTime < wallClockMs;
|
|
981
|
+
if (nextRange && hasBudget && hasWallClock) {
|
|
982
|
+
console.warn(`[Agent Scan Loop] ${consecutiveBlockedReads} consecutive blocked reads — forcing deterministic recovery to unread range ${nextRange.start}-${nextRange.end}.`);
|
|
983
|
+
}
|
|
984
|
+
else {
|
|
985
|
+
console.warn(`[Agent Scan Loop] ${consecutiveBlockedReads} consecutive blocked reads — no recovery available, force-finishing.`);
|
|
986
|
+
transcript.push({ action, observation });
|
|
987
|
+
trace.logRunCompleted('completed');
|
|
988
|
+
const incompleteSteps = investigationState.getIncompleteSteps();
|
|
989
|
+
const autoGaps = incompleteSteps.map(step => ({
|
|
990
|
+
title: `Investigation step not completed: ${step}`,
|
|
991
|
+
detail: `The agent was force-finished after repeated blocked reads without completing this required investigation step: ${step}. The investigation was incomplete and vulnerabilities may have been missed.`,
|
|
992
|
+
file: target.filePath,
|
|
993
|
+
requiredEvidence: [`Complete the ${step} step before concluding no vulnerabilities exist`],
|
|
994
|
+
suggestedNextAction: step === 'config-inspection' ? 'read_config'
|
|
995
|
+
: step === 'policy-check' ? 'check_policy'
|
|
996
|
+
: step === 'cross-file-flow' ? 'trace_flow_cross_file'
|
|
997
|
+
: step === 'route-discovery' ? 'get_endpoints'
|
|
998
|
+
: step === 'auth-symbol-search' ? 'search_code'
|
|
999
|
+
: 'continue investigation',
|
|
1000
|
+
priority: 'high',
|
|
1001
|
+
}));
|
|
1002
|
+
const unresolvedTasks = investigationState.getUnresolvedTasks();
|
|
1003
|
+
const taskGaps = unresolvedTasks.map(task => ({
|
|
1004
|
+
title: `Architecture risk unresolved: ${task.claim}`,
|
|
1005
|
+
detail: `This architecture-risk investigation task was not resolved: ${task.claim}. Required evidence: ${task.requiredEvidence.join('; ')}.`,
|
|
1006
|
+
file: task.targetFiles[0] || target.filePath,
|
|
1007
|
+
requiredEvidence: task.requiredEvidence,
|
|
1008
|
+
suggestedNextAction: task.requiredTools[0] || 'continue investigation',
|
|
1009
|
+
priority: 'high',
|
|
1010
|
+
}));
|
|
1011
|
+
const uncoveredRanges = investigationState.getUncoveredRanges(target.filePath);
|
|
1012
|
+
const rangeGaps = uncoveredRanges.map(r => ({
|
|
1013
|
+
title: `Unread range: lines ${r.start}-${r.end} of ${target.filePath}`,
|
|
1014
|
+
detail: `This range was never read during the investigation. Vulnerabilities in this range were not checked.`,
|
|
1015
|
+
file: target.filePath,
|
|
1016
|
+
requiredEvidence: [`Read lines ${r.start}-${r.end} and analyze for vulnerabilities`],
|
|
1017
|
+
suggestedNextAction: 'read_file',
|
|
1018
|
+
priority: 'high',
|
|
1019
|
+
}));
|
|
1020
|
+
const allGaps = [...autoGaps, ...taskGaps, ...rangeGaps];
|
|
1021
|
+
(0, scanState_1.terminateScan)(scanState, 'blocked_read_recovery', `Agent stuck re-reading files. ${allGaps.length} coverage gaps.`);
|
|
1022
|
+
qualityTracker.recordForcedTermination('blocked_read_recovery');
|
|
1023
|
+
const covSummary = investigationState.getCoverageSummary(target.filePath);
|
|
1024
|
+
qualityTracker.recordCoverage(covSummary.totalLines, covSummary.coveredLines, covSummary.uncoveredRangeCount, covSummary.largestUncoveredRange);
|
|
1025
|
+
qualityTracker.recordBudget(stepsTaken, stepsGranted, extensionsGranted, costSpentUsd, budget.costCapUsd, wallClockMs, Date.now() - startTime);
|
|
1026
|
+
return {
|
|
1027
|
+
status: 'completed',
|
|
1028
|
+
findings: [],
|
|
1029
|
+
transcript,
|
|
1030
|
+
stepsUsed: stepsTaken,
|
|
1031
|
+
stepsGranted,
|
|
1032
|
+
extensionsGranted,
|
|
1033
|
+
costSpentUsd,
|
|
1034
|
+
terminationReason: 'blocked_read_recovery',
|
|
1035
|
+
summary: `Investigation cut short — agent was stuck re-reading files. ${allGaps.length} coverage gaps identified.`,
|
|
1036
|
+
investigationNotes: [],
|
|
1037
|
+
coverageGaps: allGaps,
|
|
1038
|
+
};
|
|
1039
|
+
}
|
|
278
1040
|
}
|
|
279
1041
|
}
|
|
280
1042
|
else {
|
|
281
|
-
|
|
1043
|
+
trace.logToolCompleted(action.type, observation);
|
|
1044
|
+
// Only reset the blocked counter when meaningful progress was
|
|
1045
|
+
// made. Repeated searches and truncated reads do NOT reset
|
|
1046
|
+
// recovery state — only new coverage, new symbols, or new
|
|
1047
|
+
// tool calls (first invocation with these args) do.
|
|
1048
|
+
if (meaningfulProgressSinceRecovery) {
|
|
1049
|
+
consecutiveBlockedReads = 0;
|
|
1050
|
+
scanState.recovery.consecutiveBlockedActions = 0;
|
|
1051
|
+
scanState.recovery.meaningfulProgressSinceRecovery = true;
|
|
1052
|
+
}
|
|
282
1053
|
}
|
|
1054
|
+
// The agent needs both the action it tried and the block/observation
|
|
1055
|
+
// message — the original {action, observation} pair carries both.
|
|
1056
|
+
// We don't need a separate system_event for blocked (unlike
|
|
1057
|
+
// critique, the blocked case is tied to an action the agent took).
|
|
283
1058
|
transcript.push({ action, observation });
|
|
1059
|
+
// Link evidence to candidates: when the agent runs a verification
|
|
1060
|
+
// action (trace_flow, check_guard, check_policy, trace_flow_cross_file)
|
|
1061
|
+
// on a file that matches a candidate's location, add evidence to
|
|
1062
|
+
// that candidate. This auto-transitions candidates through
|
|
1063
|
+
// discovered → investigating → supported as evidence accumulates.
|
|
1064
|
+
if (!wasBlocked && candidateStore.size() > 0) {
|
|
1065
|
+
const verificationActions = new Set([
|
|
1066
|
+
'trace_flow', 'trace_flow_cross_file', 'check_guard', 'check_policy',
|
|
1067
|
+
]);
|
|
1068
|
+
if (verificationActions.has(action.type)) {
|
|
1069
|
+
const actionFile = action.filePath || action.path || target.filePath;
|
|
1070
|
+
const actionFileNorm = String(actionFile).replace(/\\/g, '/').toLowerCase();
|
|
1071
|
+
for (const candidate of candidateStore.getAll()) {
|
|
1072
|
+
const matchesLocation = candidate.locations.some(loc => loc.filePath.replace(/\\/g, '/').toLowerCase() === actionFileNorm);
|
|
1073
|
+
if (matchesLocation) {
|
|
1074
|
+
const evidenceId = `${action.type}:${actionFileNorm}:${stepsTaken}`;
|
|
1075
|
+
const category = action.type === 'trace_flow' || action.type === 'trace_flow_cross_file' ? 'flow'
|
|
1076
|
+
: action.type === 'check_guard' ? 'guard'
|
|
1077
|
+
: action.type === 'check_policy' ? 'policy'
|
|
1078
|
+
: undefined;
|
|
1079
|
+
candidateStore.addEvidence(candidate.id, evidenceId, category);
|
|
1080
|
+
}
|
|
1081
|
+
}
|
|
1082
|
+
}
|
|
1083
|
+
if (action.type === 'read_file' && candidateStore.size() > 0) {
|
|
1084
|
+
const actionFileNorm = String(action.path || target.filePath).replace(/\\/g, '/').toLowerCase();
|
|
1085
|
+
for (const candidate of candidateStore.getAll()) {
|
|
1086
|
+
const matchesLocation = candidate.locations.some(loc => loc.filePath.replace(/\\/g, '/').toLowerCase() === actionFileNorm);
|
|
1087
|
+
if (matchesLocation && candidate.evidenceCategories?.includes('source') !== true) {
|
|
1088
|
+
candidateStore.addEvidence(candidate.id, `source:${actionFileNorm}:${stepsTaken}`, 'source');
|
|
1089
|
+
}
|
|
1090
|
+
}
|
|
1091
|
+
}
|
|
1092
|
+
}
|
|
1093
|
+
// Link evidence to work items: when the agent runs an action on
|
|
1094
|
+
// a file that matches a work item's target files, add evidence
|
|
1095
|
+
// and resolve the work item if it has enough evidence.
|
|
1096
|
+
if (!wasBlocked && workItemQueue.size() > 0) {
|
|
1097
|
+
const actionFile = action.filePath || action.path || target.filePath;
|
|
1098
|
+
const actionFileNorm = String(actionFile).replace(/\\/g, '/').toLowerCase();
|
|
1099
|
+
const evidenceKindMap = {
|
|
1100
|
+
read_file: 'source-range',
|
|
1101
|
+
search_code: 'symbol-reference',
|
|
1102
|
+
trace_flow: 'cross-file-flow',
|
|
1103
|
+
trace_flow_cross_file: 'cross-file-flow',
|
|
1104
|
+
check_guard: 'guard-result',
|
|
1105
|
+
check_policy: 'policy-result',
|
|
1106
|
+
get_endpoints: 'handler-inventory',
|
|
1107
|
+
list_imports: 'symbol-reference',
|
|
1108
|
+
find_definition: 'symbol-definition',
|
|
1109
|
+
find_references: 'symbol-reference',
|
|
1110
|
+
find_tests: 'test-location',
|
|
1111
|
+
run_tests: 'test-result',
|
|
1112
|
+
read_config: 'config-result',
|
|
1113
|
+
call_graph: 'cross-file-flow',
|
|
1114
|
+
};
|
|
1115
|
+
const evidenceKind = evidenceKindMap[action.type] || 'source-range';
|
|
1116
|
+
for (const item of workItemQueue.getExecutable()) {
|
|
1117
|
+
const matchesFile = item.targetFiles.length === 0 ||
|
|
1118
|
+
item.targetFiles.some(f => f.replace(/\\/g, '/').toLowerCase() === actionFileNorm);
|
|
1119
|
+
if (matchesFile) {
|
|
1120
|
+
const evidenceId = `${action.type}:${actionFileNorm}:${stepsTaken}`;
|
|
1121
|
+
workItemQueue.addEvidence(item.id, evidenceId);
|
|
1122
|
+
for (const req of item.requirements) {
|
|
1123
|
+
if (!req.acceptedKinds.includes(evidenceKind))
|
|
1124
|
+
continue;
|
|
1125
|
+
if (req.targetFiles && req.targetFiles.length > 0) {
|
|
1126
|
+
const reqMatches = req.targetFiles.some(f => f.replace(/\\/g, '/').toLowerCase() === actionFileNorm);
|
|
1127
|
+
if (!reqMatches)
|
|
1128
|
+
continue;
|
|
1129
|
+
}
|
|
1130
|
+
if (req.requiredTools && req.requiredTools.length > 0) {
|
|
1131
|
+
if (!req.requiredTools.includes(action.type))
|
|
1132
|
+
continue;
|
|
1133
|
+
}
|
|
1134
|
+
workItemQueue.addEvidenceForRequirement(item.id, req.id, evidenceId);
|
|
1135
|
+
}
|
|
1136
|
+
if (workItemQueue.isFullyResolved(item.id)) {
|
|
1137
|
+
workItemQueue.resolve(item.id);
|
|
1138
|
+
}
|
|
1139
|
+
}
|
|
1140
|
+
}
|
|
1141
|
+
}
|
|
1142
|
+
// Re-sync candidates-verified after evidence linking
|
|
1143
|
+
if (candidateStore.allReadyForJuror()) {
|
|
1144
|
+
investigationState.markCandidatesVerified();
|
|
1145
|
+
}
|
|
1146
|
+
// Evidence ledger: record evidence for each action type. This
|
|
1147
|
+
// populates the evidence ledger so the scheduler can suggest
|
|
1148
|
+
// actions for unsatisfied requirements and the finish gate can
|
|
1149
|
+
// verify all evidence requirements are met.
|
|
1150
|
+
if (!wasBlocked) {
|
|
1151
|
+
const actionFile = String(action.filePath || action.path || target.filePath);
|
|
1152
|
+
const evidenceKindMap = {
|
|
1153
|
+
read_file: 'source-range',
|
|
1154
|
+
search_code: 'symbol-reference',
|
|
1155
|
+
trace_flow: 'cross-file-flow',
|
|
1156
|
+
trace_flow_cross_file: 'cross-file-flow',
|
|
1157
|
+
check_guard: 'guard-result',
|
|
1158
|
+
check_policy: 'policy-result',
|
|
1159
|
+
get_endpoints: 'handler-inventory',
|
|
1160
|
+
list_imports: 'symbol-reference',
|
|
1161
|
+
find_definition: 'symbol-definition',
|
|
1162
|
+
find_references: 'symbol-reference',
|
|
1163
|
+
find_tests: 'test-location',
|
|
1164
|
+
run_tests: 'test-result',
|
|
1165
|
+
read_config: 'config-result',
|
|
1166
|
+
call_graph: 'cross-file-flow',
|
|
1167
|
+
};
|
|
1168
|
+
const kind = evidenceKindMap[action.type];
|
|
1169
|
+
if (kind) {
|
|
1170
|
+
evidenceLedger.recordEvidence({
|
|
1171
|
+
kind: kind,
|
|
1172
|
+
tool: action.type,
|
|
1173
|
+
filePath: actionFile,
|
|
1174
|
+
range: action.type === 'read_file'
|
|
1175
|
+
? { start: action.startLine || 1, end: action.endLine || 1 }
|
|
1176
|
+
: undefined,
|
|
1177
|
+
symbol: action.symbol || action.pattern || undefined,
|
|
1178
|
+
outcome: wasBlocked ? 'blocked' : 'positive',
|
|
1179
|
+
transcriptStep: stepsTaken,
|
|
1180
|
+
});
|
|
1181
|
+
}
|
|
1182
|
+
}
|
|
1183
|
+
// Flow verification: classify trace_flow/trace_flow_cross_file
|
|
1184
|
+
// results using structured data from the executor, not observation
|
|
1185
|
+
// text heuristics. This prevents the cross-file-flow checklist step
|
|
1186
|
+
// from completing until a flow result has been explicitly classified.
|
|
1187
|
+
if (action.type === 'trace_flow' || action.type === 'trace_flow_cross_file') {
|
|
1188
|
+
const flowFile = String(action.filePath || target.filePath);
|
|
1189
|
+
if (wasBlocked) {
|
|
1190
|
+
investigationState.recordFlowVerification(flowFile, action.type, 'blocked', 0, 'Action was blocked');
|
|
1191
|
+
}
|
|
1192
|
+
else if (flowResult) {
|
|
1193
|
+
investigationState.recordFlowVerification(flowFile, action.type, flowResult.status, flowResult.hops.length, flowResult.error || `${flowResult.hops.length} hop(s)`, {
|
|
1194
|
+
source: flowResult.source,
|
|
1195
|
+
sink: flowResult.sink,
|
|
1196
|
+
hops: flowResult.hops,
|
|
1197
|
+
truncated: flowResult.truncated,
|
|
1198
|
+
error: flowResult.error,
|
|
1199
|
+
});
|
|
1200
|
+
}
|
|
1201
|
+
else {
|
|
1202
|
+
investigationState.recordFlowVerification(flowFile, action.type, 'inconclusive', 0, 'No structured flow result available');
|
|
1203
|
+
}
|
|
1204
|
+
}
|
|
1205
|
+
// handlers and populate the handler inventory. Create handler
|
|
1206
|
+
// review work items for security-sensitive handlers.
|
|
1207
|
+
if (!wasBlocked && action.type === 'get_endpoints') {
|
|
1208
|
+
try {
|
|
1209
|
+
const endpoints = await (0, endpointDiscovery_1.discoverEndpoints)(ctx.workspaceRoot, action.glob);
|
|
1210
|
+
if (endpoints.length > 0) {
|
|
1211
|
+
handlerInventory.addFromEndpoints(endpoints, target.filePath);
|
|
1212
|
+
// Create work items for newly discovered handlers
|
|
1213
|
+
for (const handler of handlerInventory.getAll()) {
|
|
1214
|
+
if (!workItemQueue.all().some(w => w.title === `Review handler: ${handler.symbol || 'unknown'}`)) {
|
|
1215
|
+
workItemQueue.add((0, workItem_1.createHandlerReviewWorkItem)(handler.filePath, handler.symbol || 'unknown'));
|
|
1216
|
+
}
|
|
1217
|
+
}
|
|
1218
|
+
}
|
|
1219
|
+
else {
|
|
1220
|
+
// No handlers found — auto-complete all-handlers-reviewed
|
|
1221
|
+
investigationState.markAllHandlersReviewed();
|
|
1222
|
+
}
|
|
1223
|
+
}
|
|
1224
|
+
catch { /* best-effort */ }
|
|
1225
|
+
}
|
|
1226
|
+
// Auto-complete all-handlers-reviewed if handler inventory is
|
|
1227
|
+
// empty (no handlers to review for this file type)
|
|
1228
|
+
if (handlerInventory.size() === 0 && !investigationState.getCompletedSteps().includes('all-handlers-reviewed')) {
|
|
1229
|
+
investigationState.markAllHandlersReviewed();
|
|
1230
|
+
}
|
|
1231
|
+
// Handler review tracking: when the agent reads a file range that
|
|
1232
|
+
// covers a handler's range, mark that handler as reviewed.
|
|
1233
|
+
if (!wasBlocked && action.type === 'read_file' && handlerInventory.size() > 0) {
|
|
1234
|
+
const readStart = action.startLine || 1;
|
|
1235
|
+
const readEnd = action.endLine || 9999;
|
|
1236
|
+
const readFileNorm = String(action.path || target.filePath).replace(/\\/g, '/').toLowerCase();
|
|
1237
|
+
for (const handler of handlerInventory.getAll()) {
|
|
1238
|
+
if (!handler.reviewed) {
|
|
1239
|
+
const handlerFileNorm = handler.filePath.replace(/\\/g, '/').toLowerCase();
|
|
1240
|
+
if (handlerFileNorm === readFileNorm) {
|
|
1241
|
+
if (handler.range.start >= readStart && handler.range.end <= readEnd) {
|
|
1242
|
+
handlerInventory.markReviewed(handler.id);
|
|
1243
|
+
}
|
|
1244
|
+
}
|
|
1245
|
+
}
|
|
1246
|
+
}
|
|
1247
|
+
}
|
|
1248
|
+
// Deterministic recovery: on the third consecutive blocked read,
|
|
1249
|
+
// skip the API and directly execute a recovery action selected by
|
|
1250
|
+
// the MCP. This prevents wasting API calls on an agent that is
|
|
1251
|
+
// stuck re-reading files.
|
|
1252
|
+
if (wasBlocked && consecutiveBlockedReads >= 3) {
|
|
1253
|
+
const recoveryAction = selectDeterministicRecoveryAction(investigationState, target, ctx);
|
|
1254
|
+
if (recoveryAction) {
|
|
1255
|
+
stepsTaken++;
|
|
1256
|
+
budget.stepsRemaining--;
|
|
1257
|
+
trace.nextStep();
|
|
1258
|
+
trace.logActionSelected(recoveryAction.type, 'deterministic-recovery', false);
|
|
1259
|
+
if (options.onProgress) {
|
|
1260
|
+
options.onProgress(stepsTaken, stepsGranted, `Step ${stepsTaken}: deterministic recovery: ${describeAction(recoveryAction)}`);
|
|
1261
|
+
}
|
|
1262
|
+
let recoveryObservation;
|
|
1263
|
+
if (recoveryAction.type === 'read_file') {
|
|
1264
|
+
const readResult = await (0, agentScanExecutor_1.executeReadFileAction)(recoveryAction, ctx);
|
|
1265
|
+
recoveryObservation = readResult.observation;
|
|
1266
|
+
if (readResult.totalLines > 0) {
|
|
1267
|
+
investigationState.recordActualRead(recoveryAction.path, readResult.actualStart, readResult.actualEnd, readResult.totalLines, readResult.truncated);
|
|
1268
|
+
}
|
|
1269
|
+
}
|
|
1270
|
+
else {
|
|
1271
|
+
recoveryObservation = await (0, agentScanExecutor_1.executeAction)(recoveryAction, ctx, startResp.runId, client, target);
|
|
1272
|
+
}
|
|
1273
|
+
qualityTracker.recordToolUse(recoveryAction.type);
|
|
1274
|
+
investigationState.recordToolUse(recoveryAction.type);
|
|
1275
|
+
consecutiveBlockedReads = 0; // recovery resets the counter
|
|
1276
|
+
meaningfulProgressSinceRecovery = true; // recovery is meaningful progress
|
|
1277
|
+
trace.logToolCompleted(recoveryAction.type, recoveryObservation);
|
|
1278
|
+
transcript.push({ action: recoveryAction, observation: `[DETERMINISTIC RECOVERY] ${recoveryObservation}` });
|
|
1279
|
+
}
|
|
1280
|
+
}
|
|
284
1281
|
}
|
|
285
1282
|
}
|
|
286
1283
|
catch (err) {
|
|
@@ -289,11 +1286,162 @@ async function runAgentScan(ctx, target, options = {}) {
|
|
|
289
1286
|
findings: [],
|
|
290
1287
|
transcript: [],
|
|
291
1288
|
stepsUsed: stepsTaken,
|
|
1289
|
+
stepsGranted,
|
|
1290
|
+
extensionsGranted,
|
|
292
1291
|
costSpentUsd,
|
|
1292
|
+
terminationReason: 'api_error',
|
|
293
1293
|
error: err.message || String(err),
|
|
1294
|
+
investigationNotes: [],
|
|
1295
|
+
coverageGaps: [],
|
|
294
1296
|
};
|
|
295
1297
|
}
|
|
296
1298
|
}
|
|
1299
|
+
function isCoherentText(text) {
|
|
1300
|
+
if (!text || text.length < 5)
|
|
1301
|
+
return false;
|
|
1302
|
+
const asciiLetters = (text.match(/[a-zA-Z]/g) || []).length;
|
|
1303
|
+
const totalChars = text.length;
|
|
1304
|
+
if (totalChars > 0 && asciiLetters / totalChars < 0.3)
|
|
1305
|
+
return false;
|
|
1306
|
+
const controlChars = (text.match(/[\x00-\x08\x0B\x0C\x0E-\x1F\uFFFD]/g) || []).length;
|
|
1307
|
+
if (controlChars > 0)
|
|
1308
|
+
return false;
|
|
1309
|
+
const words = text.split(/\s+/).filter(w => w.length > 1);
|
|
1310
|
+
if (words.length < 3)
|
|
1311
|
+
return false;
|
|
1312
|
+
return true;
|
|
1313
|
+
}
|
|
1314
|
+
const TRANSCRIPT_CHAR_BUDGET = 120_000;
|
|
1315
|
+
const TRANSCRIPT_KEEP_RECENT = 12;
|
|
1316
|
+
const TRANSCRIPT_SUMMARY_LEN = 200;
|
|
1317
|
+
function estimateTranscriptSize(transcript) {
|
|
1318
|
+
let size = 0;
|
|
1319
|
+
for (const step of transcript) {
|
|
1320
|
+
size += (step.observation || '').length;
|
|
1321
|
+
size += JSON.stringify(step.action || {}).length;
|
|
1322
|
+
}
|
|
1323
|
+
return size;
|
|
1324
|
+
}
|
|
1325
|
+
function compactTranscript(transcript) {
|
|
1326
|
+
const totalSize = estimateTranscriptSize(transcript);
|
|
1327
|
+
if (totalSize <= TRANSCRIPT_CHAR_BUDGET || transcript.length <= TRANSCRIPT_KEEP_RECENT) {
|
|
1328
|
+
return transcript;
|
|
1329
|
+
}
|
|
1330
|
+
const cutoff = transcript.length - TRANSCRIPT_KEEP_RECENT;
|
|
1331
|
+
const compacted = [];
|
|
1332
|
+
for (let i = 0; i < transcript.length; i++) {
|
|
1333
|
+
if (i < cutoff) {
|
|
1334
|
+
const obs = transcript[i].observation || '';
|
|
1335
|
+
const actionType = transcript[i].action?.type || 'unknown';
|
|
1336
|
+
if (actionType === 'system_event' || actionType === 'finish') {
|
|
1337
|
+
compacted.push(transcript[i]);
|
|
1338
|
+
}
|
|
1339
|
+
else if (obs.length > TRANSCRIPT_SUMMARY_LEN) {
|
|
1340
|
+
compacted.push({
|
|
1341
|
+
action: transcript[i].action,
|
|
1342
|
+
observation: obs.slice(0, TRANSCRIPT_SUMMARY_LEN) + '\n...[compacted]',
|
|
1343
|
+
});
|
|
1344
|
+
}
|
|
1345
|
+
else {
|
|
1346
|
+
compacted.push(transcript[i]);
|
|
1347
|
+
}
|
|
1348
|
+
}
|
|
1349
|
+
else {
|
|
1350
|
+
compacted.push(transcript[i]);
|
|
1351
|
+
}
|
|
1352
|
+
}
|
|
1353
|
+
return compacted;
|
|
1354
|
+
}
|
|
1355
|
+
function compactTranscriptAggressive(transcript) {
|
|
1356
|
+
const keepRecent = 6;
|
|
1357
|
+
const summaryLen = 100;
|
|
1358
|
+
if (transcript.length <= keepRecent) {
|
|
1359
|
+
return transcript;
|
|
1360
|
+
}
|
|
1361
|
+
const cutoff = transcript.length - keepRecent;
|
|
1362
|
+
const compacted = [];
|
|
1363
|
+
for (let i = 0; i < transcript.length; i++) {
|
|
1364
|
+
if (i < cutoff) {
|
|
1365
|
+
const obs = transcript[i].observation || '';
|
|
1366
|
+
const actionType = transcript[i].action?.type || 'unknown';
|
|
1367
|
+
if (actionType === 'system_event' || actionType === 'finish') {
|
|
1368
|
+
compacted.push(transcript[i]);
|
|
1369
|
+
}
|
|
1370
|
+
else if (obs.length > summaryLen) {
|
|
1371
|
+
compacted.push({
|
|
1372
|
+
action: transcript[i].action,
|
|
1373
|
+
observation: obs.slice(0, summaryLen) + '\n...[compacted]',
|
|
1374
|
+
});
|
|
1375
|
+
}
|
|
1376
|
+
else {
|
|
1377
|
+
compacted.push(transcript[i]);
|
|
1378
|
+
}
|
|
1379
|
+
}
|
|
1380
|
+
else {
|
|
1381
|
+
compacted.push(transcript[i]);
|
|
1382
|
+
}
|
|
1383
|
+
}
|
|
1384
|
+
return compacted;
|
|
1385
|
+
}
|
|
1386
|
+
const VALID_SEVERITIES = new Set(['critical', 'high', 'medium', 'low']);
|
|
1387
|
+
function sanitizeFindingText(text) {
|
|
1388
|
+
if (!text)
|
|
1389
|
+
return text;
|
|
1390
|
+
return text.replace(/[\x00-\x08\x0B\x0C\x0E-\x1F\uFFFD]/g, '').trim();
|
|
1391
|
+
}
|
|
1392
|
+
function validateFinding(f, index) {
|
|
1393
|
+
const issues = [];
|
|
1394
|
+
const sanitized = { ...f };
|
|
1395
|
+
if (typeof sanitized.line !== 'number' || sanitized.line < 1 || !Number.isFinite(sanitized.line)) {
|
|
1396
|
+
issues.push(`finding[${index}]: invalid line "${sanitized.line}" — defaulting to 0`);
|
|
1397
|
+
sanitized.line = 0;
|
|
1398
|
+
}
|
|
1399
|
+
if (typeof sanitized.type !== 'string' || sanitized.type.trim().length === 0) {
|
|
1400
|
+
issues.push(`finding[${index}]: missing or empty type — defaulting to "unknown"`);
|
|
1401
|
+
sanitized.type = 'unknown';
|
|
1402
|
+
}
|
|
1403
|
+
if (typeof sanitized.severity !== 'string' || !VALID_SEVERITIES.has(sanitized.severity)) {
|
|
1404
|
+
issues.push(`finding[${index}]: invalid severity "${sanitized.severity}" — defaulting to "low"`);
|
|
1405
|
+
sanitized.severity = 'low';
|
|
1406
|
+
}
|
|
1407
|
+
if (typeof sanitized.confidence !== 'number' || !Number.isFinite(sanitized.confidence)) {
|
|
1408
|
+
issues.push(`finding[${index}]: invalid confidence — defaulting to 0.3`);
|
|
1409
|
+
sanitized.confidence = 0.3;
|
|
1410
|
+
}
|
|
1411
|
+
else {
|
|
1412
|
+
sanitized.confidence = Math.max(0, Math.min(1, sanitized.confidence));
|
|
1413
|
+
}
|
|
1414
|
+
if (sanitized.lineEnd !== undefined && (typeof sanitized.lineEnd !== 'number' || sanitized.lineEnd < sanitized.line)) {
|
|
1415
|
+
delete sanitized.lineEnd;
|
|
1416
|
+
}
|
|
1417
|
+
sanitized.why = sanitizeFindingText(sanitized.why || '');
|
|
1418
|
+
if (!isCoherentText(sanitized.why)) {
|
|
1419
|
+
issues.push(`finding[${index}]: incoherent "why" — downgrading severity`);
|
|
1420
|
+
sanitized.why = `[Quality warning: model produced incoherent explanation] ${sanitized.why || '(empty)'}`;
|
|
1421
|
+
if (sanitized.severity === 'high' || sanitized.severity === 'critical') {
|
|
1422
|
+
sanitized.severity = 'medium';
|
|
1423
|
+
}
|
|
1424
|
+
sanitized.confidence = Math.min(sanitized.confidence, 0.3);
|
|
1425
|
+
}
|
|
1426
|
+
sanitized.evidence = sanitizeFindingText(sanitized.evidence || '');
|
|
1427
|
+
return { finding: sanitized, issues };
|
|
1428
|
+
}
|
|
1429
|
+
function sanitizeFindings(findings) {
|
|
1430
|
+
if (!Array.isArray(findings))
|
|
1431
|
+
return [];
|
|
1432
|
+
const seen = new Set();
|
|
1433
|
+
const result = [];
|
|
1434
|
+
for (let i = 0; i < findings.length; i++) {
|
|
1435
|
+
const { finding, issues } = validateFinding(findings[i], i);
|
|
1436
|
+
const dedupKey = `${finding.type}:${finding.line}`;
|
|
1437
|
+
if (seen.has(dedupKey)) {
|
|
1438
|
+
continue;
|
|
1439
|
+
}
|
|
1440
|
+
seen.add(dedupKey);
|
|
1441
|
+
result.push(finding);
|
|
1442
|
+
}
|
|
1443
|
+
return result;
|
|
1444
|
+
}
|
|
297
1445
|
function describeAction(action) {
|
|
298
1446
|
switch (action.type) {
|
|
299
1447
|
case 'read_file': return `read_file(${action.path})`;
|
|
@@ -305,7 +1453,63 @@ function describeAction(action) {
|
|
|
305
1453
|
case 'get_endpoints': return `get_endpoints(${action.glob || 'all'})`;
|
|
306
1454
|
case 'list_imports': return `list_imports(${action.filePath})`;
|
|
307
1455
|
case 'list_files': return `list_files(${action.path || 'root'})`;
|
|
1456
|
+
case 'call_graph': return `call_graph(${action.filePath})`;
|
|
1457
|
+
case 'git_blame': return `git_blame(${action.filePath})`;
|
|
1458
|
+
case 'git_history': return `git_history(${action.filePath || 'repo'})`;
|
|
1459
|
+
case 'git_diff': return `git_diff(${action.baseRef}..${action.headRef || 'HEAD'})`;
|
|
1460
|
+
case 'check_dependencies': return 'check_dependencies';
|
|
1461
|
+
case 'read_config': return `read_config(${action.configKind || 'all'})`;
|
|
1462
|
+
case 'find_definition': return `find_definition(${action.symbol})`;
|
|
1463
|
+
case 'find_references': return `find_references(${action.symbol})`;
|
|
1464
|
+
case 'find_tests': return `find_tests(${action.filePath}${action.symbol ? ':' + action.symbol : ''})`;
|
|
1465
|
+
case 'run_tests': return `run_tests(${action.mode}${action.testFiles?.length ? ':' + action.testFiles.length + ' files' : ''})`;
|
|
308
1466
|
case 'finish': return 'finish';
|
|
1467
|
+
case 'system_event': return `system_event(${action.eventType})`;
|
|
1468
|
+
}
|
|
1469
|
+
}
|
|
1470
|
+
/**
|
|
1471
|
+
* Select a deterministic recovery action based on investigation state.
|
|
1472
|
+
*
|
|
1473
|
+
* Priority:
|
|
1474
|
+
* 1. Unread range exists on the target file → read_file(next unread range)
|
|
1475
|
+
* 2. route-discovery missing → get_endpoints
|
|
1476
|
+
* 3. policy-check missing → check_policy
|
|
1477
|
+
* 4. auth-symbol-search missing → search_code
|
|
1478
|
+
* 5. cross-file-flow missing → trace_flow_cross_file
|
|
1479
|
+
* 6. config-inspection missing → read_config
|
|
1480
|
+
* 7. tests-found missing → find_tests
|
|
1481
|
+
* 8. otherwise → null (force-finish will handle it)
|
|
1482
|
+
*/
|
|
1483
|
+
function selectDeterministicRecoveryAction(state, target, ctx) {
|
|
1484
|
+
// Priority 1: unread range on the target file
|
|
1485
|
+
const nextRange = state.getNextUnreadRange(target.filePath);
|
|
1486
|
+
if (nextRange) {
|
|
1487
|
+
return {
|
|
1488
|
+
type: 'read_file',
|
|
1489
|
+
path: target.filePath,
|
|
1490
|
+
startLine: nextRange.start,
|
|
1491
|
+
endLine: nextRange.end,
|
|
1492
|
+
rationale: 'Deterministic recovery: reading next unread range',
|
|
1493
|
+
};
|
|
1494
|
+
}
|
|
1495
|
+
// Priority 2-7: incomplete investigation steps
|
|
1496
|
+
const incomplete = state.getIncompleteSteps();
|
|
1497
|
+
for (const step of incomplete) {
|
|
1498
|
+
switch (step) {
|
|
1499
|
+
case 'route-discovery':
|
|
1500
|
+
return { type: 'get_endpoints', rationale: 'Deterministic recovery: route discovery' };
|
|
1501
|
+
case 'policy-check':
|
|
1502
|
+
return { type: 'check_policy', filePath: target.filePath, rationale: 'Deterministic recovery: policy check' };
|
|
1503
|
+
case 'auth-symbol-search':
|
|
1504
|
+
return { type: 'search_code', pattern: 'auth|require|guard|permission|owner', rationale: 'Deterministic recovery: auth symbol search' };
|
|
1505
|
+
case 'cross-file-flow':
|
|
1506
|
+
return { type: 'trace_flow_cross_file', filePath: target.filePath, rationale: 'Deterministic recovery: cross-file flow' };
|
|
1507
|
+
case 'config-inspection':
|
|
1508
|
+
return { type: 'read_config', configKind: 'all', rationale: 'Deterministic recovery: config inspection' };
|
|
1509
|
+
case 'tests-found':
|
|
1510
|
+
return { type: 'find_tests', filePath: target.filePath, rationale: 'Deterministic recovery: find tests' };
|
|
1511
|
+
}
|
|
309
1512
|
}
|
|
1513
|
+
return null;
|
|
310
1514
|
}
|
|
311
1515
|
//# sourceMappingURL=agentScanLoop.js.map
|