@clear-capabilities/agentic-security-scanner 0.128.1 → 0.132.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +223 -0
- package/bin/agentic-security.js +52 -2
- package/dist/11.index.js +2 -2
- package/dist/113.index.js +498 -7
- package/dist/178.index.js +1 -1
- package/dist/207.index.js +220 -0
- package/dist/238.index.js +218 -0
- package/dist/259.index.js +975 -0
- package/dist/384.index.js +1 -1
- package/dist/415.index.js +1 -1
- package/dist/435.index.js +4 -4
- package/dist/526.index.js +844 -0
- package/dist/637.index.js +1 -1
- package/dist/830.index.js +1 -1
- package/dist/agentic-security.mjs +106 -194
- package/dist/agentic-security.mjs.sha256 +1 -1
- package/package.json +33 -17
- package/src/dataflow/CLAUDE.md +4 -1
- package/src/dataflow/async-sequencing.js +8 -3
- package/src/dataflow/catalog.js +278 -11
- package/src/dataflow/cross-repo.js +1 -1
- package/src/dataflow/cross-service-taint.js +1 -1
- package/src/dataflow/engine.js +182 -61
- package/src/dataflow/ifds.js +10 -5
- package/src/dataflow/index.js +15 -3
- package/src/dataflow/points-to.js +8 -2
- package/src/dataflow/proof-gate.js +7 -0
- package/src/dataflow/sanitizer-gate.js +89 -0
- package/src/dataflow/tabulation.js +14 -3
- package/src/engine.js +170 -7
- package/src/integrations/index.js +1 -1
- package/src/ir/CLAUDE.md +49 -4
- package/src/ir/call-sites.js +66 -0
- package/src/ir/callgraph.js +174 -7
- package/src/ir/class-hierarchy.js +22 -2
- package/src/ir/index.js +138 -51
- package/src/ir/ir-stats.js +126 -0
- package/src/ir/parser-cpp.js +829 -0
- package/src/ir/parser-cs.js +4 -1
- package/src/ir/parser-go.js +4 -1
- package/src/ir/parser-js.js +13 -1
- package/src/ir/parser-kt.js +4 -1
- package/src/ir/parser-php.js +10 -3
- package/src/ir/parser-py-cst.js +62 -10
- package/src/ir/tree-sitter-loader.js +13 -1
- package/src/llm-validator/index.js +9 -2
- package/src/llm-validator/redact.js +157 -0
- package/src/mcp/tools.js +2 -2
- package/src/posture/CLAUDE.md +193 -1
- package/src/posture/accuracy-scorecard.js +317 -0
- package/src/posture/api-contract.js +1 -1
- package/src/posture/attestation.js +202 -0
- package/src/posture/auditor-walkthrough.js +12 -3
- package/src/posture/compliance-policy.js +1 -1
- package/src/posture/corpus-enroll.js +303 -0
- package/src/posture/corpus-match.js +52 -0
- package/src/posture/cross-lang-openapi.js +1 -1
- package/src/posture/custom-rules.js +3 -3
- package/src/posture/execution-proof.js +92 -0
- package/src/posture/exploitability-probability.js +1 -1
- package/src/posture/falsification.js +45 -1
- package/src/posture/fix-metrics.js +197 -0
- package/src/posture/fix-verify.js +129 -2
- package/src/posture/license-policy.js +1 -1
- package/src/posture/profile.js +1 -1
- package/src/posture/proof-tier.js +33 -0
- package/src/posture/relevance.js +379 -0
- package/src/posture/root-cause-sweep.js +0 -0
- package/src/posture/rule-overrides.js +1 -1
- package/src/posture/sca-policy.js +1 -1
- package/src/posture/scan-checkpoint.js +277 -0
- package/src/posture/suppressions.js +1 -1
- package/src/posture/test-runner.js +147 -0
- package/src/posture/verification-separation.js +131 -0
- package/src/report/index.js +11 -0
- package/src/runScan.js +5 -7
- package/src/sandbox/CLAUDE.md +340 -0
- package/src/sandbox/backend-disabled.js +14 -0
- package/src/sandbox/backend-namespace.js +335 -0
- package/src/sandbox/backend-userspace.js +83 -0
- package/src/sandbox/capabilities.js +181 -0
- package/src/sandbox/index.js +30 -0
- package/src/sandbox/limits.js +63 -0
- package/src/sandbox/result.js +104 -0
- package/src/sca/dep-confusion.js +1 -1
- package/src/util/glob.js +173 -0
- package/src/util/yaml.js +24 -0
package/dist/113.index.js
CHANGED
|
@@ -1,7 +1,220 @@
|
|
|
1
1
|
export const id = 113;
|
|
2
|
-
export const ids = [113,
|
|
2
|
+
export const ids = [113,238,526];
|
|
3
3
|
export const modules = {
|
|
4
4
|
|
|
5
|
+
/***/ 2238:
|
|
6
|
+
/***/ ((__unused_webpack___webpack_module__, __webpack_exports__, __webpack_require__) => {
|
|
7
|
+
|
|
8
|
+
/* harmony export */ __webpack_require__.d(__webpack_exports__, {
|
|
9
|
+
/* harmony export */ I3: () => (/* binding */ recordFixAttempt),
|
|
10
|
+
/* harmony export */ fixDurationReport: () => (/* binding */ fixDurationReport),
|
|
11
|
+
/* harmony export */ renderFixDurationSummary: () => (/* binding */ renderFixDurationSummary)
|
|
12
|
+
/* harmony export */ });
|
|
13
|
+
/* unused harmony exports FIX_STAGES, loadFixAttempts, bucketOf, summarizeFixDurations, _internals */
|
|
14
|
+
/* harmony import */ var node_fs__WEBPACK_IMPORTED_MODULE_0__ = __webpack_require__(3024);
|
|
15
|
+
/* harmony import */ var node_path__WEBPACK_IMPORTED_MODULE_1__ = __webpack_require__(6760);
|
|
16
|
+
/* harmony import */ var _state_dir_js__WEBPACK_IMPORTED_MODULE_2__ = __webpack_require__(1174);
|
|
17
|
+
// Time-to-validated-fix (R5, the reporting half).
|
|
18
|
+
//
|
|
19
|
+
// `verifyFix` already RUNS the stages and `test-runner.js` already times the
|
|
20
|
+
// slowest one. What did not exist was anything durable to read afterwards, so
|
|
21
|
+
// "how long does a fix actually take to validate" had no answer from real runs
|
|
22
|
+
// — only an estimate (`time-to-fix.js` guesses engineering hours from family
|
|
23
|
+
// and patch shape, before anything runs). This module is the opposite: it
|
|
24
|
+
// records what the pipeline observed and reports the distribution.
|
|
25
|
+
//
|
|
26
|
+
// THE HONESTY RULES, which are most of why this file is longer than a mean:
|
|
27
|
+
//
|
|
28
|
+
// 1. A failed attempt is NOT a data point about how long a fix takes. Fixes
|
|
29
|
+
// that fail verification fail fast (a re-scan that still sees the finding
|
|
30
|
+
// never reaches the test suite), so blending them into one average makes
|
|
31
|
+
// the pipeline look faster the worse it performs. Validated and failed
|
|
32
|
+
// attempts are summarised separately and never merged.
|
|
33
|
+
//
|
|
34
|
+
// 2. "Tests skipped" is not "tests passed". A project with no detectable
|
|
35
|
+
// suite can reach `ok:true` having run only the re-scan and the linter.
|
|
36
|
+
// That is a weaker claim than a fix whose suite executed, and it is also
|
|
37
|
+
// much faster, so counting the two together would quietly deflate the
|
|
38
|
+
// headline. They get their own bucket: `validated` means the suite ran and
|
|
39
|
+
// passed, `validatedWithoutTests` means there was no suite to run.
|
|
40
|
+
//
|
|
41
|
+
// 3. Every figure carries its `n`, and a percentile computed from too few
|
|
42
|
+
// samples is labelled unreliable rather than omitted or silently reported.
|
|
43
|
+
// Same precedent as the accuracy scorecard's `{n, d}` rates: a number
|
|
44
|
+
// without its denominator is not a measurement.
|
|
45
|
+
//
|
|
46
|
+
// Storage is append-only JSONL at `<scanRoot>/.agentic-security/fix-metrics.jsonl`,
|
|
47
|
+
// one record per verification attempt. Nothing here throws (posture
|
|
48
|
+
// convention) — an unwritable or corrupt log degrades to "no metrics", never
|
|
49
|
+
// to a failed verification.
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
const STATE_DIR = '.agentic-security';
|
|
56
|
+
const LOG_FILE = 'fix-metrics.jsonl';
|
|
57
|
+
|
|
58
|
+
// Below this many samples a percentile is an artifact of the sample, not a
|
|
59
|
+
// property of the pipeline. Reported anyway (hiding it invites re-deriving it
|
|
60
|
+
// wrong downstream) but flagged, so a caller cannot quote it as settled.
|
|
61
|
+
const RELIABLE_N = 10;
|
|
62
|
+
|
|
63
|
+
// The stages verifyFix runs, in execution order. Kept here so the recorder and
|
|
64
|
+
// the summariser cannot drift apart on stage naming.
|
|
65
|
+
const FIX_STAGES = Object.freeze(['rescan', 'lint', 'tests', 'honesty', 'poc']);
|
|
66
|
+
|
|
67
|
+
function _logPath(scanRoot) {
|
|
68
|
+
return node_path__WEBPACK_IMPORTED_MODULE_1__.join(scanRoot, STATE_DIR, LOG_FILE);
|
|
69
|
+
}
|
|
70
|
+
|
|
71
|
+
/**
|
|
72
|
+
* Append one verification attempt. Best-effort and silent on failure: metrics
|
|
73
|
+
* must never be able to fail a fix that otherwise verified.
|
|
74
|
+
*
|
|
75
|
+
* @returns {boolean} whether the record was written (for tests, not callers).
|
|
76
|
+
*/
|
|
77
|
+
function recordFixAttempt(scanRoot, record) {
|
|
78
|
+
if (!scanRoot || !record || typeof record !== 'object') return false;
|
|
79
|
+
try {
|
|
80
|
+
const dir = node_path__WEBPACK_IMPORTED_MODULE_1__.join(scanRoot, STATE_DIR);
|
|
81
|
+
if (!(0,_state_dir_js__WEBPACK_IMPORTED_MODULE_2__.isSafeStateDir)(dir)) return false;
|
|
82
|
+
node_fs__WEBPACK_IMPORTED_MODULE_0__.mkdirSync(dir, { recursive: true });
|
|
83
|
+
// One writeSync of one newline-terminated line: a concurrent reader sees
|
|
84
|
+
// whole records or nothing, and a torn tail is dropped on read.
|
|
85
|
+
node_fs__WEBPACK_IMPORTED_MODULE_0__.appendFileSync(_logPath(scanRoot), JSON.stringify(record) + '\n', 'utf8');
|
|
86
|
+
return true;
|
|
87
|
+
} catch { return false; }
|
|
88
|
+
}
|
|
89
|
+
|
|
90
|
+
/**
|
|
91
|
+
* Read every well-formed attempt. A line that does not parse is skipped, not
|
|
92
|
+
* fatal — the last line of an interrupted write is the expected case.
|
|
93
|
+
*/
|
|
94
|
+
function loadFixAttempts(scanRoot) {
|
|
95
|
+
try {
|
|
96
|
+
const raw = node_fs__WEBPACK_IMPORTED_MODULE_0__.readFileSync(_logPath(scanRoot), 'utf8');
|
|
97
|
+
const out = [];
|
|
98
|
+
for (const line of raw.split('\n')) {
|
|
99
|
+
if (!line.trim()) continue;
|
|
100
|
+
try {
|
|
101
|
+
const rec = JSON.parse(line);
|
|
102
|
+
if (rec && typeof rec === 'object' && typeof rec.totalMs === 'number') out.push(rec);
|
|
103
|
+
} catch { /* torn or hand-edited line — drop it, keep the rest */ }
|
|
104
|
+
}
|
|
105
|
+
return out;
|
|
106
|
+
} catch { return []; }
|
|
107
|
+
}
|
|
108
|
+
|
|
109
|
+
// Nearest-rank percentile over an ascending array. Nearest-rank rather than
|
|
110
|
+
// interpolated because these are observed durations, and an interpolated p50
|
|
111
|
+
// reports a duration that no run actually took.
|
|
112
|
+
function _pct(sorted, p) {
|
|
113
|
+
if (!sorted.length) return null;
|
|
114
|
+
const rank = Math.ceil((p / 100) * sorted.length);
|
|
115
|
+
return sorted[Math.min(sorted.length - 1, Math.max(0, rank - 1))];
|
|
116
|
+
}
|
|
117
|
+
|
|
118
|
+
function _dist(values) {
|
|
119
|
+
const v = values.filter(x => typeof x === 'number' && Number.isFinite(x) && x >= 0).sort((a, b) => a - b);
|
|
120
|
+
if (!v.length) return { n: 0, minMs: null, p50Ms: null, p90Ms: null, maxMs: null, meanMs: null, reliable: false };
|
|
121
|
+
const sum = v.reduce((a, b) => a + b, 0);
|
|
122
|
+
return {
|
|
123
|
+
n: v.length,
|
|
124
|
+
minMs: v[0],
|
|
125
|
+
p50Ms: _pct(v, 50),
|
|
126
|
+
p90Ms: _pct(v, 90),
|
|
127
|
+
maxMs: v[v.length - 1],
|
|
128
|
+
meanMs: Math.round(sum / v.length),
|
|
129
|
+
// Says whether the percentiles above may be quoted, not whether the count
|
|
130
|
+
// is real. n and min/max/mean are exact at any sample size.
|
|
131
|
+
reliable: v.length >= RELIABLE_N,
|
|
132
|
+
};
|
|
133
|
+
}
|
|
134
|
+
|
|
135
|
+
// Which bucket an attempt belongs to. Deliberately total: every attempt lands
|
|
136
|
+
// in exactly one, so the bucket counts always sum to the attempt count and a
|
|
137
|
+
// mis-shaped record cannot silently vanish from the denominator.
|
|
138
|
+
function bucketOf(a) {
|
|
139
|
+
if (!a?.ok) return 'failed';
|
|
140
|
+
return a.testsRan ? 'validated' : 'validatedWithoutTests';
|
|
141
|
+
}
|
|
142
|
+
|
|
143
|
+
/**
|
|
144
|
+
* Summarise a set of attempts into the reported distribution.
|
|
145
|
+
*
|
|
146
|
+
* `validated` is the headline: attempts that verified AND whose test suite
|
|
147
|
+
* actually ran and passed. The other two buckets exist so that headline cannot
|
|
148
|
+
* be inflated by counting weaker or faster outcomes inside it.
|
|
149
|
+
*/
|
|
150
|
+
function summarizeFixDurations(attempts) {
|
|
151
|
+
const all = Array.isArray(attempts) ? attempts : [];
|
|
152
|
+
const buckets = { validated: [], validatedWithoutTests: [], failed: [] };
|
|
153
|
+
for (const a of all) buckets[bucketOf(a)].push(a);
|
|
154
|
+
|
|
155
|
+
const byStage = {};
|
|
156
|
+
for (const stage of FIX_STAGES) {
|
|
157
|
+
// Per-stage timings come from validated runs only. A stage's duration in a
|
|
158
|
+
// failed run is truncated by the failure (the pipeline stops), so mixing
|
|
159
|
+
// them in would understate every stage after the first failure point.
|
|
160
|
+
byStage[stage] = _dist(buckets.validated.map(a => a?.stages?.[stage]));
|
|
161
|
+
}
|
|
162
|
+
|
|
163
|
+
return {
|
|
164
|
+
attempts: all.length,
|
|
165
|
+
counts: {
|
|
166
|
+
validated: buckets.validated.length,
|
|
167
|
+
validatedWithoutTests: buckets.validatedWithoutTests.length,
|
|
168
|
+
failed: buckets.failed.length,
|
|
169
|
+
},
|
|
170
|
+
timeToValidatedFix: _dist(buckets.validated.map(a => a.totalMs)),
|
|
171
|
+
timeToValidatedFixWithoutTests: _dist(buckets.validatedWithoutTests.map(a => a.totalMs)),
|
|
172
|
+
timeToFailure: _dist(buckets.failed.map(a => a.totalMs)),
|
|
173
|
+
byStage,
|
|
174
|
+
reliableAtOrAbove: RELIABLE_N,
|
|
175
|
+
};
|
|
176
|
+
}
|
|
177
|
+
|
|
178
|
+
/** Read + summarise in one step. */
|
|
179
|
+
function fixDurationReport(scanRoot) {
|
|
180
|
+
return summarizeFixDurations(loadFixAttempts(scanRoot));
|
|
181
|
+
}
|
|
182
|
+
|
|
183
|
+
function _ms(v) {
|
|
184
|
+
if (v == null) return '—';
|
|
185
|
+
return v >= 1000 ? `${(v / 1000).toFixed(1)}s` : `${v}ms`;
|
|
186
|
+
}
|
|
187
|
+
|
|
188
|
+
/**
|
|
189
|
+
* One-paragraph human summary. Returns null when there is nothing measured —
|
|
190
|
+
* callers print nothing rather than printing an empty table.
|
|
191
|
+
*/
|
|
192
|
+
function renderFixDurationSummary(sum) {
|
|
193
|
+
if (!sum || !sum.attempts) return null;
|
|
194
|
+
const d = sum.timeToValidatedFix;
|
|
195
|
+
const parts = [];
|
|
196
|
+
if (d.n) {
|
|
197
|
+
parts.push(
|
|
198
|
+
`time-to-validated-fix: median ${_ms(d.p50Ms)}, p90 ${_ms(d.p90Ms)} `
|
|
199
|
+
+ `(n=${d.n}${d.reliable ? '' : `, below ${sum.reliableAtOrAbove} — percentiles not yet reliable`})`,
|
|
200
|
+
);
|
|
201
|
+
} else {
|
|
202
|
+
parts.push('time-to-validated-fix: no fix has both verified and had its test suite run yet');
|
|
203
|
+
}
|
|
204
|
+
if (sum.counts.validatedWithoutTests) {
|
|
205
|
+
parts.push(`${sum.counts.validatedWithoutTests} verified with no detectable test suite (excluded from the median above)`);
|
|
206
|
+
}
|
|
207
|
+
if (sum.counts.failed) {
|
|
208
|
+
parts.push(`${sum.counts.failed} failed verification, median ${_ms(sum.timeToFailure.p50Ms)} (counted separately)`);
|
|
209
|
+
}
|
|
210
|
+
return parts.join('; ') + '.';
|
|
211
|
+
}
|
|
212
|
+
|
|
213
|
+
const _internals = { _dist, _pct, RELIABLE_N };
|
|
214
|
+
|
|
215
|
+
|
|
216
|
+
/***/ }),
|
|
217
|
+
|
|
5
218
|
/***/ 4113:
|
|
6
219
|
/***/ ((__unused_webpack___webpack_module__, __webpack_exports__, __webpack_require__) => {
|
|
7
220
|
|
|
@@ -12,7 +225,7 @@ export const modules = {
|
|
|
12
225
|
/* harmony import */ var node_child_process__WEBPACK_IMPORTED_MODULE_0__ = __webpack_require__(1421);
|
|
13
226
|
/* harmony import */ var node_fs__WEBPACK_IMPORTED_MODULE_1__ = __webpack_require__(3024);
|
|
14
227
|
/* harmony import */ var node_path__WEBPACK_IMPORTED_MODULE_2__ = __webpack_require__(6760);
|
|
15
|
-
/* harmony import */ var _fix_verify_js__WEBPACK_IMPORTED_MODULE_3__ = __webpack_require__(
|
|
228
|
+
/* harmony import */ var _fix_verify_js__WEBPACK_IMPORTED_MODULE_3__ = __webpack_require__(3526);
|
|
16
229
|
// Closed-loop fix verification (v0.68).
|
|
17
230
|
//
|
|
18
231
|
// Existing `fix-verify.js` does scan + lint. This module adds the third
|
|
@@ -174,7 +387,7 @@ function _summarize(legs, verdict) {
|
|
|
174
387
|
|
|
175
388
|
/***/ }),
|
|
176
389
|
|
|
177
|
-
/***/
|
|
390
|
+
/***/ 3526:
|
|
178
391
|
/***/ ((__unused_webpack___webpack_module__, __webpack_exports__, __webpack_require__) => {
|
|
179
392
|
|
|
180
393
|
// ESM COMPAT FLAG
|
|
@@ -193,8 +406,8 @@ var external_node_child_process_ = __webpack_require__(1421);
|
|
|
193
406
|
var external_node_fs_ = __webpack_require__(3024);
|
|
194
407
|
// EXTERNAL MODULE: external "node:path"
|
|
195
408
|
var external_node_path_ = __webpack_require__(6760);
|
|
196
|
-
// EXTERNAL MODULE: ./src/engine.js +
|
|
197
|
-
var engine = __webpack_require__(
|
|
409
|
+
// EXTERNAL MODULE: ./src/engine.js + 595 modules
|
|
410
|
+
var engine = __webpack_require__(4660);
|
|
198
411
|
;// CONCATENATED MODULE: ./src/posture/fix-honesty-gate.js
|
|
199
412
|
// Deterministic honesty gates on fix / finding output (#7).
|
|
200
413
|
//
|
|
@@ -372,6 +585,157 @@ function gateFixOutput({ residual, verdict, evidence, signals } = {}) {
|
|
|
372
585
|
|
|
373
586
|
const _internals = Object.freeze({ BANNED_RESIDUAL_PHRASES, CITATION_RE, FP_VERDICTS });
|
|
374
587
|
|
|
588
|
+
;// CONCATENATED MODULE: ./src/posture/test-runner.js
|
|
589
|
+
// R5 (partial, roadmap) — the project's own test suite as a verification
|
|
590
|
+
// stage for `verifyFix()` (see `fix-verify.js`). Closes the gap where a
|
|
591
|
+
// "verified" fix only proved a finding's stableId stopped firing — a patch
|
|
592
|
+
// that deletes the feature entirely would satisfy that just as well as a
|
|
593
|
+
// real fix. Running the project's own tests is the cheapest available check
|
|
594
|
+
// that the application still works.
|
|
595
|
+
//
|
|
596
|
+
// Execution-safety note: this spawns the TARGET PROJECT's own test command
|
|
597
|
+
// in the target project's own directory. That is deliberately NOT routed
|
|
598
|
+
// through the R1 confinement sandbox (`../sandbox/`). That sandbox exists to
|
|
599
|
+
// contain untrusted proof-of-concept exploit code the scanner itself
|
|
600
|
+
// synthesizes — code nobody has vetted, being run for the first time. A
|
|
601
|
+
// project's pre-existing test suite is the opposite case: it is the
|
|
602
|
+
// project's own trusted source, already sitting on disk, and running it is
|
|
603
|
+
// exactly what a human developer does by hand before trusting a fix.
|
|
604
|
+
// Wrapping "npm test" / "pytest" / "go test" in the PoC sandbox's
|
|
605
|
+
// syscall/network/filesystem restrictions would break the large majority of
|
|
606
|
+
// real test suites (they bind local ports, spawn child processes, write temp
|
|
607
|
+
// fixtures, etc.) for no corresponding security benefit.
|
|
608
|
+
|
|
609
|
+
|
|
610
|
+
|
|
611
|
+
|
|
612
|
+
|
|
613
|
+
const DEFAULT_TIMEOUT_MS = 300_000;
|
|
614
|
+
const NPM_PLACEHOLDER = /Error: no test specified/i;
|
|
615
|
+
|
|
616
|
+
function _exists(scanRoot, rel) {
|
|
617
|
+
try { return external_node_fs_.existsSync(external_node_path_.join(scanRoot, rel)); } catch { return false; }
|
|
618
|
+
}
|
|
619
|
+
|
|
620
|
+
function _isDir(scanRoot, rel) {
|
|
621
|
+
try { return external_node_fs_.statSync(external_node_path_.join(scanRoot, rel)).isDirectory(); } catch { return false; }
|
|
622
|
+
}
|
|
623
|
+
|
|
624
|
+
function _binaryAvailable(cmd) {
|
|
625
|
+
try {
|
|
626
|
+
const r = (0,external_node_child_process_.spawnSync)(cmd, ['--version'], { timeout: 5_000, stdio: 'ignore' });
|
|
627
|
+
return !(r.error && r.error.code === 'ENOENT');
|
|
628
|
+
} catch {
|
|
629
|
+
return false;
|
|
630
|
+
}
|
|
631
|
+
}
|
|
632
|
+
|
|
633
|
+
// Detect the project's test command. Read-only — never spawns the actual
|
|
634
|
+
// test run, only (optionally) a cheap `--version` probe to confirm a tool
|
|
635
|
+
// like `pytest` is actually installed before committing to it. Returns
|
|
636
|
+
// `null` when nothing detectable is found — most scanned repos will hit
|
|
637
|
+
// this path, and that must not be treated as a failure by callers.
|
|
638
|
+
function detectTestCommand(scanRoot) {
|
|
639
|
+
if (!scanRoot) return null;
|
|
640
|
+
|
|
641
|
+
// JS/TS — package.json with a real (non-placeholder) `scripts.test`.
|
|
642
|
+
let pkg = null;
|
|
643
|
+
try { pkg = JSON.parse(external_node_fs_.readFileSync(external_node_path_.join(scanRoot, 'package.json'), 'utf8')); } catch { pkg = null; }
|
|
644
|
+
const testScript = pkg && pkg.scripts && pkg.scripts.test;
|
|
645
|
+
if (testScript && !NPM_PLACEHOLDER.test(String(testScript))) {
|
|
646
|
+
if (_exists(scanRoot, 'pnpm-lock.yaml')) return { cmd: 'pnpm', args: ['test'], kind: 'pnpm' };
|
|
647
|
+
if (_exists(scanRoot, 'yarn.lock')) return { cmd: 'yarn', args: ['test'], kind: 'yarn' };
|
|
648
|
+
if (_exists(scanRoot, 'bun.lockb') || _exists(scanRoot, 'bun.lock')) return { cmd: 'bun', args: ['test'], kind: 'bun' };
|
|
649
|
+
return { cmd: 'npm', args: ['test', '--silent'], kind: 'npm' };
|
|
650
|
+
}
|
|
651
|
+
|
|
652
|
+
// Python — pytest.ini / pyproject.toml / tox.ini / a tests/ dir, and the
|
|
653
|
+
// `pytest` binary actually available. If pytest isn't installed we do NOT
|
|
654
|
+
// report a python test command — falling through lets a later language
|
|
655
|
+
// marker (e.g. go.mod in a polyglot repo) still be detected.
|
|
656
|
+
const pyMarker = _exists(scanRoot, 'pytest.ini') || _exists(scanRoot, 'pyproject.toml') ||
|
|
657
|
+
_exists(scanRoot, 'tox.ini') || _isDir(scanRoot, 'tests');
|
|
658
|
+
if (pyMarker && _binaryAvailable('pytest')) {
|
|
659
|
+
return { cmd: 'pytest', args: ['-q'], kind: 'pytest' };
|
|
660
|
+
}
|
|
661
|
+
|
|
662
|
+
// Go
|
|
663
|
+
if (_exists(scanRoot, 'go.mod')) {
|
|
664
|
+
return { cmd: 'go', args: ['test', './...'], kind: 'go' };
|
|
665
|
+
}
|
|
666
|
+
|
|
667
|
+
return null;
|
|
668
|
+
}
|
|
669
|
+
|
|
670
|
+
// Run the detected test command with a walltime budget. Always returns a
|
|
671
|
+
// result object — never throws. Distinguishes four outcomes:
|
|
672
|
+
// - no detectable/runnable command -> status: 'skipped' (does NOT fail)
|
|
673
|
+
// - ran and exited 0 -> status: 'passed'
|
|
674
|
+
// - ran and exited non-zero -> status: 'failed'
|
|
675
|
+
// - ran past the timeout budget -> status: 'failed', timedOut: true
|
|
676
|
+
function runProjectTests(scanRoot, { timeoutMs = DEFAULT_TIMEOUT_MS } = {}) {
|
|
677
|
+
const startedAt = Date.now();
|
|
678
|
+
const command = detectTestCommand(scanRoot);
|
|
679
|
+
if (!command) {
|
|
680
|
+
return {
|
|
681
|
+
status: 'skipped', passed: null, skipped: true,
|
|
682
|
+
reason: 'no-test-command-detected', exitCode: null, timedOut: false,
|
|
683
|
+
durationMs: Date.now() - startedAt,
|
|
684
|
+
};
|
|
685
|
+
}
|
|
686
|
+
|
|
687
|
+
let r;
|
|
688
|
+
try {
|
|
689
|
+
r = (0,external_node_child_process_.spawnSync)(command.cmd, command.args, {
|
|
690
|
+
cwd: scanRoot,
|
|
691
|
+
encoding: 'utf8',
|
|
692
|
+
timeout: timeoutMs,
|
|
693
|
+
env: { ...process.env, CI: '1' },
|
|
694
|
+
});
|
|
695
|
+
} catch (e) {
|
|
696
|
+
// The spawn call itself threw (rare — e.g. cwd vanished). Treat as
|
|
697
|
+
// "could not run", not "ran and failed".
|
|
698
|
+
return {
|
|
699
|
+
status: 'skipped', passed: null, skipped: true,
|
|
700
|
+
reason: `spawn-error: ${e.message}`, exitCode: null, timedOut: false,
|
|
701
|
+
durationMs: Date.now() - startedAt,
|
|
702
|
+
};
|
|
703
|
+
}
|
|
704
|
+
const durationMs = Date.now() - startedAt;
|
|
705
|
+
|
|
706
|
+
if (r.error && r.error.code === 'ENOENT') {
|
|
707
|
+
// The detected tool isn't actually installed on this machine. Not a
|
|
708
|
+
// test failure — the suite never ran.
|
|
709
|
+
return {
|
|
710
|
+
status: 'skipped', passed: null, skipped: true,
|
|
711
|
+
reason: `${command.kind}-not-installed`, exitCode: null, timedOut: false, durationMs,
|
|
712
|
+
};
|
|
713
|
+
}
|
|
714
|
+
|
|
715
|
+
if (r.status === null) {
|
|
716
|
+
// spawnSync sets status:null both on timeout-kill and on being killed by
|
|
717
|
+
// another signal; either way the run did not complete, which is a
|
|
718
|
+
// verification failure, never a skip — we asked for a result and the
|
|
719
|
+
// process was terminated before producing one.
|
|
720
|
+
return {
|
|
721
|
+
status: 'failed', passed: false, skipped: false,
|
|
722
|
+
reason: 'timed-out', exitCode: null, timedOut: true, durationMs,
|
|
723
|
+
};
|
|
724
|
+
}
|
|
725
|
+
|
|
726
|
+
return {
|
|
727
|
+
status: r.status === 0 ? 'passed' : 'failed',
|
|
728
|
+
passed: r.status === 0,
|
|
729
|
+
skipped: false,
|
|
730
|
+
reason: r.status === 0 ? null : 'test-failures',
|
|
731
|
+
exitCode: r.status,
|
|
732
|
+
timedOut: false,
|
|
733
|
+
durationMs,
|
|
734
|
+
};
|
|
735
|
+
}
|
|
736
|
+
|
|
737
|
+
// EXTERNAL MODULE: ./src/posture/fix-metrics.js
|
|
738
|
+
var fix_metrics = __webpack_require__(2238);
|
|
375
739
|
;// CONCATENATED MODULE: ./src/posture/fix-verify.js
|
|
376
740
|
// Closed-loop /fix verification (Sentinel-parity FR-L4-4, FR-L4-5).
|
|
377
741
|
//
|
|
@@ -381,6 +745,10 @@ const _internals = Object.freeze({ BANNED_RESIDUAL_PHRASES, CITATION_RE, FP_VERD
|
|
|
381
745
|
// 1. The original finding's stableId no longer fires on the patched file.
|
|
382
746
|
// 2. No new findings at severity ≥ medium were introduced by the patch.
|
|
383
747
|
// 3. The project's existing linter (when present) passes on the patched file.
|
|
748
|
+
// 4. The project's own test suite (when detectable) still passes. This is
|
|
749
|
+
// the R5 gap-closer: a patch that silently deletes the feature would
|
|
750
|
+
// satisfy (1) and (2) just as well as a real fix — only running the
|
|
751
|
+
// tests catches that. See `test-runner.js` for detection + execution.
|
|
384
752
|
//
|
|
385
753
|
// If any of those fail, the caller is expected to NOT apply the patch and
|
|
386
754
|
// instead surface a "fix plan" — a numbered list of steps the engineer can
|
|
@@ -392,6 +760,8 @@ const _internals = Object.freeze({ BANNED_RESIDUAL_PHRASES, CITATION_RE, FP_VERD
|
|
|
392
760
|
|
|
393
761
|
|
|
394
762
|
|
|
763
|
+
|
|
764
|
+
|
|
395
765
|
const SEVERITY_RANK = { critical: 0, high: 1, medium: 2, low: 3, info: 4 };
|
|
396
766
|
|
|
397
767
|
// Run a focused re-scan over just the patched file(s) using the in-memory
|
|
@@ -494,29 +864,150 @@ function runLinter(cwd, cmd, args) {
|
|
|
494
864
|
// dishonest or over-claiming fix fails the gate. When `fixMeta` is absent
|
|
495
865
|
// (the deterministic MCP write path, which has no claims to check) the honesty
|
|
496
866
|
// gate is skipped and behavior is unchanged.
|
|
867
|
+
// R5 (partial) — the test-suite stage. Runs the target project's own tests,
|
|
868
|
+
// in the target project's own directory, against whatever is currently on
|
|
869
|
+
// disk there. See `test-runner.js`'s header comment for why that run is
|
|
870
|
+
// deliberately NOT routed through the R1 PoC-confinement sandbox: this is
|
|
871
|
+
// the project's own already-trusted suite, not untrusted synthesized code.
|
|
872
|
+
//
|
|
873
|
+
// Caveat that matters for callers: `verifyPatch` above re-scans the
|
|
874
|
+
// candidate patch purely in memory (no write to disk), but a test runner
|
|
875
|
+
// needs real files — there is no cheap way to hand a runner an in-memory
|
|
876
|
+
// overlay. So this leg reports on the CURRENT on-disk tree, not the
|
|
877
|
+
// candidate `files` map, when `verifyFix` is used as a pre-write preview
|
|
878
|
+
// (e.g. the `verify_fix` MCP tool). Callers that apply the patch first and
|
|
879
|
+
// then re-verify get the strongest signal; that ordering is not enforced
|
|
880
|
+
// here — it's the caller's responsibility, same as it already is for the
|
|
881
|
+
// closed-loop `fix-verify-loop.js` path.
|
|
882
|
+
// Does the caller's candidate patch differ from what is on disk right now?
|
|
883
|
+
// If so, any test run necessarily exercised the pre-patch tree. Compared by
|
|
884
|
+
// content so a patch that happens to match disk (already applied) is correctly
|
|
885
|
+
// treated as NOT pre-patch.
|
|
886
|
+
function _candidateDiffersFromDisk(scanRoot, files) {
|
|
887
|
+
if (!files || typeof files !== 'object') return false;
|
|
888
|
+
for (const [rel, content] of Object.entries(files)) {
|
|
889
|
+
if (typeof content !== 'string') continue;
|
|
890
|
+
try {
|
|
891
|
+
const abs = external_node_path_.resolve(scanRoot, rel);
|
|
892
|
+
if (external_node_fs_.readFileSync(abs, 'utf8') !== content) return true;
|
|
893
|
+
} catch {
|
|
894
|
+
return true; // candidate file absent on disk -> definitely not applied
|
|
895
|
+
}
|
|
896
|
+
}
|
|
897
|
+
return false;
|
|
898
|
+
}
|
|
899
|
+
|
|
497
900
|
async function verifyFix({
|
|
498
901
|
scanRoot,
|
|
499
902
|
originalFindingStableId,
|
|
500
903
|
files,
|
|
501
904
|
depFileContents,
|
|
502
905
|
fixMeta,
|
|
906
|
+
testTimeoutMs,
|
|
907
|
+
recordMetrics = true,
|
|
908
|
+
poc,
|
|
503
909
|
} = {}) {
|
|
910
|
+
// R5 (reporting half) — time each stage as it runs. Measured here rather
|
|
911
|
+
// than inside each stage because only this function knows the boundaries of
|
|
912
|
+
// one verification ATTEMPT, which is the unit the distribution is over.
|
|
913
|
+
const stages = {};
|
|
914
|
+
const t0 = Date.now();
|
|
915
|
+
let mark = t0;
|
|
916
|
+
const _lap = (name) => { const now = Date.now(); stages[name] = now - mark; mark = now; };
|
|
917
|
+
|
|
504
918
|
const rescan = await verifyPatch({ scanRoot, originalFindingStableId, files, depFileContents });
|
|
919
|
+
_lap('rescan');
|
|
505
920
|
const lint = runProjectLinter(scanRoot, Object.keys(files || {}));
|
|
921
|
+
_lap('lint');
|
|
922
|
+
const tests = runProjectTests(scanRoot, testTimeoutMs != null ? { timeoutMs: testTimeoutMs } : {});
|
|
923
|
+
_lap('tests');
|
|
924
|
+
// True when a candidate patch was supplied but has not been written, so the
|
|
925
|
+
// suite necessarily ran against the pre-patch tree. Surfaced in the summary
|
|
926
|
+
// and on the result so a caller cannot mistake it for a verified patch.
|
|
927
|
+
const _testedPrePatch = !tests.skipped && _candidateDiffersFromDisk(scanRoot, files);
|
|
928
|
+
const testsOk = tests.skipped ? true : tests.passed === true;
|
|
506
929
|
let honesty = null;
|
|
507
930
|
if (fixMeta && typeof fixMeta === 'object') {
|
|
508
931
|
try { honesty = gateFixOutput(fixMeta); } catch { honesty = null; }
|
|
509
932
|
}
|
|
510
|
-
|
|
933
|
+
_lap('honesty');
|
|
934
|
+
|
|
935
|
+
// R5 — the PoC leg. Re-run the finding's proof-of-concept against the
|
|
936
|
+
// CANDIDATE patch inside R1's sandbox. A patch that still lets the PoC
|
|
937
|
+
// demonstrate the predicted effect has not fixed anything, however green the
|
|
938
|
+
// re-scan looks: the re-scan only proves the DETECTOR stopped firing, which
|
|
939
|
+
// a cosmetic edit can achieve. Execution is the stronger claim.
|
|
940
|
+
//
|
|
941
|
+
// Direction matters and is asymmetric on purpose. `execution-proven` after
|
|
942
|
+
// the patch is a hard FAIL. Anything else is NOT a pass — a PoC that failed
|
|
943
|
+
// to run, or a sandbox that could not start, is recorded as `inconclusive`
|
|
944
|
+
// and left out of the verdict entirely. Treating "could not prove it" as
|
|
945
|
+
// "fixed" is exactly the false confidence this leg exists to prevent.
|
|
946
|
+
let pocLeg = { status: 'not-requested', reason: null, tier: null };
|
|
947
|
+
if (poc?.code) {
|
|
948
|
+
try {
|
|
949
|
+
const { proveFinding } = await __webpack_require__.e(/* import() */ 259).then(__webpack_require__.bind(__webpack_require__, 8259));
|
|
950
|
+
const proved = await proveFinding({ ...(poc.finding || {}), poc }, { files });
|
|
951
|
+
const tier = proved.proofTier;
|
|
952
|
+
pocLeg = tier === 'execution-proven'
|
|
953
|
+
? { status: 'still-exploitable', tier, reason: proved.proofEvidence?.observed || null }
|
|
954
|
+
: proved.proofEvidence?.ran
|
|
955
|
+
? { status: 'no-longer-proven', tier, reason: proved.proofEvidence?.reason || null }
|
|
956
|
+
: { status: 'inconclusive', tier, reason: proved.proofEvidence?.reason || null };
|
|
957
|
+
} catch (e) {
|
|
958
|
+
pocLeg = { status: 'inconclusive', tier: null, reason: `proof harness error: ${e.message}` };
|
|
959
|
+
}
|
|
960
|
+
}
|
|
961
|
+
_lap('poc');
|
|
962
|
+
const pocOk = pocLeg.status !== 'still-exploitable';
|
|
963
|
+
|
|
964
|
+
const ok = rescan.ok && (lint.ok || lint.skipped) && testsOk && pocOk && (honesty ? honesty.ok : true);
|
|
965
|
+
const durations = { ...stages, totalMs: Date.now() - t0 };
|
|
511
966
|
const summary = [
|
|
512
967
|
`re-scan: ${rescan.ok ? 'PASS' : 'FAIL — ' + rescan.reason}`,
|
|
513
968
|
`linter: ${lint.runner === 'none' ? 'skipped (no linter config)'
|
|
514
969
|
: lint.skipped ? `${lint.runner} not installed`
|
|
515
970
|
: lint.ok ? `${lint.runner} PASS`
|
|
516
971
|
: `${lint.runner} FAIL (exit ${lint.exitCode})`}`,
|
|
972
|
+
// Say which tree the suite actually ran against. `files` is a candidate
|
|
973
|
+
// patch held in memory; the runner needs real files, so it sees whatever is
|
|
974
|
+
// on disk. Reporting a bare "PASS" here would let a caller believe the
|
|
975
|
+
// PATCH passed the tests when the suite may have run on unpatched code.
|
|
976
|
+
`tests: ${tests.skipped ? `skipped (${tests.reason})`
|
|
977
|
+
: tests.timedOut ? 'FAIL (timed out)'
|
|
978
|
+
: tests.passed ? `PASS${_testedPrePatch ? ' — on the CURRENT on-disk tree, NOT the candidate patch' : ''}`
|
|
979
|
+
: `FAIL (exit ${tests.exitCode})`}`,
|
|
517
980
|
honesty ? `honesty: ${honesty.ok ? `PASS (${honesty.tier})` : 'FAIL — ' + honesty.violations.join('; ')}` : null,
|
|
981
|
+
// Never render `inconclusive` as a pass — say plainly that nothing was proven.
|
|
982
|
+
pocLeg.status === 'not-requested' ? null
|
|
983
|
+
: pocLeg.status === 'still-exploitable' ? `poc: FAIL — the proof-of-concept still demonstrates the vulnerability against the patch`
|
|
984
|
+
: pocLeg.status === 'no-longer-proven' ? 'poc: PASS (ran against the patch and no longer demonstrates the vulnerability)'
|
|
985
|
+
: `poc: inconclusive — not counted either way (${pocLeg.reason || 'no detail reported'})`,
|
|
518
986
|
].filter(Boolean).join('\n');
|
|
519
|
-
|
|
987
|
+
// Persist the attempt so the distribution can be reported from real runs.
|
|
988
|
+
// `testsRan` is the load-bearing field: it is what keeps "verified with no
|
|
989
|
+
// test suite to run" out of the headline time-to-validated-fix bucket.
|
|
990
|
+
// A patch that was never written to disk is recorded too, but flagged — its
|
|
991
|
+
// suite ran against the pre-patch tree, so its timing is real while its
|
|
992
|
+
// verdict is about a different tree.
|
|
993
|
+
if (recordMetrics && scanRoot) {
|
|
994
|
+
(0,fix_metrics/* recordFixAttempt */.I3)(scanRoot, {
|
|
995
|
+
at: new Date().toISOString(),
|
|
996
|
+
stableId: originalFindingStableId || null,
|
|
997
|
+
ok,
|
|
998
|
+
testsRan: !tests.skipped,
|
|
999
|
+
testsPassed: tests.skipped ? null : tests.passed === true,
|
|
1000
|
+
testedPrePatch: _testedPrePatch,
|
|
1001
|
+
lintRan: !(lint.skipped || lint.runner === 'none'),
|
|
1002
|
+
honestyGated: honesty != null,
|
|
1003
|
+
pocStatus: pocLeg.status,
|
|
1004
|
+
files: Object.keys(files || {}).length,
|
|
1005
|
+
stages,
|
|
1006
|
+
totalMs: durations.totalMs,
|
|
1007
|
+
});
|
|
1008
|
+
}
|
|
1009
|
+
|
|
1010
|
+
return { ok, rescan, lint, tests, testedPrePatch: _testedPrePatch, honesty, poc: pocLeg, durations, summary };
|
|
520
1011
|
}
|
|
521
1012
|
|
|
522
1013
|
|
package/dist/178.index.js
CHANGED
|
@@ -13,7 +13,7 @@ export const modules = {
|
|
|
13
13
|
/* harmony import */ var node_child_process__WEBPACK_IMPORTED_MODULE_0__ = __webpack_require__(1421);
|
|
14
14
|
/* harmony import */ var node_fs__WEBPACK_IMPORTED_MODULE_1__ = __webpack_require__(3024);
|
|
15
15
|
/* harmony import */ var node_path__WEBPACK_IMPORTED_MODULE_2__ = __webpack_require__(6760);
|
|
16
|
-
/* harmony import */ var _engine_js__WEBPACK_IMPORTED_MODULE_3__ = __webpack_require__(
|
|
16
|
+
/* harmony import */ var _engine_js__WEBPACK_IMPORTED_MODULE_3__ = __webpack_require__(4660);
|
|
17
17
|
// Time-travel + counterfactual scanning (v0.68).
|
|
18
18
|
//
|
|
19
19
|
// Two new modes that exploit the pure-input shape of runFullScan:
|