@clear-capabilities/agentic-security-scanner 0.130.0 → 0.133.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (50) hide show
  1. package/CHANGELOG.md +247 -0
  2. package/bin/agentic-security.js +39 -3
  3. package/dist/113.index.js +294 -5
  4. package/dist/178.index.js +1 -1
  5. package/dist/207.index.js +7 -4
  6. package/dist/238.index.js +218 -0
  7. package/dist/259.index.js +975 -0
  8. package/dist/384.index.js +1 -1
  9. package/dist/435.index.js +2 -2
  10. package/dist/526.index.js +294 -5
  11. package/dist/637.index.js +1 -1
  12. package/dist/agentic-security.mjs +18 -57
  13. package/dist/agentic-security.mjs.sha256 +1 -1
  14. package/package.json +19 -10
  15. package/src/engine.js +48 -1
  16. package/src/ir/parser-js.js +8 -0
  17. package/src/llm-validator/cost-ceiling.js +199 -0
  18. package/src/llm-validator/index.js +241 -12
  19. package/src/llm-validator/local-endpoint.js +90 -0
  20. package/src/mcp/tools.js +2 -2
  21. package/src/posture/CLAUDE.md +83 -6
  22. package/src/posture/accuracy-scorecard.js +37 -6
  23. package/src/posture/attestation.js +7 -4
  24. package/src/posture/corpus-enroll.js +303 -0
  25. package/src/posture/corpus-match.js +67 -0
  26. package/src/posture/custom-rules.js +2 -2
  27. package/src/posture/execution-proof.js +44 -4
  28. package/src/posture/fix-metrics.js +197 -0
  29. package/src/posture/fix-verify.js +76 -2
  30. package/src/posture/integrity.js +42 -9
  31. package/src/posture/learning.js +8 -1
  32. package/src/posture/model-routing.js +26 -0
  33. package/src/posture/model-trust.js +174 -0
  34. package/src/posture/poc-inprocess.js +165 -0
  35. package/src/posture/prove-findings.js +148 -0
  36. package/src/posture/root-cause-sweep.js +0 -0
  37. package/src/posture/rule-overrides.js +64 -3
  38. package/src/posture/state-dir.js +25 -0
  39. package/src/posture/vuln-archaeology.js +231 -0
  40. package/src/report/index.js +7 -0
  41. package/src/runScan.js +2 -6
  42. package/src/sandbox/CLAUDE.md +190 -46
  43. package/src/sandbox/backend-namespace.js +328 -48
  44. package/src/sandbox/backend-userspace.js +6 -19
  45. package/src/sandbox/capabilities.js +132 -4
  46. package/src/sandbox/limits.js +21 -0
  47. package/src/sandbox/result.js +1 -1
  48. package/src/sast/CLAUDE.md +4 -0
  49. package/src/sast/crypto-specialist.js +247 -0
  50. package/src/util/glob.js +173 -0
package/dist/384.index.js CHANGED
@@ -8,7 +8,7 @@ export const modules = {
8
8
  /* harmony export */ __webpack_require__.d(__webpack_exports__, {
9
9
  /* harmony export */ scanCredentials: () => (/* reexport safe */ _engine_js__WEBPACK_IMPORTED_MODULE_0__.Sv)
10
10
  /* harmony export */ });
11
- /* harmony import */ var _engine_js__WEBPACK_IMPORTED_MODULE_0__ = __webpack_require__(4660);
11
+ /* harmony import */ var _engine_js__WEBPACK_IMPORTED_MODULE_0__ = __webpack_require__(5865);
12
12
  // Secrets submodule view of the engine — credential + entropy + TODO scanning.
13
13
 
14
14
 
package/dist/435.index.js CHANGED
@@ -466,8 +466,8 @@ const cve_lookup_internals = { CACHE_DIR, CVE_RE, _stalenessTier };
466
466
 
467
467
 
468
468
 
469
- // Lazy-loaded: these transitively pull in npm packages (fast-glob,
470
- // @babel/core) that aren't available in the plugin-cache install path
469
+ // Lazy-loaded: these transitively pull in npm packages (@babel/core and
470
+ // friends) that aren't available in the plugin-cache install path
471
471
  // (no node_modules). Deferring keeps the MCP server bootable everywhere;
472
472
  // the import only runs when a tool that needs them is actually called.
473
473
  let _runScan;
package/dist/526.index.js CHANGED
@@ -1,7 +1,220 @@
1
1
  export const id = 526;
2
- export const ids = [526];
2
+ export const ids = [526,238];
3
3
  export const modules = {
4
4
 
5
+ /***/ 2238:
6
+ /***/ ((__unused_webpack___webpack_module__, __webpack_exports__, __webpack_require__) => {
7
+
8
+ /* harmony export */ __webpack_require__.d(__webpack_exports__, {
9
+ /* harmony export */ I3: () => (/* binding */ recordFixAttempt),
10
+ /* harmony export */ fixDurationReport: () => (/* binding */ fixDurationReport),
11
+ /* harmony export */ renderFixDurationSummary: () => (/* binding */ renderFixDurationSummary)
12
+ /* harmony export */ });
13
+ /* unused harmony exports FIX_STAGES, loadFixAttempts, bucketOf, summarizeFixDurations, _internals */
14
+ /* harmony import */ var node_fs__WEBPACK_IMPORTED_MODULE_0__ = __webpack_require__(3024);
15
+ /* harmony import */ var node_path__WEBPACK_IMPORTED_MODULE_1__ = __webpack_require__(6760);
16
+ /* harmony import */ var _state_dir_js__WEBPACK_IMPORTED_MODULE_2__ = __webpack_require__(1174);
17
+ // Time-to-validated-fix (R5, the reporting half).
18
+ //
19
+ // `verifyFix` already RUNS the stages and `test-runner.js` already times the
20
+ // slowest one. What did not exist was anything durable to read afterwards, so
21
+ // "how long does a fix actually take to validate" had no answer from real runs
22
+ // — only an estimate (`time-to-fix.js` guesses engineering hours from family
23
+ // and patch shape, before anything runs). This module is the opposite: it
24
+ // records what the pipeline observed and reports the distribution.
25
+ //
26
+ // THE HONESTY RULES, which are most of why this file is longer than a mean:
27
+ //
28
+ // 1. A failed attempt is NOT a data point about how long a fix takes. Fixes
29
+ // that fail verification fail fast (a re-scan that still sees the finding
30
+ // never reaches the test suite), so blending them into one average makes
31
+ // the pipeline look faster the worse it performs. Validated and failed
32
+ // attempts are summarised separately and never merged.
33
+ //
34
+ // 2. "Tests skipped" is not "tests passed". A project with no detectable
35
+ // suite can reach `ok:true` having run only the re-scan and the linter.
36
+ // That is a weaker claim than a fix whose suite executed, and it is also
37
+ // much faster, so counting the two together would quietly deflate the
38
+ // headline. They get their own bucket: `validated` means the suite ran and
39
+ // passed, `validatedWithoutTests` means there was no suite to run.
40
+ //
41
+ // 3. Every figure carries its `n`, and a percentile computed from too few
42
+ // samples is labelled unreliable rather than omitted or silently reported.
43
+ // Same precedent as the accuracy scorecard's `{n, d}` rates: a number
44
+ // without its denominator is not a measurement.
45
+ //
46
+ // Storage is append-only JSONL at `<scanRoot>/.agentic-security/fix-metrics.jsonl`,
47
+ // one record per verification attempt. Nothing here throws (posture
48
+ // convention) — an unwritable or corrupt log degrades to "no metrics", never
49
+ // to a failed verification.
50
+
51
+
52
+
53
+
54
+
55
+ const STATE_DIR = '.agentic-security';
56
+ const LOG_FILE = 'fix-metrics.jsonl';
57
+
58
+ // Below this many samples a percentile is an artifact of the sample, not a
59
+ // property of the pipeline. Reported anyway (hiding it invites re-deriving it
60
+ // wrong downstream) but flagged, so a caller cannot quote it as settled.
61
+ const RELIABLE_N = 10;
62
+
63
+ // The stages verifyFix runs, in execution order. Kept here so the recorder and
64
+ // the summariser cannot drift apart on stage naming.
65
+ const FIX_STAGES = Object.freeze(['rescan', 'lint', 'tests', 'honesty', 'poc']);
66
+
67
+ function _logPath(scanRoot) {
68
+ return node_path__WEBPACK_IMPORTED_MODULE_1__.join(scanRoot, STATE_DIR, LOG_FILE);
69
+ }
70
+
71
+ /**
72
+ * Append one verification attempt. Best-effort and silent on failure: metrics
73
+ * must never be able to fail a fix that otherwise verified.
74
+ *
75
+ * @returns {boolean} whether the record was written (for tests, not callers).
76
+ */
77
+ function recordFixAttempt(scanRoot, record) {
78
+ if (!scanRoot || !record || typeof record !== 'object') return false;
79
+ try {
80
+ const dir = node_path__WEBPACK_IMPORTED_MODULE_1__.join(scanRoot, STATE_DIR);
81
+ if (!(0,_state_dir_js__WEBPACK_IMPORTED_MODULE_2__.isSafeStateDir)(dir)) return false;
82
+ node_fs__WEBPACK_IMPORTED_MODULE_0__.mkdirSync(dir, { recursive: true });
83
+ // One writeSync of one newline-terminated line: a concurrent reader sees
84
+ // whole records or nothing, and a torn tail is dropped on read.
85
+ node_fs__WEBPACK_IMPORTED_MODULE_0__.appendFileSync(_logPath(scanRoot), JSON.stringify(record) + '\n', 'utf8');
86
+ return true;
87
+ } catch { return false; }
88
+ }
89
+
90
+ /**
91
+ * Read every well-formed attempt. A line that does not parse is skipped, not
92
+ * fatal — the last line of an interrupted write is the expected case.
93
+ */
94
+ function loadFixAttempts(scanRoot) {
95
+ try {
96
+ const raw = node_fs__WEBPACK_IMPORTED_MODULE_0__.readFileSync(_logPath(scanRoot), 'utf8');
97
+ const out = [];
98
+ for (const line of raw.split('\n')) {
99
+ if (!line.trim()) continue;
100
+ try {
101
+ const rec = JSON.parse(line);
102
+ if (rec && typeof rec === 'object' && typeof rec.totalMs === 'number') out.push(rec);
103
+ } catch { /* torn or hand-edited line — drop it, keep the rest */ }
104
+ }
105
+ return out;
106
+ } catch { return []; }
107
+ }
108
+
109
+ // Nearest-rank percentile over an ascending array. Nearest-rank rather than
110
+ // interpolated because these are observed durations, and an interpolated p50
111
+ // reports a duration that no run actually took.
112
+ function _pct(sorted, p) {
113
+ if (!sorted.length) return null;
114
+ const rank = Math.ceil((p / 100) * sorted.length);
115
+ return sorted[Math.min(sorted.length - 1, Math.max(0, rank - 1))];
116
+ }
117
+
118
+ function _dist(values) {
119
+ const v = values.filter(x => typeof x === 'number' && Number.isFinite(x) && x >= 0).sort((a, b) => a - b);
120
+ if (!v.length) return { n: 0, minMs: null, p50Ms: null, p90Ms: null, maxMs: null, meanMs: null, reliable: false };
121
+ const sum = v.reduce((a, b) => a + b, 0);
122
+ return {
123
+ n: v.length,
124
+ minMs: v[0],
125
+ p50Ms: _pct(v, 50),
126
+ p90Ms: _pct(v, 90),
127
+ maxMs: v[v.length - 1],
128
+ meanMs: Math.round(sum / v.length),
129
+ // Says whether the percentiles above may be quoted, not whether the count
130
+ // is real. n and min/max/mean are exact at any sample size.
131
+ reliable: v.length >= RELIABLE_N,
132
+ };
133
+ }
134
+
135
+ // Which bucket an attempt belongs to. Deliberately total: every attempt lands
136
+ // in exactly one, so the bucket counts always sum to the attempt count and a
137
+ // mis-shaped record cannot silently vanish from the denominator.
138
+ function bucketOf(a) {
139
+ if (!a?.ok) return 'failed';
140
+ return a.testsRan ? 'validated' : 'validatedWithoutTests';
141
+ }
142
+
143
+ /**
144
+ * Summarise a set of attempts into the reported distribution.
145
+ *
146
+ * `validated` is the headline: attempts that verified AND whose test suite
147
+ * actually ran and passed. The other two buckets exist so that headline cannot
148
+ * be inflated by counting weaker or faster outcomes inside it.
149
+ */
150
+ function summarizeFixDurations(attempts) {
151
+ const all = Array.isArray(attempts) ? attempts : [];
152
+ const buckets = { validated: [], validatedWithoutTests: [], failed: [] };
153
+ for (const a of all) buckets[bucketOf(a)].push(a);
154
+
155
+ const byStage = {};
156
+ for (const stage of FIX_STAGES) {
157
+ // Per-stage timings come from validated runs only. A stage's duration in a
158
+ // failed run is truncated by the failure (the pipeline stops), so mixing
159
+ // them in would understate every stage after the first failure point.
160
+ byStage[stage] = _dist(buckets.validated.map(a => a?.stages?.[stage]));
161
+ }
162
+
163
+ return {
164
+ attempts: all.length,
165
+ counts: {
166
+ validated: buckets.validated.length,
167
+ validatedWithoutTests: buckets.validatedWithoutTests.length,
168
+ failed: buckets.failed.length,
169
+ },
170
+ timeToValidatedFix: _dist(buckets.validated.map(a => a.totalMs)),
171
+ timeToValidatedFixWithoutTests: _dist(buckets.validatedWithoutTests.map(a => a.totalMs)),
172
+ timeToFailure: _dist(buckets.failed.map(a => a.totalMs)),
173
+ byStage,
174
+ reliableAtOrAbove: RELIABLE_N,
175
+ };
176
+ }
177
+
178
+ /** Read + summarise in one step. */
179
+ function fixDurationReport(scanRoot) {
180
+ return summarizeFixDurations(loadFixAttempts(scanRoot));
181
+ }
182
+
183
+ function _ms(v) {
184
+ if (v == null) return '—';
185
+ return v >= 1000 ? `${(v / 1000).toFixed(1)}s` : `${v}ms`;
186
+ }
187
+
188
+ /**
189
+ * One-paragraph human summary. Returns null when there is nothing measured —
190
+ * callers print nothing rather than printing an empty table.
191
+ */
192
+ function renderFixDurationSummary(sum) {
193
+ if (!sum || !sum.attempts) return null;
194
+ const d = sum.timeToValidatedFix;
195
+ const parts = [];
196
+ if (d.n) {
197
+ parts.push(
198
+ `time-to-validated-fix: median ${_ms(d.p50Ms)}, p90 ${_ms(d.p90Ms)} `
199
+ + `(n=${d.n}${d.reliable ? '' : `, below ${sum.reliableAtOrAbove} — percentiles not yet reliable`})`,
200
+ );
201
+ } else {
202
+ parts.push('time-to-validated-fix: no fix has both verified and had its test suite run yet');
203
+ }
204
+ if (sum.counts.validatedWithoutTests) {
205
+ parts.push(`${sum.counts.validatedWithoutTests} verified with no detectable test suite (excluded from the median above)`);
206
+ }
207
+ if (sum.counts.failed) {
208
+ parts.push(`${sum.counts.failed} failed verification, median ${_ms(sum.timeToFailure.p50Ms)} (counted separately)`);
209
+ }
210
+ return parts.join('; ') + '.';
211
+ }
212
+
213
+ const _internals = { _dist, _pct, RELIABLE_N };
214
+
215
+
216
+ /***/ }),
217
+
5
218
  /***/ 3526:
6
219
  /***/ ((__unused_webpack___webpack_module__, __webpack_exports__, __webpack_require__) => {
7
220
 
@@ -21,8 +234,8 @@ var external_node_child_process_ = __webpack_require__(1421);
21
234
  var external_node_fs_ = __webpack_require__(3024);
22
235
  // EXTERNAL MODULE: external "node:path"
23
236
  var external_node_path_ = __webpack_require__(6760);
24
- // EXTERNAL MODULE: ./src/engine.js + 595 modules
25
- var engine = __webpack_require__(4660);
237
+ // EXTERNAL MODULE: ./src/engine.js + 600 modules
238
+ var engine = __webpack_require__(5865);
26
239
  ;// CONCATENATED MODULE: ./src/posture/fix-honesty-gate.js
27
240
  // Deterministic honesty gates on fix / finding output (#7).
28
241
  //
@@ -349,6 +562,8 @@ function runProjectTests(scanRoot, { timeoutMs = DEFAULT_TIMEOUT_MS } = {}) {
349
562
  };
350
563
  }
351
564
 
565
+ // EXTERNAL MODULE: ./src/posture/fix-metrics.js
566
+ var fix_metrics = __webpack_require__(2238);
352
567
  ;// CONCATENATED MODULE: ./src/posture/fix-verify.js
353
568
  // Closed-loop /fix verification (Sentinel-parity FR-L4-4, FR-L4-5).
354
569
  //
@@ -374,6 +589,7 @@ function runProjectTests(scanRoot, { timeoutMs = DEFAULT_TIMEOUT_MS } = {}) {
374
589
 
375
590
 
376
591
 
592
+
377
593
  const SEVERITY_RANK = { critical: 0, high: 1, medium: 2, low: 3, info: 4 };
378
594
 
379
595
  // Run a focused re-scan over just the patched file(s) using the in-memory
@@ -516,10 +732,23 @@ async function verifyFix({
516
732
  depFileContents,
517
733
  fixMeta,
518
734
  testTimeoutMs,
735
+ recordMetrics = true,
736
+ poc,
519
737
  } = {}) {
738
+ // R5 (reporting half) — time each stage as it runs. Measured here rather
739
+ // than inside each stage because only this function knows the boundaries of
740
+ // one verification ATTEMPT, which is the unit the distribution is over.
741
+ const stages = {};
742
+ const t0 = Date.now();
743
+ let mark = t0;
744
+ const _lap = (name) => { const now = Date.now(); stages[name] = now - mark; mark = now; };
745
+
520
746
  const rescan = await verifyPatch({ scanRoot, originalFindingStableId, files, depFileContents });
747
+ _lap('rescan');
521
748
  const lint = runProjectLinter(scanRoot, Object.keys(files || {}));
749
+ _lap('lint');
522
750
  const tests = runProjectTests(scanRoot, testTimeoutMs != null ? { timeoutMs: testTimeoutMs } : {});
751
+ _lap('tests');
523
752
  // True when a candidate patch was supplied but has not been written, so the
524
753
  // suite necessarily ran against the pre-patch tree. Surfaced in the summary
525
754
  // and on the result so a caller cannot mistake it for a verified patch.
@@ -529,7 +758,39 @@ async function verifyFix({
529
758
  if (fixMeta && typeof fixMeta === 'object') {
530
759
  try { honesty = gateFixOutput(fixMeta); } catch { honesty = null; }
531
760
  }
532
- const ok = rescan.ok && (lint.ok || lint.skipped) && testsOk && (honesty ? honesty.ok : true);
761
+ _lap('honesty');
762
+
763
+ // R5 — the PoC leg. Re-run the finding's proof-of-concept against the
764
+ // CANDIDATE patch inside R1's sandbox. A patch that still lets the PoC
765
+ // demonstrate the predicted effect has not fixed anything, however green the
766
+ // re-scan looks: the re-scan only proves the DETECTOR stopped firing, which
767
+ // a cosmetic edit can achieve. Execution is the stronger claim.
768
+ //
769
+ // Direction matters and is asymmetric on purpose. `execution-proven` after
770
+ // the patch is a hard FAIL. Anything else is NOT a pass — a PoC that failed
771
+ // to run, or a sandbox that could not start, is recorded as `inconclusive`
772
+ // and left out of the verdict entirely. Treating "could not prove it" as
773
+ // "fixed" is exactly the false confidence this leg exists to prevent.
774
+ let pocLeg = { status: 'not-requested', reason: null, tier: null };
775
+ if (poc?.code) {
776
+ try {
777
+ const { proveFinding } = await Promise.resolve(/* import() */).then(__webpack_require__.bind(__webpack_require__, 1291));
778
+ const proved = await proveFinding({ ...(poc.finding || {}), poc }, { files });
779
+ const tier = proved.proofTier;
780
+ pocLeg = tier === 'execution-proven'
781
+ ? { status: 'still-exploitable', tier, reason: proved.proofEvidence?.observed || null }
782
+ : proved.proofEvidence?.ran
783
+ ? { status: 'no-longer-proven', tier, reason: proved.proofEvidence?.reason || null }
784
+ : { status: 'inconclusive', tier, reason: proved.proofEvidence?.reason || null };
785
+ } catch (e) {
786
+ pocLeg = { status: 'inconclusive', tier: null, reason: `proof harness error: ${e.message}` };
787
+ }
788
+ }
789
+ _lap('poc');
790
+ const pocOk = pocLeg.status !== 'still-exploitable';
791
+
792
+ const ok = rescan.ok && (lint.ok || lint.skipped) && testsOk && pocOk && (honesty ? honesty.ok : true);
793
+ const durations = { ...stages, totalMs: Date.now() - t0 };
533
794
  const summary = [
534
795
  `re-scan: ${rescan.ok ? 'PASS' : 'FAIL — ' + rescan.reason}`,
535
796
  `linter: ${lint.runner === 'none' ? 'skipped (no linter config)'
@@ -545,8 +806,36 @@ async function verifyFix({
545
806
  : tests.passed ? `PASS${_testedPrePatch ? ' — on the CURRENT on-disk tree, NOT the candidate patch' : ''}`
546
807
  : `FAIL (exit ${tests.exitCode})`}`,
547
808
  honesty ? `honesty: ${honesty.ok ? `PASS (${honesty.tier})` : 'FAIL — ' + honesty.violations.join('; ')}` : null,
809
+ // Never render `inconclusive` as a pass — say plainly that nothing was proven.
810
+ pocLeg.status === 'not-requested' ? null
811
+ : pocLeg.status === 'still-exploitable' ? `poc: FAIL — the proof-of-concept still demonstrates the vulnerability against the patch`
812
+ : pocLeg.status === 'no-longer-proven' ? 'poc: PASS (ran against the patch and no longer demonstrates the vulnerability)'
813
+ : `poc: inconclusive — not counted either way (${pocLeg.reason || 'no detail reported'})`,
548
814
  ].filter(Boolean).join('\n');
549
- return { ok, rescan, lint, tests, testedPrePatch: _testedPrePatch, honesty, summary };
815
+ // Persist the attempt so the distribution can be reported from real runs.
816
+ // `testsRan` is the load-bearing field: it is what keeps "verified with no
817
+ // test suite to run" out of the headline time-to-validated-fix bucket.
818
+ // A patch that was never written to disk is recorded too, but flagged — its
819
+ // suite ran against the pre-patch tree, so its timing is real while its
820
+ // verdict is about a different tree.
821
+ if (recordMetrics && scanRoot) {
822
+ (0,fix_metrics/* recordFixAttempt */.I3)(scanRoot, {
823
+ at: new Date().toISOString(),
824
+ stableId: originalFindingStableId || null,
825
+ ok,
826
+ testsRan: !tests.skipped,
827
+ testsPassed: tests.skipped ? null : tests.passed === true,
828
+ testedPrePatch: _testedPrePatch,
829
+ lintRan: !(lint.skipped || lint.runner === 'none'),
830
+ honestyGated: honesty != null,
831
+ pocStatus: pocLeg.status,
832
+ files: Object.keys(files || {}).length,
833
+ stages,
834
+ totalMs: durations.totalMs,
835
+ });
836
+ }
837
+
838
+ return { ok, rescan, lint, tests, testedPrePatch: _testedPrePatch, honesty, poc: pocLeg, durations, summary };
550
839
  }
551
840
 
552
841
 
package/dist/637.index.js CHANGED
@@ -11,7 +11,7 @@ export const modules = {
11
11
  /* harmony export */ renderPrDeltaText: () => (/* binding */ renderPrDeltaText)
12
12
  /* harmony export */ });
13
13
  /* harmony import */ var node_child_process__WEBPACK_IMPORTED_MODULE_0__ = __webpack_require__(1421);
14
- /* harmony import */ var _engine_js__WEBPACK_IMPORTED_MODULE_1__ = __webpack_require__(4660);
14
+ /* harmony import */ var _engine_js__WEBPACK_IMPORTED_MODULE_1__ = __webpack_require__(5865);
15
15
  // Shadowscan / security-DELTA on PR (v0.72).
16
16
  //
17
17
  // Most SAST PR-comment integrations show absolute counts — "12 findings