progmune-runtime 2.1.6 → 3.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +108 -468
- package/dist/ablation-study.js +144 -0
- package/dist/ablation-study.test.js +18 -0
- package/dist/action-runtime.js +3 -1
- package/dist/active-learning.js +211 -0
- package/dist/analytics.js +139 -0
- package/dist/asset-factory.js +309 -0
- package/dist/asset-growth.js +244 -0
- package/dist/asset-promotion.js +382 -0
- package/dist/asset-quality.js +550 -0
- package/dist/audit/business-translator.js +285 -0
- package/dist/audit/cli.js +66 -0
- package/dist/audit/formatters/html.js +379 -0
- package/dist/audit/formatters/json.js +11 -0
- package/dist/audit/formatters/markdown.js +192 -0
- package/dist/audit/formatters/terminal.js +189 -0
- package/dist/audit/index.js +25 -0
- package/dist/audit/report-builder.js +318 -0
- package/dist/audit/types.js +8 -0
- package/dist/audit.js +3 -3
- package/dist/auto-benchmark-generator.js +137 -0
- package/dist/auto-benchmark-generator.test.js +45 -0
- package/dist/auto-protocol-synthesizer.js +362 -0
- package/dist/auto-protocol-synthesizer.test.js +82 -0
- package/dist/autonomous-patch.js +175 -0
- package/dist/autonomous-patch.test.js +128 -0
- package/dist/badge/badge-server.js +98 -0
- package/dist/behavior-miner.js +442 -0
- package/dist/belief-layer.js +475 -0
- package/dist/benchmark-count.js +5 -0
- package/dist/benchmark-generator.js +211 -0
- package/dist/benchmark-harness.js +201 -0
- package/dist/benchmark-pass-rate.js +7 -0
- package/dist/benchmark-report.js +8 -3
- package/dist/benchmark-save.js +14 -1
- package/dist/bootstrap-validation.js +197 -0
- package/dist/bootstrap-validation.test.js +51 -0
- package/dist/branch-ledger.js +1 -1
- package/dist/capability-gap.js +130 -0
- package/dist/certify-html.js +351 -0
- package/dist/certify.js +326 -0
- package/dist/check.js +4 -4
- package/dist/compliance-miner.js +447 -0
- package/dist/continuous-benchmark.js +194 -0
- package/dist/continuous-benchmark.test.js +116 -0
- package/dist/corpus-stats.js +173 -0
- package/dist/counterfactual-engine.js +288 -0
- package/dist/coverage-dashboard.js +109 -0
- package/dist/coverage-system.test.js +205 -0
- package/dist/cross-repo-precision.js +352 -0
- package/dist/cve-benchmark.js +180 -0
- package/dist/cve-benchmark.test.js +28 -0
- package/dist/cve-collector.js +73 -0
- package/dist/data-quality.js +141 -0
- package/dist/decision-engine.js +388 -0
- package/dist/derive-metadata.js +250 -0
- package/dist/difficulty-active.test.js +198 -0
- package/dist/difficulty-map.js +244 -0
- package/dist/discovery-analytics.js +125 -0
- package/dist/discovery-model.js +149 -0
- package/dist/discovery-optimize.test.js +199 -0
- package/dist/discovery-trace.js +276 -0
- package/dist/discovery-trace.test.js +97 -0
- package/dist/emitter.js +83 -1
- package/dist/enterprise-dashboard.js +405 -0
- package/dist/eval-hardening.js +297 -0
- package/dist/eval-hardening.test.js +85 -0
- package/dist/evaluation-campaign.js +359 -0
- package/dist/evaluation-campaign.test.js +181 -0
- package/dist/evidence-growth.js +143 -0
- package/dist/evidence-repository.js +209 -0
- package/dist/evidence-system.js +441 -0
- package/dist/execute.js +15 -7
- package/dist/experimental/software-physics.js +291 -0
- package/dist/experimental/state-inference.js +516 -0
- package/dist/experimental/unsupervised-physics.js +230 -0
- package/dist/extract-ir-python.js +54 -7
- package/dist/extract-ir.js +376 -12
- package/dist/failure-collector.js +2 -2
- package/dist/failure-corpus.js +322 -9
- package/dist/feedback.js +16 -5
- package/dist/feedback.test.js +49 -0
- package/dist/file-lock.js +1 -1
- package/dist/flywheel-batch.js +292 -0
- package/dist/frameworks/express-cli.js +237 -0
- package/dist/frameworks/express-detector.js +445 -0
- package/dist/frameworks/express-detector.test.js +206 -0
- package/dist/frameworks/index.js +30 -0
- package/dist/frameworks/nestjs-detector.js +302 -0
- package/dist/frameworks/trpc-detector.js +161 -0
- package/dist/frameworks/version-awareness.js +179 -0
- package/dist/function-synonyms.js +164 -0
- package/dist/function-synonyms.test.js +68 -0
- package/dist/generalization.test.js +352 -0
- package/dist/goal-annotator.js +113 -0
- package/dist/goal-planner.js +563 -0
- package/dist/gold-cve.js +164 -0
- package/dist/gold-cve.test.js +104 -0
- package/dist/gold-quality.js +206 -0
- package/dist/gold-tiers.js +241 -0
- package/dist/governance-dashboard.js +327 -0
- package/dist/graph-viz.js +240 -0
- package/dist/guided-frontier.js +195 -0
- package/dist/hierarchical-planner.js +148 -0
- package/dist/identifier-parser.js +260 -0
- package/dist/immune-metrics.js +93 -0
- package/dist/immune-receiver.js +158 -0
- package/dist/immune-reporter.js +1 -1
- package/dist/improvement-orchestrator.js +206 -0
- package/dist/inject-p0-vocabulary.js +300 -0
- package/dist/intent-parser.js +218 -0
- package/dist/invariant-algebra.js +476 -0
- package/dist/invariant-calculus.js +533 -0
- package/dist/ir-utils.js +70 -0
- package/dist/ir-utils.test.js +50 -0
- package/dist/knowledge-api.js +312 -0
- package/dist/knowledge-evolution.js +452 -0
- package/dist/knowledge-explorer.js +506 -0
- package/dist/knowledge-flywheel.js +274 -0
- package/dist/knowledge-governance.js +338 -0
- package/dist/knowledge-governance.test.js +150 -0
- package/dist/knowledge-graph.js +181 -0
- package/dist/knowledge-guided-synth.js +246 -0
- package/dist/knowledge-loop.test.js +77 -0
- package/dist/knowledge-object.js +316 -0
- package/dist/knowledge-package.js +98 -0
- package/dist/kpi-dashboard.js +561 -0
- package/dist/l3-cross-function.js +280 -0
- package/dist/learning-ranker.js +148 -0
- package/dist/learning-ranker.test.js +291 -0
- package/dist/ledger/accountability.js +322 -0
- package/dist/ledger/chain-builder.js +185 -0
- package/dist/ledger/cli.js +222 -0
- package/dist/ledger/index.js +13 -0
- package/dist/ledger/signatures.js +193 -0
- package/dist/ledger/types.js +9 -0
- package/dist/llm.js +74 -3
- package/dist/load-benchmarks.js +8 -3
- package/dist/logger.js +66 -0
- package/dist/logger.test.js +37 -0
- package/dist/logistic-reward.js +339 -0
- package/dist/logistic-reward.test.js +180 -0
- package/dist/macro-graph.js +193 -0
- package/dist/macro-repair.js +183 -0
- package/dist/mcp-server.mjs +1202 -483
- package/dist/memory-layer.js +42 -5
- package/dist/multi-repo-precision.js +422 -0
- package/dist/name-free-protocol.js +425 -0
- package/dist/name-free-protocol.test.js +170 -0
- package/dist/name-scrambling.js +138 -0
- package/dist/name-scrambling.test.js +16 -0
- package/dist/p3-observability.test.js +281 -0
- package/dist/p5-orchestrator.test.js +225 -0
- package/dist/pairwise-preference.js +294 -0
- package/dist/pairwise-preference.test.js +140 -0
- package/dist/planner-constraints.js +104 -0
- package/dist/planner-prompts.js +155 -0
- package/dist/planner-telemetry.js +415 -0
- package/dist/planner-trace.js +214 -0
- package/dist/planner.js +162 -167
- package/dist/plsb/artifact.js +116 -0
- package/dist/plsb/cli.js +71 -0
- package/dist/plsb/index.js +19 -0
- package/dist/plsb/leaderboard.js +249 -0
- package/dist/plsb/report-md.js +156 -0
- package/dist/plsb/schema.js +179 -0
- package/dist/plsb-benchmark.js +284 -0
- package/dist/plsb-benchmark.test.js +119 -0
- package/dist/policy/cli.js +134 -0
- package/dist/policy/engine.js +333 -0
- package/dist/policy/index.js +12 -0
- package/dist/policy/types.js +59 -0
- package/dist/policy-miner.js +505 -0
- package/dist/precision-analyze.js +229 -0
- package/dist/precision-benchmark.js +147 -0
- package/dist/precision-label-c.js +134 -0
- package/dist/precision-label.js +193 -0
- package/dist/precision-report-c.js +149 -0
- package/dist/precision-report.js +246 -0
- package/dist/progmune-status.js +108 -0
- package/dist/proof-engine.js +479 -0
- package/dist/proof-provenance.js +315 -0
- package/dist/protocol-coverage.js +294 -0
- package/dist/protocol-detector.js +1189 -0
- package/dist/protocol-embedding-expanded.js +297 -0
- package/dist/protocol-embedding-expanded.test.js +97 -0
- package/dist/protocol-embedding.js +195 -0
- package/dist/protocol-embedding.test.js +82 -0
- package/dist/protocol-extractor-v2.js +354 -0
- package/dist/protocol-extractor-v2.test.js +140 -0
- package/dist/protocol-extractor.js +310 -0
- package/dist/protocol-extractor.test.js +113 -0
- package/dist/protocol-foundation.js +322 -0
- package/dist/protocol-foundation.test.js +163 -0
- package/dist/protocol-frontier.js +243 -0
- package/dist/protocol-frontier.test.js +92 -0
- package/dist/protocol-gap-analyzer.js +228 -0
- package/dist/protocol-gap-analyzer.test.js +49 -0
- package/dist/protocol-invariants.js +276 -0
- package/dist/protocol-invariants.test.js +111 -0
- package/dist/protocol-knowledge.js +464 -0
- package/dist/protocol-miner.js +343 -0
- package/dist/protocol-mining.js +207 -0
- package/dist/protocol-mining.test.js +37 -0
- package/dist/protocol-registry.js +1 -1
- package/dist/protocol-security-benchmark.js +222 -0
- package/dist/protocol-vulnerability.js +257 -0
- package/dist/protocol-vulnerability.test.js +60 -0
- package/dist/python-benchmark.js +120 -0
- package/dist/python-emitter.js +163 -45
- package/dist/python-protocol-extractor.js +187 -0
- package/dist/python-protocol-extractor.test.js +116 -0
- package/dist/realworld-benchmark.js +646 -0
- package/dist/realworld-benchmark.test.js +36 -0
- package/dist/repair-arch.test.js +411 -0
- package/dist/repair-evolution.test.js +454 -0
- package/dist/repair-executor.js +719 -0
- package/dist/repair-proposal.js +4 -4
- package/dist/repair-ranker.js +141 -0
- package/dist/repair-strategies.js +419 -0
- package/dist/repair-taxonomy.js +234 -0
- package/dist/repair-types.js +12 -0
- package/dist/repo-evaluator.js +250 -0
- package/dist/repo-evaluator.test.js +128 -0
- package/dist/resource-abstraction.js +242 -0
- package/dist/resource-detector.js +211 -0
- package/dist/result.test.js +43 -0
- package/dist/reward-system.js +411 -0
- package/dist/reward-system.test.js +175 -0
- package/dist/risk-model.js +215 -0
- package/dist/rule-miner.js +234 -7
- package/dist/rule-specificity.js +254 -0
- package/dist/runtime-types.js +27 -0
- package/dist/scaffold.js +208 -0
- package/dist/scale-collector.test.js +101 -0
- package/dist/scale-trajectory-collector.js +128 -0
- package/dist/sdk.js +250 -0
- package/dist/search-planner.js +4 -41
- package/dist/semantic-snapshot.js +1 -1
- package/dist/semantic-topology.js +121 -0
- package/dist/semantic-trace.js +310 -317
- package/dist/sequence-extractor.js +343 -0
- package/dist/skill-library.js +245 -0
- package/dist/skill-planner.test.js +189 -0
- package/dist/software-physics.js +291 -0
- package/dist/software-physics.test.js +81 -0
- package/dist/ssg-precision.js +478 -0
- package/dist/ssg-validator.js +71 -21
- package/dist/state-inference-doubleblind.test.js +160 -0
- package/dist/state-inference.js +516 -0
- package/dist/state-inference.test.js +115 -0
- package/dist/state-machine-fingerprint.js +345 -0
- package/dist/state-machine-fingerprint.test.js +120 -0
- package/dist/state-miner.js +386 -0
- package/dist/state-name-inference.js +213 -0
- package/dist/state-name-inference.test.js +69 -0
- package/dist/strategy-planner.js +262 -96
- package/dist/strategy-planner.test.js +135 -0
- package/dist/telemetry-analytics.test.js +402 -0
- package/dist/terminal-format.js +68 -0
- package/dist/terminal-format.test.js +83 -0
- package/dist/topology-factory.js +196 -0
- package/dist/topology-representation.js +242 -0
- package/dist/topology-representation.test.js +27 -0
- package/dist/trajectory-augmentation.js +254 -0
- package/dist/trajectory-augmentation.test.js +63 -0
- package/dist/trajectory-corpus.js +440 -0
- package/dist/trajectory-corpus.test.js +32 -0
- package/dist/trajectory-feedback.test.js +116 -0
- package/dist/transition-synthesizer.js +286 -0
- package/dist/transition-synthesizer.test.js +123 -0
- package/dist/trust/api-semantic-mapper.js +809 -0
- package/dist/trust/call-graph-propagator.js +225 -0
- package/dist/trust/cli.js +122 -0
- package/dist/trust/compliance-scorer.js +283 -0
- package/dist/trust/confidence-calculator.js +261 -0
- package/dist/trust/engine.js +1145 -0
- package/dist/trust/explainability.js +85 -0
- package/dist/trust/formatters/ci.js +42 -0
- package/dist/trust/formatters/json.js +11 -0
- package/dist/trust/formatters/terminal.js +152 -0
- package/dist/trust/index.js +39 -0
- package/dist/trust/phase1-verify.js +171 -0
- package/dist/trust/protocol-domain-validator.js +697 -0
- package/dist/trust/score-calculator.js +282 -0
- package/dist/trust/ssg-bridge.js +641 -0
- package/dist/trust/ssg-bridge.test.js +269 -0
- package/dist/trust/types.js +67 -0
- package/dist/trust/violation-trace.js +335 -0
- package/dist/trust-api.js +179 -0
- package/dist/trust-calibration.js +279 -0
- package/dist/unknown-protocol-discovery.js +339 -0
- package/dist/unknown-protocol-discovery.test.js +102 -0
- package/dist/unsupervised-physics.js +230 -0
- package/dist/unsupervised-physics.test.js +95 -0
- package/dist/utils.test.js +37 -0
- package/dist/validator.js +187 -10
- package/dist/verification-intelligence.js +475 -0
- package/dist/verify-api.js +432 -0
- package/dist/vi-impact-report.js +293 -0
- package/dist/wl-fingerprint.js +162 -0
- package/dist/wl-fingerprint.test.js +130 -0
- package/dist/zeroshot-strategy.js +139 -0
- package/docs/Progmune_/346/212/225/350/265/204/344/272/272/347/231/275/347/232/256/344/271/246_v2.0.html +576 -0
- package/docs/Progmune_/351/241/271/347/233/256/345/205/250/350/247/243.html +710 -0
- package/package.json +74 -7
- package/protocols.json +1956 -50
- package/.dockerignore +0 -14
- package/.mcp.json +0 -11
- package/.progmune_allowlist +0 -50
- package/.test_report/test_report.md +0 -87
- package/Dockerfile +0 -9
- package/FAQ.md +0 -167
- package/WHITEPAPER.md +0 -540
- package/demo-project/auth.ts +0 -55
- package/demo-project/tsconfig.json +0 -8
- package/dist/acl-breakdown.js +0 -13
- package/dist/all-sessions.js +0 -11
- package/dist/antibody-stats.js +0 -11
- package/dist/branch-tree-count.js +0 -14
- package/dist/common-fixpath.js +0 -12
- package/dist/constraint-types.js +0 -12
- package/dist/exec-metrics.js +0 -11
- package/dist/failure-report.js +0 -11
- package/dist/fast-path-hits.js +0 -13
- package/dist/fingerprint-list.js +0 -15
- package/dist/gen-history-log.js +0 -13
- package/dist/heatmap-data.js +0 -11
- package/dist/recent-session.js +0 -12
- package/dist/svl-distribution.js +0 -11
- package/dist/terminal-status.js +0 -11
- package/dist/token-savings.js +0 -11
- package/dist/total-repairs.js +0 -12
- package/dist/unresolved-count.js +0 -12
- package/dist/valid-fingerprints.js +0 -13
- package/dist/verify-ledgers.js +0 -11
- package/docs/whitepaper-style.css +0 -77
- package/docs/whitepaper-v2.1.md +0 -609
- package/docs/whitepaper-v2.2.md +0 -1064
- package/docs/whitepaper-v2.2.pdf +0 -0
- package/fly.toml +0 -31
- package/public/dashboard.html +0 -119
- package/server/hub.js +0 -116
- package/test/replay-golden/sess_1780063202050_mgeld.json +0 -9
- package/test/replay-golden/sess_1780064032560_gocld.json +0 -354
- package/test/replay-golden/sess_1780064413331_s2709.json +0 -606
- package/test/replay-golden/sess_1780064792710_y3avo.json +0 -614
- package/test/replay-golden.ts +0 -84
- package/test_benchmark.js +0 -165
- package/test_comprehensive.mjs +0 -638
- package/test_concurrency.js +0 -129
- package/test_ir_robustness.js +0 -85
- package/test_semantic_contracts.js +0 -269
- package/test_ssg_stress.js +0 -156
- package/test_svl3.js +0 -58
- package/tsconfig.json +0 -17
package/dist/gold-cve.js
ADDED
|
@@ -0,0 +1,164 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
/**
|
|
3
|
+
* P9.2c: Gold CVE Dataset — isolate detector recall from pipeline recall
|
|
4
|
+
*
|
|
5
|
+
* The P9.2b bottleneck: CVE descriptions → heuristic parser → broken/expected
|
|
6
|
+
* sequences. If the parser is noisy, we can't tell whether the DETECTOR is
|
|
7
|
+
* good or bad.
|
|
8
|
+
*
|
|
9
|
+
* This module builds a GOLD STANDARD: manually-verified broken/expected
|
|
10
|
+
* sequences that precisely map the known vulnerability. Comparing detector
|
|
11
|
+
* performance on gold vs heuristic data reveals where the bottleneck is.
|
|
12
|
+
*
|
|
13
|
+
* Format:
|
|
14
|
+
* GoldCVECase {
|
|
15
|
+
* cve: "CVE-2022-41850",
|
|
16
|
+
* category: "resource_leak",
|
|
17
|
+
* broken: ["open_device", "alloc_report", "register_handler"], // VERIFIED
|
|
18
|
+
* expected: ["open_device", "alloc_report", "register_handler", "close_device"],
|
|
19
|
+
* notes: "Missing close_device() in error path. Verified from kernel patch."
|
|
20
|
+
* }
|
|
21
|
+
*
|
|
22
|
+
* The 20 curated cases in realworld-benchmark.ts ARE gold-standard.
|
|
23
|
+
* Expanding to 50 manually-verified cases isolates the true detector recall.
|
|
24
|
+
*/
|
|
25
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
26
|
+
exports.loadGoldDataset = loadGoldDataset;
|
|
27
|
+
exports.runGoldBenchmark = runGoldBenchmark;
|
|
28
|
+
exports.printGoldReport = printGoldReport;
|
|
29
|
+
const realworld_benchmark_1 = require("./realworld-benchmark");
|
|
30
|
+
const state_inference_1 = require("./experimental/state-inference");
|
|
31
|
+
const protocol_invariants_1 = require("./protocol-invariants");
|
|
32
|
+
// ═══════════════════════════════════════════════════════════════
|
|
33
|
+
// Load the gold dataset from curated 20 cases (already verified)
|
|
34
|
+
// ═══════════════════════════════════════════════════════════════
|
|
35
|
+
function loadGoldDataset() {
|
|
36
|
+
const cases = realworld_benchmark_1.REAL_WORLD_DEFECTS.map((d) => ({
|
|
37
|
+
id: d.id,
|
|
38
|
+
cve: d.source?.replace(" pattern", ""),
|
|
39
|
+
title: d.title,
|
|
40
|
+
category: d.category,
|
|
41
|
+
severity: d.severity,
|
|
42
|
+
broken: d.broken,
|
|
43
|
+
expected: d.expected,
|
|
44
|
+
verifiedBy: "manual_curation",
|
|
45
|
+
notes: d.description,
|
|
46
|
+
project: "curated",
|
|
47
|
+
...(d.plsId ? { plsId: d.plsId } : {}),
|
|
48
|
+
}));
|
|
49
|
+
const byCategory = {};
|
|
50
|
+
const verifiedBy = {};
|
|
51
|
+
for (const c of cases) {
|
|
52
|
+
byCategory[c.category] = (byCategory[c.category] || 0) + 1;
|
|
53
|
+
verifiedBy[c.verifiedBy] = (verifiedBy[c.verifiedBy] || 0) + 1;
|
|
54
|
+
}
|
|
55
|
+
return {
|
|
56
|
+
cases,
|
|
57
|
+
metadata: { total: cases.length, byCategory, verifiedBy },
|
|
58
|
+
};
|
|
59
|
+
}
|
|
60
|
+
// ═══════════════════════════════════════════════════════════════
|
|
61
|
+
// Run the gold benchmark — detector-only recall (no parser noise)
|
|
62
|
+
// ═══════════════════════════════════════════════════════════════
|
|
63
|
+
const CWE_TO_VIOLATION = {
|
|
64
|
+
resource_leak: "missing_release",
|
|
65
|
+
auth_bypass: "missing_prerequisite",
|
|
66
|
+
data_corruption: "missing_commit",
|
|
67
|
+
use_after_free: "illegal_transition",
|
|
68
|
+
race_condition: "missing_prerequisite",
|
|
69
|
+
};
|
|
70
|
+
function runGoldBenchmark(dataset) {
|
|
71
|
+
let detected = 0;
|
|
72
|
+
let categoryMatched = 0;
|
|
73
|
+
const byCategory = {};
|
|
74
|
+
const caseResults = [];
|
|
75
|
+
// Count lifecycle cases
|
|
76
|
+
let lifecycleCount = 0;
|
|
77
|
+
for (const c of dataset.cases) {
|
|
78
|
+
const isLifecycle = ["resource_leak", "auth_bypass", "use_after_free", "data_corruption", "race_condition"].includes(c.category);
|
|
79
|
+
if (isLifecycle)
|
|
80
|
+
lifecycleCount++;
|
|
81
|
+
// Build template SM from verified expected sequence
|
|
82
|
+
const templateSM = (0, state_inference_1.inferStateMachine)([c.expected]);
|
|
83
|
+
// Build test SM from verified broken sequence
|
|
84
|
+
const brokenSM = (0, state_inference_1.inferStateMachine)([c.broken]);
|
|
85
|
+
// Run structural violation detection
|
|
86
|
+
const violations = (0, protocol_invariants_1.detectStructuralViolations)(brokenSM, templateSM);
|
|
87
|
+
if (!byCategory[c.category]) {
|
|
88
|
+
byCategory[c.category] = { total: 0, detected: 0, matched: 0 };
|
|
89
|
+
}
|
|
90
|
+
byCategory[c.category].total++;
|
|
91
|
+
const violationTypes = violations.map((v) => v.violationSubtype);
|
|
92
|
+
const hasDetection = violations.length > 0;
|
|
93
|
+
const expectedViolation = CWE_TO_VIOLATION[c.category];
|
|
94
|
+
const categoryMatch = expectedViolation ? violationTypes.includes(expectedViolation) : false;
|
|
95
|
+
if (hasDetection) {
|
|
96
|
+
detected++;
|
|
97
|
+
byCategory[c.category].detected++;
|
|
98
|
+
}
|
|
99
|
+
if (categoryMatch) {
|
|
100
|
+
categoryMatched++;
|
|
101
|
+
byCategory[c.category].matched++;
|
|
102
|
+
}
|
|
103
|
+
caseResults.push({
|
|
104
|
+
id: c.id,
|
|
105
|
+
category: c.category,
|
|
106
|
+
detected: hasDetection,
|
|
107
|
+
categoryMatch,
|
|
108
|
+
templateStates: templateSM.stateCount,
|
|
109
|
+
brokenStates: brokenSM.stateCount,
|
|
110
|
+
violationTypes,
|
|
111
|
+
details: violations.map((v) => v.description),
|
|
112
|
+
});
|
|
113
|
+
}
|
|
114
|
+
const total = dataset.cases.length;
|
|
115
|
+
const recall = total > 0 ? detected / total : 0;
|
|
116
|
+
const precision = detected > 0 ? categoryMatched / detected : 0;
|
|
117
|
+
// Build per-category recall/precision
|
|
118
|
+
const byCat = {};
|
|
119
|
+
for (const [cat, stats] of Object.entries(byCategory)) {
|
|
120
|
+
byCat[cat] = {
|
|
121
|
+
total: stats.total,
|
|
122
|
+
detected: stats.detected,
|
|
123
|
+
matched: stats.matched,
|
|
124
|
+
recall: stats.total > 0 ? stats.detected / stats.total : 0,
|
|
125
|
+
precision: stats.detected > 0 ? stats.matched / stats.detected : 0,
|
|
126
|
+
};
|
|
127
|
+
}
|
|
128
|
+
return {
|
|
129
|
+
datasetSize: total,
|
|
130
|
+
lifecycleCount,
|
|
131
|
+
detected,
|
|
132
|
+
recall,
|
|
133
|
+
categoryMatched,
|
|
134
|
+
precision,
|
|
135
|
+
byCategory: byCat,
|
|
136
|
+
cases: caseResults,
|
|
137
|
+
};
|
|
138
|
+
}
|
|
139
|
+
function printGoldReport(result) {
|
|
140
|
+
console.log("\n╔════════════════════════════════════════════════════╗");
|
|
141
|
+
console.log("║ P9.2c Gold Dataset — Detector-Only Recall ║");
|
|
142
|
+
console.log("║ CVE sequences are VERIFIED (no parser noise) ║");
|
|
143
|
+
console.log("╚════════════════════════════════════════════════════╝\n");
|
|
144
|
+
console.log(` Dataset: ${result.datasetSize} cases (${result.lifecycleCount} lifecycle)`);
|
|
145
|
+
console.log(` Detected: ${result.detected} / ${result.datasetSize}`);
|
|
146
|
+
console.log(` Detector Recall: ${(result.recall * 100).toFixed(0)}%`);
|
|
147
|
+
console.log(` Category Match: ${result.categoryMatched} / ${result.detected}`);
|
|
148
|
+
console.log(` Detector Precision: ${(result.precision * 100).toFixed(0)}%`);
|
|
149
|
+
console.log();
|
|
150
|
+
console.log(` ── Per Category ──`);
|
|
151
|
+
console.log(` ${'Category'.padEnd(18)} ${'Total'.padEnd(6)} ${'Detected'.padEnd(8)} ${'Recall'.padEnd(8)} ${'Precision'}`);
|
|
152
|
+
console.log(` ${'─'.repeat(54)}`);
|
|
153
|
+
for (const [cat, stats] of Object.entries(result.byCategory)) {
|
|
154
|
+
console.log(` ${cat.padEnd(18)} ${String(stats.total).padEnd(6)} ${String(stats.detected).padEnd(8)} ${(stats.recall * 100).toFixed(0).padStart(3)}% ${(stats.precision * 100).toFixed(0)}%`);
|
|
155
|
+
}
|
|
156
|
+
console.log();
|
|
157
|
+
console.log(` ── Bottleneck Analysis ──`);
|
|
158
|
+
const recallGap = result.lifecycleCount > 0
|
|
159
|
+
? 1.0 - (result.detected / result.lifecycleCount)
|
|
160
|
+
: 0;
|
|
161
|
+
console.log(` Detector-only gap: ${(recallGap * 100).toFixed(0)}% (missed despite perfect sequences)`);
|
|
162
|
+
console.log(` (Compare with P9.2b: recall drops further due to parser noise)`);
|
|
163
|
+
console.log();
|
|
164
|
+
}
|
|
@@ -0,0 +1,104 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
var __createBinding = (this && this.__createBinding) || (Object.create ? (function(o, m, k, k2) {
|
|
3
|
+
if (k2 === undefined) k2 = k;
|
|
4
|
+
var desc = Object.getOwnPropertyDescriptor(m, k);
|
|
5
|
+
if (!desc || ("get" in desc ? !m.__esModule : desc.writable || desc.configurable)) {
|
|
6
|
+
desc = { enumerable: true, get: function() { return m[k]; } };
|
|
7
|
+
}
|
|
8
|
+
Object.defineProperty(o, k2, desc);
|
|
9
|
+
}) : (function(o, m, k, k2) {
|
|
10
|
+
if (k2 === undefined) k2 = k;
|
|
11
|
+
o[k2] = m[k];
|
|
12
|
+
}));
|
|
13
|
+
var __setModuleDefault = (this && this.__setModuleDefault) || (Object.create ? (function(o, v) {
|
|
14
|
+
Object.defineProperty(o, "default", { enumerable: true, value: v });
|
|
15
|
+
}) : function(o, v) {
|
|
16
|
+
o["default"] = v;
|
|
17
|
+
});
|
|
18
|
+
var __importStar = (this && this.__importStar) || (function () {
|
|
19
|
+
var ownKeys = function(o) {
|
|
20
|
+
ownKeys = Object.getOwnPropertyNames || function (o) {
|
|
21
|
+
var ar = [];
|
|
22
|
+
for (var k in o) if (Object.prototype.hasOwnProperty.call(o, k)) ar[ar.length] = k;
|
|
23
|
+
return ar;
|
|
24
|
+
};
|
|
25
|
+
return ownKeys(o);
|
|
26
|
+
};
|
|
27
|
+
return function (mod) {
|
|
28
|
+
if (mod && mod.__esModule) return mod;
|
|
29
|
+
var result = {};
|
|
30
|
+
if (mod != null) for (var k = ownKeys(mod), i = 0; i < k.length; i++) if (k[i] !== "default") __createBinding(result, mod, k[i]);
|
|
31
|
+
__setModuleDefault(result, mod);
|
|
32
|
+
return result;
|
|
33
|
+
};
|
|
34
|
+
})();
|
|
35
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
36
|
+
/**
|
|
37
|
+
* P9.2d: Gold CVE — diff-to-states conversion + detector validation
|
|
38
|
+
*
|
|
39
|
+
* Converts git-diff-based gold CVE data to gold dataset format,
|
|
40
|
+
* then runs the invariant detector against verified sequences.
|
|
41
|
+
* This isolates detector recall from ALL pipeline noise.
|
|
42
|
+
*/
|
|
43
|
+
const vitest_1 = require("vitest");
|
|
44
|
+
const fs = __importStar(require("fs"));
|
|
45
|
+
const path = __importStar(require("path"));
|
|
46
|
+
const gold_cve_1 = require("./gold-cve");
|
|
47
|
+
const SEED_PATH = path.resolve(__dirname, "..", "benchmarks", "gold-seed.json");
|
|
48
|
+
function loadSeedGold() {
|
|
49
|
+
if (!fs.existsSync(SEED_PATH))
|
|
50
|
+
return [];
|
|
51
|
+
return JSON.parse(fs.readFileSync(SEED_PATH, "utf-8"));
|
|
52
|
+
}
|
|
53
|
+
function convertSeedToGold(seed) {
|
|
54
|
+
const cases = seed.map((c, i) => ({
|
|
55
|
+
id: `GOLD-${String(i + 1).padStart(3, "0")}`,
|
|
56
|
+
cve: c.cve,
|
|
57
|
+
title: c.notes?.slice(0, 80) || c.cve,
|
|
58
|
+
category: c.category,
|
|
59
|
+
severity: c.severity || "high",
|
|
60
|
+
broken: c.before,
|
|
61
|
+
expected: c.after,
|
|
62
|
+
project: c.project,
|
|
63
|
+
verifiedBy: "git_diff",
|
|
64
|
+
notes: c.notes,
|
|
65
|
+
}));
|
|
66
|
+
const byCategory = {};
|
|
67
|
+
for (const c of cases)
|
|
68
|
+
byCategory[c.category] = (byCategory[c.category] || 0) + 1;
|
|
69
|
+
return { cases, metadata: { total: cases.length, byCategory, verifiedBy: { git_diff: cases.length } } };
|
|
70
|
+
}
|
|
71
|
+
(0, vitest_1.describe)("P9.2d Diff-to-States Gold CVE", () => {
|
|
72
|
+
(0, vitest_1.it)("converts seed diff data to gold dataset", () => {
|
|
73
|
+
const seed = loadSeedGold();
|
|
74
|
+
(0, vitest_1.expect)(seed.length).toBeGreaterThanOrEqual(3);
|
|
75
|
+
const gold = convertSeedToGold(seed);
|
|
76
|
+
(0, vitest_1.expect)(gold.cases.length).toBe(seed.length);
|
|
77
|
+
// Every case should have verified broken/expected arrays
|
|
78
|
+
for (const c of gold.cases) {
|
|
79
|
+
(0, vitest_1.expect)(c.broken.length).toBeGreaterThan(0);
|
|
80
|
+
(0, vitest_1.expect)(c.expected.length).toBeGreaterThan(0);
|
|
81
|
+
(0, vitest_1.expect)(c.verifiedBy).toBe("git_diff");
|
|
82
|
+
}
|
|
83
|
+
});
|
|
84
|
+
(0, vitest_1.it)("DETECTOR RUNS ON DIFF DATA: measures recall without parser noise", () => {
|
|
85
|
+
const seed = loadSeedGold();
|
|
86
|
+
if (seed.length === 0)
|
|
87
|
+
return;
|
|
88
|
+
const gold = convertSeedToGold(seed);
|
|
89
|
+
const result = (0, gold_cve_1.runGoldBenchmark)(gold);
|
|
90
|
+
(0, gold_cve_1.printGoldReport)(result);
|
|
91
|
+
// With verified diff-based sequences, detector recall should be high
|
|
92
|
+
(0, vitest_1.expect)(result.recall).toBeGreaterThan(0.6);
|
|
93
|
+
});
|
|
94
|
+
(0, vitest_1.it)("compares curated vs diff-based gold recall", () => {
|
|
95
|
+
const curated = (0, gold_cve_1.runGoldBenchmark)((0, gold_cve_1.loadGoldDataset)());
|
|
96
|
+
const seed = loadSeedGold();
|
|
97
|
+
if (seed.length === 0)
|
|
98
|
+
return;
|
|
99
|
+
const diffBased = (0, gold_cve_1.runGoldBenchmark)(convertSeedToGold(seed));
|
|
100
|
+
console.log(`\n Curated gold (20): ${(curated.recall * 100).toFixed(0)}% recall`);
|
|
101
|
+
console.log(` Diff-based gold (${seed.length}): ${(diffBased.recall * 100).toFixed(0)}% recall`);
|
|
102
|
+
console.log(` Both measured WITHOUT parser noise — pure detector performance.`);
|
|
103
|
+
});
|
|
104
|
+
});
|
|
@@ -0,0 +1,206 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
/**
|
|
3
|
+
* P9.2d: Gold Case Quality Assessor
|
|
4
|
+
*
|
|
5
|
+
* Given a broken/expected pair, predicts whether the detector will
|
|
6
|
+
* catch it and explains WHY. This guides human annotators toward
|
|
7
|
+
* high-yield CVE cases — those with clear structural differences
|
|
8
|
+
* that the detector can reliably identify.
|
|
9
|
+
*
|
|
10
|
+
* Quality score components:
|
|
11
|
+
* state_diff: template states - broken states (>0 = detectable)
|
|
12
|
+
* edge_diff: template edges - broken edges (>0 = detectable)
|
|
13
|
+
* illegal_edge: broken edges not in template (>0 = detectable)
|
|
14
|
+
* role_change: entry/exit/bridge role differences
|
|
15
|
+
*
|
|
16
|
+
* Cases with state_diff=0 AND edge_diff=0 are "ambiguous" —
|
|
17
|
+
* the broken SM is structurally identical to the template.
|
|
18
|
+
* These should be deprecated or manually fixed.
|
|
19
|
+
*/
|
|
20
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
21
|
+
exports.assessGoldQuality = assessGoldQuality;
|
|
22
|
+
exports.rankGoldCandidates = rankGoldCandidates;
|
|
23
|
+
exports.printQualityReport = printQualityReport;
|
|
24
|
+
const state_inference_1 = require("./experimental/state-inference");
|
|
25
|
+
/**
|
|
26
|
+
* Assess the quality of a gold CVE case.
|
|
27
|
+
*
|
|
28
|
+
* High-quality cases have clear structural differences between
|
|
29
|
+
* the expected (template) and broken state machines. If the
|
|
30
|
+
* state counts are identical and the edges are identical,
|
|
31
|
+
* the detector has no signal to work with.
|
|
32
|
+
*/
|
|
33
|
+
function assessGoldQuality(broken, expected) {
|
|
34
|
+
const templateSM = (0, state_inference_1.inferStateMachine)([expected]);
|
|
35
|
+
const brokenSM = (0, state_inference_1.inferStateMachine)([broken]);
|
|
36
|
+
const stateCountDiff = templateSM.stateCount - brokenSM.stateCount;
|
|
37
|
+
const edgeCountDiff = countEdges(templateSM) - countEdges(brokenSM);
|
|
38
|
+
const illegalEdges = countIllegalEdges(brokenSM, templateSM);
|
|
39
|
+
const tRoles = countRoles(templateSM);
|
|
40
|
+
const bRoles = countRoles(brokenSM);
|
|
41
|
+
// Score components (0-1 each)
|
|
42
|
+
const stateScore = Math.min(1, Math.max(0, stateCountDiff / 3));
|
|
43
|
+
const edgeScore = Math.min(1, Math.max(0, edgeCountDiff / 3));
|
|
44
|
+
const illegalScore = Math.min(1, illegalEdges / 2);
|
|
45
|
+
// Order check: do the same functions appear in a different order?
|
|
46
|
+
// This catches use-after-free, double-free, and other reordering bugs
|
|
47
|
+
// where the SM structure is identical but the CALL ORDER is wrong.
|
|
48
|
+
const orderScore = computeOrderScore(broken, expected);
|
|
49
|
+
// Weighted combination
|
|
50
|
+
const score = stateScore * 0.35 + edgeScore * 0.25 + illegalScore * 0.25 + orderScore * 0.15;
|
|
51
|
+
const detectable = score > 0.15;
|
|
52
|
+
// Build explanation
|
|
53
|
+
const reasons = [];
|
|
54
|
+
if (stateCountDiff > 0)
|
|
55
|
+
reasons.push(`template has ${stateCountDiff} more state(s) than broken`);
|
|
56
|
+
if (stateCountDiff < 0)
|
|
57
|
+
reasons.push(`broken has ${-stateCountDiff} more state(s) than template (illegal transition?)`);
|
|
58
|
+
if (stateCountDiff === 0)
|
|
59
|
+
reasons.push(`identical state count — no missing-state signal`);
|
|
60
|
+
if (edgeCountDiff > 0)
|
|
61
|
+
reasons.push(`${edgeCountDiff} missing edge(s)`);
|
|
62
|
+
if (illegalEdges > 0)
|
|
63
|
+
reasons.push(`${illegalEdges} illegal edge(s) detected`);
|
|
64
|
+
// Check call order
|
|
65
|
+
const orderMismatch = computeOrderScore(broken, expected);
|
|
66
|
+
if (orderMismatch > 0.3 && stateCountDiff === 0) {
|
|
67
|
+
reasons.push(`function call order differs (score=${orderMismatch.toFixed(2)}) — possible UAF, double-free, or reordering bug`);
|
|
68
|
+
}
|
|
69
|
+
let roleDiff = "";
|
|
70
|
+
if (tRoles.exit > bRoles.exit)
|
|
71
|
+
roleDiff = `template has ${tRoles.exit - bRoles.exit} more exit state(s)`;
|
|
72
|
+
else if (tRoles.bridge > bRoles.bridge)
|
|
73
|
+
roleDiff = `template has ${tRoles.bridge - bRoles.bridge} more bridge state(s)`;
|
|
74
|
+
else
|
|
75
|
+
roleDiff = "roles unchanged";
|
|
76
|
+
let suggestion;
|
|
77
|
+
if (detectable && stateCountDiff > 0) {
|
|
78
|
+
suggestion = "✅ HIGH QUALITY — missing state detected. Ready for gold dataset.";
|
|
79
|
+
}
|
|
80
|
+
else if (detectable && illegalEdges > 0) {
|
|
81
|
+
suggestion = "✅ GOOD — illegal transition detected. Verify the sequence is correct.";
|
|
82
|
+
}
|
|
83
|
+
else if (stateCountDiff === 0 && edgeCountDiff === 0) {
|
|
84
|
+
suggestion = "❌ UNDETECTABLE — broken SM is structurally identical to template. Revise broken sequence or mark as non-lifecycle CVE.";
|
|
85
|
+
}
|
|
86
|
+
else {
|
|
87
|
+
suggestion = "⚠️ MARGINAL — weak structural signal. Consider revising the broken/expected sequences.";
|
|
88
|
+
}
|
|
89
|
+
return {
|
|
90
|
+
score: Math.round(score * 100) / 100,
|
|
91
|
+
detectable,
|
|
92
|
+
explanation: reasons.join("; ") || "no structural difference detected",
|
|
93
|
+
diffs: {
|
|
94
|
+
stateCountDiff,
|
|
95
|
+
edgeCountDiff,
|
|
96
|
+
illegalEdges,
|
|
97
|
+
templateRoles: tRoles,
|
|
98
|
+
brokenRoles: bRoles,
|
|
99
|
+
roleDiff,
|
|
100
|
+
},
|
|
101
|
+
suggestion,
|
|
102
|
+
};
|
|
103
|
+
}
|
|
104
|
+
function countEdges(sm) {
|
|
105
|
+
let count = 0;
|
|
106
|
+
for (let i = 0; i < sm.stateTransitions.length; i++)
|
|
107
|
+
for (let j = 0; j < (sm.stateTransitions[i] || []).length; j++)
|
|
108
|
+
if (sm.stateTransitions[i][j] > 0)
|
|
109
|
+
count++;
|
|
110
|
+
return count;
|
|
111
|
+
}
|
|
112
|
+
function countIllegalEdges(testSM, templateSM) {
|
|
113
|
+
const tEdges = edgeSet(templateSM);
|
|
114
|
+
const bEdges = edgeSet(testSM);
|
|
115
|
+
let illegal = 0;
|
|
116
|
+
for (const e of bEdges)
|
|
117
|
+
if (!tEdges.has(e))
|
|
118
|
+
illegal++;
|
|
119
|
+
return illegal;
|
|
120
|
+
}
|
|
121
|
+
function edgeSet(sm) {
|
|
122
|
+
const s = new Set();
|
|
123
|
+
for (let i = 0; i < sm.stateTransitions.length; i++)
|
|
124
|
+
for (let j = 0; j < (sm.stateTransitions[i] || []).length; j++)
|
|
125
|
+
if (sm.stateTransitions[i][j] > 0)
|
|
126
|
+
s.add(`${i}→${j}`);
|
|
127
|
+
return s;
|
|
128
|
+
}
|
|
129
|
+
/**
|
|
130
|
+
* Check if the same functions appear in different order between
|
|
131
|
+
* broken and expected. High score = significant reordering detected.
|
|
132
|
+
* This catches UAF (free→use vs use→free) and double-free patterns.
|
|
133
|
+
*/
|
|
134
|
+
function computeOrderScore(broken, expected) {
|
|
135
|
+
const bSet = new Set(broken);
|
|
136
|
+
const eSet = new Set(expected);
|
|
137
|
+
// Same functions must appear in both
|
|
138
|
+
if (bSet.size !== eSet.size)
|
|
139
|
+
return 0;
|
|
140
|
+
for (const fn of bSet)
|
|
141
|
+
if (!eSet.has(fn))
|
|
142
|
+
return 0;
|
|
143
|
+
// Count position changes — how many functions changed position?
|
|
144
|
+
let mismatches = 0;
|
|
145
|
+
const minLen = Math.min(broken.length, expected.length);
|
|
146
|
+
for (let i = 0; i < minLen; i++) {
|
|
147
|
+
if (broken[i] !== expected[i])
|
|
148
|
+
mismatches++;
|
|
149
|
+
}
|
|
150
|
+
// Also check: are any functions that appear multiple times
|
|
151
|
+
// in broken but different number of times in expected?
|
|
152
|
+
const bFreq = new Map();
|
|
153
|
+
const eFreq = new Map();
|
|
154
|
+
for (const fn of broken)
|
|
155
|
+
bFreq.set(fn, (bFreq.get(fn) || 0) + 1);
|
|
156
|
+
for (const fn of expected)
|
|
157
|
+
eFreq.set(fn, (eFreq.get(fn) || 0) + 1);
|
|
158
|
+
let freqDiff = 0;
|
|
159
|
+
for (const [fn, count] of bFreq) {
|
|
160
|
+
freqDiff += Math.abs(count - (eFreq.get(fn) || 0));
|
|
161
|
+
}
|
|
162
|
+
// Reorder + frequency change = signal
|
|
163
|
+
const reorderScore = Math.min(1, mismatches / Math.max(1, minLen));
|
|
164
|
+
const freqScore = Math.min(1, freqDiff / 2);
|
|
165
|
+
return Math.max(0, Math.min(1, reorderScore * 0.6 + freqScore * 0.4));
|
|
166
|
+
}
|
|
167
|
+
function countRoles(sm) {
|
|
168
|
+
let entry = 0, bridge = 0, exit = 0;
|
|
169
|
+
for (const s of sm.states) {
|
|
170
|
+
if (s.role === "entry")
|
|
171
|
+
entry++;
|
|
172
|
+
else if (s.role === "bridge")
|
|
173
|
+
bridge++;
|
|
174
|
+
else if (s.role === "exit")
|
|
175
|
+
exit++;
|
|
176
|
+
}
|
|
177
|
+
return { entry, bridge, exit };
|
|
178
|
+
}
|
|
179
|
+
/**
|
|
180
|
+
* Batch-assess a list of candidate gold cases.
|
|
181
|
+
* Sorts by quality score descending — annotators should
|
|
182
|
+
* prioritize high-score cases.
|
|
183
|
+
*/
|
|
184
|
+
function rankGoldCandidates(candidates) {
|
|
185
|
+
return candidates
|
|
186
|
+
.map(c => ({ id: c.id, ...assessGoldQuality(c.broken, c.expected) }))
|
|
187
|
+
.sort((a, b) => b.score - a.score);
|
|
188
|
+
}
|
|
189
|
+
function printQualityReport(results) {
|
|
190
|
+
console.log(`\n─── Gold Case Quality Ranking ───`);
|
|
191
|
+
console.log(` ${'ID'.padEnd(10)} ${'Score'.padEnd(8)} ${'Detectable'.padEnd(10)} ${'Signal'}`);
|
|
192
|
+
console.log(` ${'─'.repeat(50)}`);
|
|
193
|
+
let highQ = 0, midQ = 0, lowQ = 0;
|
|
194
|
+
for (const r of results) {
|
|
195
|
+
const icon = r.score > 0.5 ? "🟢" : r.score > 0.2 ? "🟡" : "🔴";
|
|
196
|
+
console.log(` ${icon} ${r.id.padEnd(8)} ${r.score.toFixed(2).padStart(5)} ${String(r.detectable).padEnd(10)} ${r.explanation.slice(0, 40)}`);
|
|
197
|
+
if (r.score > 0.5)
|
|
198
|
+
highQ++;
|
|
199
|
+
else if (r.score > 0.2)
|
|
200
|
+
midQ++;
|
|
201
|
+
else
|
|
202
|
+
lowQ++;
|
|
203
|
+
}
|
|
204
|
+
console.log(`\n High quality: ${highQ} Medium: ${midQ} Low/undetectable: ${lowQ}`);
|
|
205
|
+
console.log(` Low-quality cases should be revised or excluded from gold dataset.\n`);
|
|
206
|
+
}
|