progmune-runtime 2.1.6 → 3.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +108 -468
- package/dist/ablation-study.js +144 -0
- package/dist/ablation-study.test.js +18 -0
- package/dist/action-runtime.js +3 -1
- package/dist/active-learning.js +211 -0
- package/dist/analytics.js +139 -0
- package/dist/asset-factory.js +309 -0
- package/dist/asset-growth.js +244 -0
- package/dist/asset-promotion.js +382 -0
- package/dist/asset-quality.js +550 -0
- package/dist/audit/business-translator.js +285 -0
- package/dist/audit/cli.js +66 -0
- package/dist/audit/formatters/html.js +379 -0
- package/dist/audit/formatters/json.js +11 -0
- package/dist/audit/formatters/markdown.js +192 -0
- package/dist/audit/formatters/terminal.js +189 -0
- package/dist/audit/index.js +25 -0
- package/dist/audit/report-builder.js +318 -0
- package/dist/audit/types.js +8 -0
- package/dist/audit.js +3 -3
- package/dist/auto-benchmark-generator.js +137 -0
- package/dist/auto-benchmark-generator.test.js +45 -0
- package/dist/auto-protocol-synthesizer.js +362 -0
- package/dist/auto-protocol-synthesizer.test.js +82 -0
- package/dist/autonomous-patch.js +175 -0
- package/dist/autonomous-patch.test.js +128 -0
- package/dist/badge/badge-server.js +98 -0
- package/dist/behavior-miner.js +442 -0
- package/dist/belief-layer.js +475 -0
- package/dist/benchmark-count.js +5 -0
- package/dist/benchmark-generator.js +211 -0
- package/dist/benchmark-harness.js +201 -0
- package/dist/benchmark-pass-rate.js +7 -0
- package/dist/benchmark-report.js +8 -3
- package/dist/benchmark-save.js +14 -1
- package/dist/bootstrap-validation.js +197 -0
- package/dist/bootstrap-validation.test.js +51 -0
- package/dist/branch-ledger.js +1 -1
- package/dist/capability-gap.js +130 -0
- package/dist/certify-html.js +351 -0
- package/dist/certify.js +326 -0
- package/dist/check.js +4 -4
- package/dist/compliance-miner.js +447 -0
- package/dist/continuous-benchmark.js +194 -0
- package/dist/continuous-benchmark.test.js +116 -0
- package/dist/corpus-stats.js +173 -0
- package/dist/counterfactual-engine.js +288 -0
- package/dist/coverage-dashboard.js +109 -0
- package/dist/coverage-system.test.js +205 -0
- package/dist/cross-repo-precision.js +352 -0
- package/dist/cve-benchmark.js +180 -0
- package/dist/cve-benchmark.test.js +28 -0
- package/dist/cve-collector.js +73 -0
- package/dist/data-quality.js +141 -0
- package/dist/decision-engine.js +388 -0
- package/dist/derive-metadata.js +250 -0
- package/dist/difficulty-active.test.js +198 -0
- package/dist/difficulty-map.js +244 -0
- package/dist/discovery-analytics.js +125 -0
- package/dist/discovery-model.js +149 -0
- package/dist/discovery-optimize.test.js +199 -0
- package/dist/discovery-trace.js +276 -0
- package/dist/discovery-trace.test.js +97 -0
- package/dist/emitter.js +83 -1
- package/dist/enterprise-dashboard.js +405 -0
- package/dist/eval-hardening.js +297 -0
- package/dist/eval-hardening.test.js +85 -0
- package/dist/evaluation-campaign.js +359 -0
- package/dist/evaluation-campaign.test.js +181 -0
- package/dist/evidence-growth.js +143 -0
- package/dist/evidence-repository.js +209 -0
- package/dist/evidence-system.js +441 -0
- package/dist/execute.js +15 -7
- package/dist/experimental/software-physics.js +291 -0
- package/dist/experimental/state-inference.js +516 -0
- package/dist/experimental/unsupervised-physics.js +230 -0
- package/dist/extract-ir-python.js +54 -7
- package/dist/extract-ir.js +376 -12
- package/dist/failure-collector.js +2 -2
- package/dist/failure-corpus.js +322 -9
- package/dist/feedback.js +16 -5
- package/dist/feedback.test.js +49 -0
- package/dist/file-lock.js +1 -1
- package/dist/flywheel-batch.js +292 -0
- package/dist/frameworks/express-cli.js +237 -0
- package/dist/frameworks/express-detector.js +445 -0
- package/dist/frameworks/express-detector.test.js +206 -0
- package/dist/frameworks/index.js +30 -0
- package/dist/frameworks/nestjs-detector.js +302 -0
- package/dist/frameworks/trpc-detector.js +161 -0
- package/dist/frameworks/version-awareness.js +179 -0
- package/dist/function-synonyms.js +164 -0
- package/dist/function-synonyms.test.js +68 -0
- package/dist/generalization.test.js +352 -0
- package/dist/goal-annotator.js +113 -0
- package/dist/goal-planner.js +563 -0
- package/dist/gold-cve.js +164 -0
- package/dist/gold-cve.test.js +104 -0
- package/dist/gold-quality.js +206 -0
- package/dist/gold-tiers.js +241 -0
- package/dist/governance-dashboard.js +327 -0
- package/dist/graph-viz.js +240 -0
- package/dist/guided-frontier.js +195 -0
- package/dist/hierarchical-planner.js +148 -0
- package/dist/identifier-parser.js +260 -0
- package/dist/immune-metrics.js +93 -0
- package/dist/immune-receiver.js +158 -0
- package/dist/immune-reporter.js +1 -1
- package/dist/improvement-orchestrator.js +206 -0
- package/dist/inject-p0-vocabulary.js +300 -0
- package/dist/intent-parser.js +218 -0
- package/dist/invariant-algebra.js +476 -0
- package/dist/invariant-calculus.js +533 -0
- package/dist/ir-utils.js +70 -0
- package/dist/ir-utils.test.js +50 -0
- package/dist/knowledge-api.js +312 -0
- package/dist/knowledge-evolution.js +452 -0
- package/dist/knowledge-explorer.js +506 -0
- package/dist/knowledge-flywheel.js +274 -0
- package/dist/knowledge-governance.js +338 -0
- package/dist/knowledge-governance.test.js +150 -0
- package/dist/knowledge-graph.js +181 -0
- package/dist/knowledge-guided-synth.js +246 -0
- package/dist/knowledge-loop.test.js +77 -0
- package/dist/knowledge-object.js +316 -0
- package/dist/knowledge-package.js +98 -0
- package/dist/kpi-dashboard.js +561 -0
- package/dist/l3-cross-function.js +280 -0
- package/dist/learning-ranker.js +148 -0
- package/dist/learning-ranker.test.js +291 -0
- package/dist/ledger/accountability.js +322 -0
- package/dist/ledger/chain-builder.js +185 -0
- package/dist/ledger/cli.js +222 -0
- package/dist/ledger/index.js +13 -0
- package/dist/ledger/signatures.js +193 -0
- package/dist/ledger/types.js +9 -0
- package/dist/llm.js +74 -3
- package/dist/load-benchmarks.js +8 -3
- package/dist/logger.js +66 -0
- package/dist/logger.test.js +37 -0
- package/dist/logistic-reward.js +339 -0
- package/dist/logistic-reward.test.js +180 -0
- package/dist/macro-graph.js +193 -0
- package/dist/macro-repair.js +183 -0
- package/dist/mcp-server.mjs +1202 -483
- package/dist/memory-layer.js +42 -5
- package/dist/multi-repo-precision.js +422 -0
- package/dist/name-free-protocol.js +425 -0
- package/dist/name-free-protocol.test.js +170 -0
- package/dist/name-scrambling.js +138 -0
- package/dist/name-scrambling.test.js +16 -0
- package/dist/p3-observability.test.js +281 -0
- package/dist/p5-orchestrator.test.js +225 -0
- package/dist/pairwise-preference.js +294 -0
- package/dist/pairwise-preference.test.js +140 -0
- package/dist/planner-constraints.js +104 -0
- package/dist/planner-prompts.js +155 -0
- package/dist/planner-telemetry.js +415 -0
- package/dist/planner-trace.js +214 -0
- package/dist/planner.js +162 -167
- package/dist/plsb/artifact.js +116 -0
- package/dist/plsb/cli.js +71 -0
- package/dist/plsb/index.js +19 -0
- package/dist/plsb/leaderboard.js +249 -0
- package/dist/plsb/report-md.js +156 -0
- package/dist/plsb/schema.js +179 -0
- package/dist/plsb-benchmark.js +284 -0
- package/dist/plsb-benchmark.test.js +119 -0
- package/dist/policy/cli.js +134 -0
- package/dist/policy/engine.js +333 -0
- package/dist/policy/index.js +12 -0
- package/dist/policy/types.js +59 -0
- package/dist/policy-miner.js +505 -0
- package/dist/precision-analyze.js +229 -0
- package/dist/precision-benchmark.js +147 -0
- package/dist/precision-label-c.js +134 -0
- package/dist/precision-label.js +193 -0
- package/dist/precision-report-c.js +149 -0
- package/dist/precision-report.js +246 -0
- package/dist/progmune-status.js +108 -0
- package/dist/proof-engine.js +479 -0
- package/dist/proof-provenance.js +315 -0
- package/dist/protocol-coverage.js +294 -0
- package/dist/protocol-detector.js +1189 -0
- package/dist/protocol-embedding-expanded.js +297 -0
- package/dist/protocol-embedding-expanded.test.js +97 -0
- package/dist/protocol-embedding.js +195 -0
- package/dist/protocol-embedding.test.js +82 -0
- package/dist/protocol-extractor-v2.js +354 -0
- package/dist/protocol-extractor-v2.test.js +140 -0
- package/dist/protocol-extractor.js +310 -0
- package/dist/protocol-extractor.test.js +113 -0
- package/dist/protocol-foundation.js +322 -0
- package/dist/protocol-foundation.test.js +163 -0
- package/dist/protocol-frontier.js +243 -0
- package/dist/protocol-frontier.test.js +92 -0
- package/dist/protocol-gap-analyzer.js +228 -0
- package/dist/protocol-gap-analyzer.test.js +49 -0
- package/dist/protocol-invariants.js +276 -0
- package/dist/protocol-invariants.test.js +111 -0
- package/dist/protocol-knowledge.js +464 -0
- package/dist/protocol-miner.js +343 -0
- package/dist/protocol-mining.js +207 -0
- package/dist/protocol-mining.test.js +37 -0
- package/dist/protocol-registry.js +1 -1
- package/dist/protocol-security-benchmark.js +222 -0
- package/dist/protocol-vulnerability.js +257 -0
- package/dist/protocol-vulnerability.test.js +60 -0
- package/dist/python-benchmark.js +120 -0
- package/dist/python-emitter.js +163 -45
- package/dist/python-protocol-extractor.js +187 -0
- package/dist/python-protocol-extractor.test.js +116 -0
- package/dist/realworld-benchmark.js +646 -0
- package/dist/realworld-benchmark.test.js +36 -0
- package/dist/repair-arch.test.js +411 -0
- package/dist/repair-evolution.test.js +454 -0
- package/dist/repair-executor.js +719 -0
- package/dist/repair-proposal.js +4 -4
- package/dist/repair-ranker.js +141 -0
- package/dist/repair-strategies.js +419 -0
- package/dist/repair-taxonomy.js +234 -0
- package/dist/repair-types.js +12 -0
- package/dist/repo-evaluator.js +250 -0
- package/dist/repo-evaluator.test.js +128 -0
- package/dist/resource-abstraction.js +242 -0
- package/dist/resource-detector.js +211 -0
- package/dist/result.test.js +43 -0
- package/dist/reward-system.js +411 -0
- package/dist/reward-system.test.js +175 -0
- package/dist/risk-model.js +215 -0
- package/dist/rule-miner.js +234 -7
- package/dist/rule-specificity.js +254 -0
- package/dist/runtime-types.js +27 -0
- package/dist/scaffold.js +208 -0
- package/dist/scale-collector.test.js +101 -0
- package/dist/scale-trajectory-collector.js +128 -0
- package/dist/sdk.js +250 -0
- package/dist/search-planner.js +4 -41
- package/dist/semantic-snapshot.js +1 -1
- package/dist/semantic-topology.js +121 -0
- package/dist/semantic-trace.js +310 -317
- package/dist/sequence-extractor.js +343 -0
- package/dist/skill-library.js +245 -0
- package/dist/skill-planner.test.js +189 -0
- package/dist/software-physics.js +291 -0
- package/dist/software-physics.test.js +81 -0
- package/dist/ssg-precision.js +478 -0
- package/dist/ssg-validator.js +71 -21
- package/dist/state-inference-doubleblind.test.js +160 -0
- package/dist/state-inference.js +516 -0
- package/dist/state-inference.test.js +115 -0
- package/dist/state-machine-fingerprint.js +345 -0
- package/dist/state-machine-fingerprint.test.js +120 -0
- package/dist/state-miner.js +386 -0
- package/dist/state-name-inference.js +213 -0
- package/dist/state-name-inference.test.js +69 -0
- package/dist/strategy-planner.js +262 -96
- package/dist/strategy-planner.test.js +135 -0
- package/dist/telemetry-analytics.test.js +402 -0
- package/dist/terminal-format.js +68 -0
- package/dist/terminal-format.test.js +83 -0
- package/dist/topology-factory.js +196 -0
- package/dist/topology-representation.js +242 -0
- package/dist/topology-representation.test.js +27 -0
- package/dist/trajectory-augmentation.js +254 -0
- package/dist/trajectory-augmentation.test.js +63 -0
- package/dist/trajectory-corpus.js +440 -0
- package/dist/trajectory-corpus.test.js +32 -0
- package/dist/trajectory-feedback.test.js +116 -0
- package/dist/transition-synthesizer.js +286 -0
- package/dist/transition-synthesizer.test.js +123 -0
- package/dist/trust/api-semantic-mapper.js +809 -0
- package/dist/trust/call-graph-propagator.js +225 -0
- package/dist/trust/cli.js +122 -0
- package/dist/trust/compliance-scorer.js +283 -0
- package/dist/trust/confidence-calculator.js +261 -0
- package/dist/trust/engine.js +1145 -0
- package/dist/trust/explainability.js +85 -0
- package/dist/trust/formatters/ci.js +42 -0
- package/dist/trust/formatters/json.js +11 -0
- package/dist/trust/formatters/terminal.js +152 -0
- package/dist/trust/index.js +39 -0
- package/dist/trust/phase1-verify.js +171 -0
- package/dist/trust/protocol-domain-validator.js +697 -0
- package/dist/trust/score-calculator.js +282 -0
- package/dist/trust/ssg-bridge.js +641 -0
- package/dist/trust/ssg-bridge.test.js +269 -0
- package/dist/trust/types.js +67 -0
- package/dist/trust/violation-trace.js +335 -0
- package/dist/trust-api.js +179 -0
- package/dist/trust-calibration.js +279 -0
- package/dist/unknown-protocol-discovery.js +339 -0
- package/dist/unknown-protocol-discovery.test.js +102 -0
- package/dist/unsupervised-physics.js +230 -0
- package/dist/unsupervised-physics.test.js +95 -0
- package/dist/utils.test.js +37 -0
- package/dist/validator.js +187 -10
- package/dist/verification-intelligence.js +475 -0
- package/dist/verify-api.js +432 -0
- package/dist/vi-impact-report.js +293 -0
- package/dist/wl-fingerprint.js +162 -0
- package/dist/wl-fingerprint.test.js +130 -0
- package/dist/zeroshot-strategy.js +139 -0
- package/docs/Progmune_/346/212/225/350/265/204/344/272/272/347/231/275/347/232/256/344/271/246_v2.0.html +576 -0
- package/docs/Progmune_/351/241/271/347/233/256/345/205/250/350/247/243.html +710 -0
- package/package.json +74 -7
- package/protocols.json +1956 -50
- package/.dockerignore +0 -14
- package/.mcp.json +0 -11
- package/.progmune_allowlist +0 -50
- package/.test_report/test_report.md +0 -87
- package/Dockerfile +0 -9
- package/FAQ.md +0 -167
- package/WHITEPAPER.md +0 -540
- package/demo-project/auth.ts +0 -55
- package/demo-project/tsconfig.json +0 -8
- package/dist/acl-breakdown.js +0 -13
- package/dist/all-sessions.js +0 -11
- package/dist/antibody-stats.js +0 -11
- package/dist/branch-tree-count.js +0 -14
- package/dist/common-fixpath.js +0 -12
- package/dist/constraint-types.js +0 -12
- package/dist/exec-metrics.js +0 -11
- package/dist/failure-report.js +0 -11
- package/dist/fast-path-hits.js +0 -13
- package/dist/fingerprint-list.js +0 -15
- package/dist/gen-history-log.js +0 -13
- package/dist/heatmap-data.js +0 -11
- package/dist/recent-session.js +0 -12
- package/dist/svl-distribution.js +0 -11
- package/dist/terminal-status.js +0 -11
- package/dist/token-savings.js +0 -11
- package/dist/total-repairs.js +0 -12
- package/dist/unresolved-count.js +0 -12
- package/dist/valid-fingerprints.js +0 -13
- package/dist/verify-ledgers.js +0 -11
- package/docs/whitepaper-style.css +0 -77
- package/docs/whitepaper-v2.1.md +0 -609
- package/docs/whitepaper-v2.2.md +0 -1064
- package/docs/whitepaper-v2.2.pdf +0 -0
- package/fly.toml +0 -31
- package/public/dashboard.html +0 -119
- package/server/hub.js +0 -116
- package/test/replay-golden/sess_1780063202050_mgeld.json +0 -9
- package/test/replay-golden/sess_1780064032560_gocld.json +0 -354
- package/test/replay-golden/sess_1780064413331_s2709.json +0 -606
- package/test/replay-golden/sess_1780064792710_y3avo.json +0 -614
- package/test/replay-golden.ts +0 -84
- package/test_benchmark.js +0 -165
- package/test_comprehensive.mjs +0 -638
- package/test_concurrency.js +0 -129
- package/test_ir_robustness.js +0 -85
- package/test_semantic_contracts.js +0 -269
- package/test_ssg_stress.js +0 -156
- package/test_svl3.js +0 -58
- package/tsconfig.json +0 -17
|
@@ -0,0 +1,180 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
/**
|
|
3
|
+
* P9.2: Real CVE Validation — Protocol Violation → Security Bug
|
|
4
|
+
*
|
|
5
|
+
* THE decisive question: can state-machine invariant violations
|
|
6
|
+
* detect REAL security vulnerabilities, not just synthetic damage?
|
|
7
|
+
*
|
|
8
|
+
* Uses the 20 real-world defect cases (mapped to CVEs) from
|
|
9
|
+
* realworld-benchmark.ts as a ground-truth dataset.
|
|
10
|
+
*
|
|
11
|
+
* For each CVE case:
|
|
12
|
+
* 1. Build template SM from the expected (correct) sequence
|
|
13
|
+
* 2. Build test SM from the broken (vulnerable) sequence
|
|
14
|
+
* 3. Run detectStructuralViolations(testSM, templateSM)
|
|
15
|
+
* 4. Check if the detected violation type matches the CVE category
|
|
16
|
+
*
|
|
17
|
+
* Success criteria:
|
|
18
|
+
* Recall > 70% (most real vulnerabilities detected)
|
|
19
|
+
* Precision > 50% (few false positives)
|
|
20
|
+
* Explainability > 90% (violation description maps to CVE type)
|
|
21
|
+
*/
|
|
22
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
23
|
+
exports.runAnyCVEBenchmark = runAnyCVEBenchmark;
|
|
24
|
+
exports.runCVEBenchmark = runCVEBenchmark;
|
|
25
|
+
exports.printCVEReport = printCVEReport;
|
|
26
|
+
const realworld_benchmark_1 = require("./realworld-benchmark");
|
|
27
|
+
const state_inference_1 = require("./experimental/state-inference");
|
|
28
|
+
const protocol_invariants_1 = require("./protocol-invariants");
|
|
29
|
+
// ═══════════════════════════════════════════════════════════════
|
|
30
|
+
// CVE Category → Expected Violation Subtype mapping
|
|
31
|
+
// ═══════════════════════════════════════════════════════════════
|
|
32
|
+
const CWE_TO_VIOLATION = {
|
|
33
|
+
resource_leak: "missing_release",
|
|
34
|
+
auth_bypass: "missing_prerequisite",
|
|
35
|
+
data_corruption: "missing_commit",
|
|
36
|
+
use_after_free: "missing_release",
|
|
37
|
+
race_condition: "missing_prerequisite",
|
|
38
|
+
};
|
|
39
|
+
/**
|
|
40
|
+
* Run the CVE benchmark against ANY array of CVE cases.
|
|
41
|
+
* The benchmark is named after the data source: "20-case" or "100-case".
|
|
42
|
+
*/
|
|
43
|
+
function runAnyCVEBenchmark(cases) {
|
|
44
|
+
return runCVEBenchmarkInternal(cases);
|
|
45
|
+
}
|
|
46
|
+
/** Run the benchmark against the curated 20-case set. */
|
|
47
|
+
function runCVEBenchmark() {
|
|
48
|
+
const curated = realworld_benchmark_1.REAL_WORLD_DEFECTS.map((d) => ({
|
|
49
|
+
id: d.id,
|
|
50
|
+
cve: d.source?.replace(" pattern", "") || "",
|
|
51
|
+
title: d.title,
|
|
52
|
+
description: d.description,
|
|
53
|
+
severity: d.severity,
|
|
54
|
+
cwe: "",
|
|
55
|
+
category: d.category,
|
|
56
|
+
broken: d.broken,
|
|
57
|
+
expected: d.expected,
|
|
58
|
+
project: "curated",
|
|
59
|
+
affectedVersions: [],
|
|
60
|
+
source: "curated",
|
|
61
|
+
}));
|
|
62
|
+
return runCVEBenchmarkInternal(curated);
|
|
63
|
+
}
|
|
64
|
+
function runCVEBenchmarkInternal(defects) {
|
|
65
|
+
const results = {
|
|
66
|
+
total: defects.length,
|
|
67
|
+
detected: 0,
|
|
68
|
+
categoryMatched: 0,
|
|
69
|
+
recall: 0,
|
|
70
|
+
precision: 0,
|
|
71
|
+
bySeverity: {},
|
|
72
|
+
byCategory: {},
|
|
73
|
+
results: [],
|
|
74
|
+
};
|
|
75
|
+
for (const defect of defects) {
|
|
76
|
+
// Build template SM from the expected (correct) sequence
|
|
77
|
+
const templateSM = (0, state_inference_1.inferStateMachine)([defect.expected]);
|
|
78
|
+
// Build test SM from the broken (vulnerable) sequence
|
|
79
|
+
const brokenSM = (0, state_inference_1.inferStateMachine)([defect.broken]);
|
|
80
|
+
// Run structural violation detection
|
|
81
|
+
const violations = (0, protocol_invariants_1.detectStructuralViolations)(brokenSM, templateSM);
|
|
82
|
+
// Map violations to CVE categories
|
|
83
|
+
const violationTypes = violations.map(v => v.violationSubtype);
|
|
84
|
+
const expectedViolation = CWE_TO_VIOLATION[defect.category || "other"];
|
|
85
|
+
const categoryMatch = expectedViolation
|
|
86
|
+
? violationTypes.includes(expectedViolation)
|
|
87
|
+
: false;
|
|
88
|
+
const detected = violations.length > 0;
|
|
89
|
+
// Track by severity
|
|
90
|
+
const sev = defect.severity || "medium";
|
|
91
|
+
if (!results.bySeverity[sev]) {
|
|
92
|
+
results.bySeverity[sev] = { total: 0, detected: 0 };
|
|
93
|
+
}
|
|
94
|
+
results.bySeverity[sev].total++;
|
|
95
|
+
if (detected)
|
|
96
|
+
results.bySeverity[sev].detected++;
|
|
97
|
+
// Track by category
|
|
98
|
+
const cat = defect.category || "other";
|
|
99
|
+
if (!results.byCategory[cat]) {
|
|
100
|
+
results.byCategory[cat] = { total: 0, detected: 0 };
|
|
101
|
+
}
|
|
102
|
+
results.byCategory[cat].total++;
|
|
103
|
+
if (detected)
|
|
104
|
+
results.byCategory[cat].detected++;
|
|
105
|
+
if (detected)
|
|
106
|
+
results.detected++;
|
|
107
|
+
if (categoryMatch)
|
|
108
|
+
results.categoryMatched++;
|
|
109
|
+
results.results.push({
|
|
110
|
+
defectId: defect.id,
|
|
111
|
+
title: defect.title,
|
|
112
|
+
severity: defect.severity,
|
|
113
|
+
cweCategory: defect.category || "other",
|
|
114
|
+
detected,
|
|
115
|
+
violationCount: violations.length,
|
|
116
|
+
violationTypes,
|
|
117
|
+
categoryMatch,
|
|
118
|
+
templateStates: templateSM.stateCount,
|
|
119
|
+
brokenStates: brokenSM.stateCount,
|
|
120
|
+
details: violations.map(v => v.description),
|
|
121
|
+
});
|
|
122
|
+
}
|
|
123
|
+
// Compute metrics
|
|
124
|
+
results.recall = results.total > 0 ? results.detected / results.total : 0;
|
|
125
|
+
// Precision: of detected, how many have correct category?
|
|
126
|
+
results.precision = results.detected > 0
|
|
127
|
+
? results.categoryMatched / results.detected
|
|
128
|
+
: 0;
|
|
129
|
+
return results;
|
|
130
|
+
}
|
|
131
|
+
function printCVEReport(report) {
|
|
132
|
+
console.log("\n╔════════════════════════════════════════════════════╗");
|
|
133
|
+
console.log("║ P9.2 Real CVE Validation ║");
|
|
134
|
+
console.log("║ Protocol Violation → Security Bug? ║");
|
|
135
|
+
console.log("╚════════════════════════════════════════════════════╝\n");
|
|
136
|
+
console.log(` Total CVEs: ${report.total}`);
|
|
137
|
+
console.log(` Detected: ${report.detected} (${(report.recall * 100).toFixed(0)}%)`);
|
|
138
|
+
console.log(` Category matched: ${report.categoryMatched} (${(report.precision * 100).toFixed(0)}%)`);
|
|
139
|
+
console.log();
|
|
140
|
+
console.log(` ── By Severity ──`);
|
|
141
|
+
for (const [sev, stats] of Object.entries(report.bySeverity)) {
|
|
142
|
+
const rate = stats.total > 0 ? (stats.detected / stats.total * 100).toFixed(0) : "N/A";
|
|
143
|
+
const icon = rate === "100" ? "✅" : rate === "0" ? "❌" : "⚠️";
|
|
144
|
+
console.log(` ${sev.padEnd(10)} ${stats.detected}/${stats.total} (${rate}%) ${icon}`);
|
|
145
|
+
}
|
|
146
|
+
console.log();
|
|
147
|
+
console.log(` ── By Category ──`);
|
|
148
|
+
for (const [cat, stats] of Object.entries(report.byCategory)) {
|
|
149
|
+
const rate = stats.total > 0 ? (stats.detected / stats.total * 100).toFixed(0) : "N/A";
|
|
150
|
+
const icon = rate === "100" ? "✅" : rate === "0" ? "❌" : "⚠️";
|
|
151
|
+
console.log(` ${cat.padEnd(18)} ${stats.detected}/${stats.total} (${rate}%) ${icon}`);
|
|
152
|
+
}
|
|
153
|
+
console.log();
|
|
154
|
+
console.log(` ── Per-CVE Results ──`);
|
|
155
|
+
for (const r of report.results) {
|
|
156
|
+
const icon = r.detected
|
|
157
|
+
? (r.categoryMatch ? "✅" : "⚠️")
|
|
158
|
+
: "❌";
|
|
159
|
+
console.log(` ${icon} ${r.defectId} ${r.severity.padEnd(8)} tpl=${r.templateStates} broken=${r.brokenStates} violations=${r.violationCount} types=${r.violationTypes.join(",") || "none"}`);
|
|
160
|
+
if (r.details.length > 0) {
|
|
161
|
+
for (const d of r.details.slice(0, 2)) {
|
|
162
|
+
console.log(` ${d.slice(0, 100)}`);
|
|
163
|
+
}
|
|
164
|
+
}
|
|
165
|
+
}
|
|
166
|
+
console.log();
|
|
167
|
+
console.log(` ── Verdict ──`);
|
|
168
|
+
const recallOk = report.recall > 0.7;
|
|
169
|
+
const precisionOk = report.precision > 0.5;
|
|
170
|
+
if (recallOk && precisionOk) {
|
|
171
|
+
console.log(` ✅ COMMERCIAL VIABILITY: Protocol violations → Real CVE detection works.`);
|
|
172
|
+
}
|
|
173
|
+
else if (recallOk || precisionOk) {
|
|
174
|
+
console.log(` ⚠️ PARTIAL: One metric passes, one fails. Need investigation.`);
|
|
175
|
+
}
|
|
176
|
+
else {
|
|
177
|
+
console.log(` ❌ GAP EXISTS: Protocol structure alone insufficient for real CVE detection.`);
|
|
178
|
+
}
|
|
179
|
+
console.log();
|
|
180
|
+
}
|
|
@@ -0,0 +1,28 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
3
|
+
const vitest_1 = require("vitest");
|
|
4
|
+
const cve_benchmark_1 = require("./cve-benchmark");
|
|
5
|
+
(0, vitest_1.describe)("P9.2 Real CVE Validation", () => {
|
|
6
|
+
(0, vitest_1.it)("detects protocol violations in real-world CVE cases", () => {
|
|
7
|
+
const report = (0, cve_benchmark_1.runCVEBenchmark)();
|
|
8
|
+
(0, cve_benchmark_1.printCVEReport)(report);
|
|
9
|
+
(0, vitest_1.expect)(report.total).toBe(20);
|
|
10
|
+
(0, vitest_1.expect)(report.detected).toBeGreaterThan(0);
|
|
11
|
+
(0, vitest_1.expect)(report.recall).toBeGreaterThanOrEqual(0);
|
|
12
|
+
(0, vitest_1.expect)(report.precision).toBeGreaterThanOrEqual(0);
|
|
13
|
+
});
|
|
14
|
+
(0, vitest_1.it)("resource_leak CVEs detected as missing_release", () => {
|
|
15
|
+
const report = (0, cve_benchmark_1.runCVEBenchmark)();
|
|
16
|
+
const rl = report.byCategory["resource_leak"];
|
|
17
|
+
(0, vitest_1.expect)(rl).toBeTruthy();
|
|
18
|
+
// Most resource leaks should be detected
|
|
19
|
+
(0, vitest_1.expect)(rl.detected).toBeGreaterThan(0);
|
|
20
|
+
});
|
|
21
|
+
(0, vitest_1.it)("auth_bypass CVEs detected as missing_prerequisite", () => {
|
|
22
|
+
const report = (0, cve_benchmark_1.runCVEBenchmark)();
|
|
23
|
+
const ab = report.byCategory["auth_bypass"];
|
|
24
|
+
(0, vitest_1.expect)(ab).toBeTruthy();
|
|
25
|
+
// At least some auth bypasses should be detected
|
|
26
|
+
(0, vitest_1.expect)(ab.detected).toBeGreaterThanOrEqual(0);
|
|
27
|
+
});
|
|
28
|
+
});
|
|
@@ -0,0 +1,73 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
3
|
+
exports.fetchCVEsFromNVD = fetchCVEsFromNVD;
|
|
4
|
+
const NVD_API_BASE = 'https://services.nvd.nist.gov/rest/json/cves/2.0';
|
|
5
|
+
async function fetchCVEsFromNVD(options = {}) {
|
|
6
|
+
const { limit = 1000, cwe, severity } = options;
|
|
7
|
+
const params = new URLSearchParams({
|
|
8
|
+
resultsPerPage: Math.min(limit, 2000).toString(),
|
|
9
|
+
});
|
|
10
|
+
if (cwe)
|
|
11
|
+
params.append('cweId', cwe);
|
|
12
|
+
if (severity)
|
|
13
|
+
params.append('cvssV3Severity', severity);
|
|
14
|
+
const url = `${NVD_API_BASE}?${params.toString()}`;
|
|
15
|
+
const response = await fetch(url);
|
|
16
|
+
if (!response.ok)
|
|
17
|
+
throw new Error(`NVD API error: ${response.status} ${response.statusText}`);
|
|
18
|
+
const data = (await response.json());
|
|
19
|
+
const cves = [];
|
|
20
|
+
for (const vuln of data.vulnerabilities || []) {
|
|
21
|
+
const cve = vuln.cve;
|
|
22
|
+
const id = cve.id;
|
|
23
|
+
const description = cve.descriptions?.find((d) => d.lang === 'en')?.value || '';
|
|
24
|
+
const severityScore = cve.metrics?.cvssMetricV3?.[0]?.cvssData?.baseSeverity || 'medium';
|
|
25
|
+
const cweId = cve.weaknesses?.[0]?.description?.[0]?.value || '';
|
|
26
|
+
const project = extractProjectName(cve);
|
|
27
|
+
const affectedVersions = cve.affects?.vendor?.vendor_data?.[0]?.product?.product_data?.[0]?.version?.version_data
|
|
28
|
+
?.map((v) => v.version_value) || [];
|
|
29
|
+
const { broken, expected } = inferFromDescription(description);
|
|
30
|
+
cves.push({
|
|
31
|
+
id,
|
|
32
|
+
cve: id,
|
|
33
|
+
title: cve.descriptions?.find((d) => d.lang === 'en')?.value?.slice(0, 80) || id,
|
|
34
|
+
description,
|
|
35
|
+
severity: severityScore.toLowerCase(),
|
|
36
|
+
cwe: cweId,
|
|
37
|
+
project,
|
|
38
|
+
affectedVersions,
|
|
39
|
+
broken,
|
|
40
|
+
expected,
|
|
41
|
+
source: 'nvd',
|
|
42
|
+
});
|
|
43
|
+
}
|
|
44
|
+
return cves.slice(0, limit);
|
|
45
|
+
}
|
|
46
|
+
function inferFromDescription(desc) {
|
|
47
|
+
const verbs = ['open', 'close', 'read', 'write', 'connect', 'disconnect', 'auth', 'verify', 'commit', 'rollback', 'malloc', 'free'];
|
|
48
|
+
const found = verbs.filter(v => desc.toLowerCase().includes(v));
|
|
49
|
+
if (found.length > 0) {
|
|
50
|
+
return { broken: found.slice(0, -1), expected: found };
|
|
51
|
+
}
|
|
52
|
+
return { broken: ['unknown'], expected: ['unknown'] };
|
|
53
|
+
}
|
|
54
|
+
function extractProjectName(cve) {
|
|
55
|
+
const vendorData = cve.affects?.vendor?.vendor_data;
|
|
56
|
+
if (vendorData && vendorData.length > 0) {
|
|
57
|
+
const product = vendorData[0]?.product?.product_data?.[0]?.product_name;
|
|
58
|
+
if (product)
|
|
59
|
+
return product;
|
|
60
|
+
}
|
|
61
|
+
const refs = cve.references?.reference_data || [];
|
|
62
|
+
for (const ref of refs) {
|
|
63
|
+
const url = ref.url;
|
|
64
|
+
if (url.includes('github.com')) {
|
|
65
|
+
const parts = url.split('/');
|
|
66
|
+
const idx = parts.indexOf('github.com');
|
|
67
|
+
if (idx !== -1 && parts.length > idx + 2) {
|
|
68
|
+
return parts[idx + 1] + '/' + parts[idx + 2];
|
|
69
|
+
}
|
|
70
|
+
}
|
|
71
|
+
}
|
|
72
|
+
return 'unknown';
|
|
73
|
+
}
|
|
@@ -0,0 +1,141 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
/**
|
|
3
|
+
* P3.1: Data Quality Layer
|
|
4
|
+
*
|
|
5
|
+
* Separates raw acceptance from verified correctness.
|
|
6
|
+
*
|
|
7
|
+
* Risk: accepted ≠ correct, rejected ≠ wrong.
|
|
8
|
+
* A user accepting a fast-but-leaky repair produces toxic training data
|
|
9
|
+
* for the Reward Model. A user rejecting a correct-but-slow repair
|
|
10
|
+
* deprives the system of a positive signal.
|
|
11
|
+
*
|
|
12
|
+
* RepairOutcome adds three independent verification signals:
|
|
13
|
+
* 1. executionSucceeded — did it actually run without errors?
|
|
14
|
+
* 2. postValidationPassed — did the SSG validator accept the final state?
|
|
15
|
+
* 3. regressionTestsPassed — did existing tests still pass?
|
|
16
|
+
*
|
|
17
|
+
* Future P4 Reward Model trains on verified outcomes, not raw acceptance.
|
|
18
|
+
*/
|
|
19
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
20
|
+
exports.computeQualityScore = computeQualityScore;
|
|
21
|
+
exports.computeRewardSignal = computeRewardSignal;
|
|
22
|
+
exports.generateQualityReport = generateQualityReport;
|
|
23
|
+
exports.printQualityReport = printQualityReport;
|
|
24
|
+
// ═══════════════════════════════════════════════════════════════
|
|
25
|
+
// Quality Scoring
|
|
26
|
+
// ═══════════════════════════════════════════════════════════════
|
|
27
|
+
/** Default weights for quality score computation. */
|
|
28
|
+
const DEFAULT_QUALITY_WEIGHTS = {
|
|
29
|
+
execution: 0.4, // did it run?
|
|
30
|
+
validation: 0.4, // did the SSG accept the result?
|
|
31
|
+
regression: 0.2, // did existing tests pass?
|
|
32
|
+
};
|
|
33
|
+
/**
|
|
34
|
+
* Compute a quality score from a RepairOutcome.
|
|
35
|
+
*
|
|
36
|
+
* If verification signals are missing, the score degrades gracefully:
|
|
37
|
+
* - No execution data → weight redistributed to validation
|
|
38
|
+
* - No validation data → weight redistributed to execution
|
|
39
|
+
* - Neither → 0.5 (neutral prior)
|
|
40
|
+
*/
|
|
41
|
+
function computeQualityScore(outcome, weights) {
|
|
42
|
+
const w = { ...DEFAULT_QUALITY_WEIGHTS, ...weights };
|
|
43
|
+
let score = 0;
|
|
44
|
+
let totalWeight = 0;
|
|
45
|
+
if (outcome.executionSucceeded !== undefined) {
|
|
46
|
+
score += (outcome.executionSucceeded ? 1 : 0) * w.execution;
|
|
47
|
+
totalWeight += w.execution;
|
|
48
|
+
}
|
|
49
|
+
if (outcome.postValidationPassed !== undefined) {
|
|
50
|
+
score += (outcome.postValidationPassed ? 1 : 0) * w.validation;
|
|
51
|
+
totalWeight += w.validation;
|
|
52
|
+
}
|
|
53
|
+
if (outcome.regressionTestsPassed !== undefined) {
|
|
54
|
+
score += (outcome.regressionTestsPassed ? 1 : 0) * w.regression;
|
|
55
|
+
totalWeight += w.regression;
|
|
56
|
+
}
|
|
57
|
+
return totalWeight > 0 ? score / totalWeight : 0.5;
|
|
58
|
+
}
|
|
59
|
+
// ═══════════════════════════════════════════════════════════════
|
|
60
|
+
// Quality-aware reward signal
|
|
61
|
+
// ═══════════════════════════════════════════════════════════════
|
|
62
|
+
/**
|
|
63
|
+
* Compute a quality-aware reward for a repair.
|
|
64
|
+
*
|
|
65
|
+
* reward = accepted * 0.4 + validationPassed * 0.4 + executionSucceeded * 0.2
|
|
66
|
+
*
|
|
67
|
+
* This is the training signal for P4 Reward Model.
|
|
68
|
+
* Raw acceptance alone is NOT sufficient — a fast-but-leaky repair
|
|
69
|
+
* that users love should NOT get a high reward.
|
|
70
|
+
*/
|
|
71
|
+
function computeRewardSignal(outcome) {
|
|
72
|
+
const accepted = outcome.accepted ? 1 : 0;
|
|
73
|
+
const validation = outcome.postValidationPassed ? 1 : 0;
|
|
74
|
+
const execution = outcome.executionSucceeded ? 1 : 0;
|
|
75
|
+
const regression = outcome.regressionTestsPassed ? 1 : 0;
|
|
76
|
+
// If validation or regression data is available, it dominates
|
|
77
|
+
if (outcome.postValidationPassed !== undefined || outcome.regressionTestsPassed !== undefined) {
|
|
78
|
+
return (accepted * 0.4 +
|
|
79
|
+
validation * 0.4 +
|
|
80
|
+
(execution * 0.1 + regression * 0.1));
|
|
81
|
+
}
|
|
82
|
+
// Fallback: execution + acceptance only
|
|
83
|
+
if (outcome.executionSucceeded !== undefined) {
|
|
84
|
+
return accepted * 0.5 + execution * 0.5;
|
|
85
|
+
}
|
|
86
|
+
// Minimal: no verification data → neutral prior (avoid overfitting to raw acceptance)
|
|
87
|
+
return 0.5;
|
|
88
|
+
}
|
|
89
|
+
/**
|
|
90
|
+
* Generate a quality report from telemetry data.
|
|
91
|
+
* Identifies contradictory outcomes where acceptance disagrees with execution.
|
|
92
|
+
*/
|
|
93
|
+
function generateQualityReport(outcomes) {
|
|
94
|
+
const total = outcomes.length;
|
|
95
|
+
if (total === 0) {
|
|
96
|
+
return {
|
|
97
|
+
totalOutcomes: 0, rawAcceptanceRate: 0, executionSuccessRate: 0,
|
|
98
|
+
validationPassRate: 0, regressionPassRate: 0, qualityScoreAvg: 0,
|
|
99
|
+
contradictoryOutcomes: 0,
|
|
100
|
+
};
|
|
101
|
+
}
|
|
102
|
+
const accepted = outcomes.filter(o => o.accepted).length;
|
|
103
|
+
const execOk = outcomes.filter(o => o.executionSucceeded === true).length;
|
|
104
|
+
const execTotal = outcomes.filter(o => o.executionSucceeded !== undefined).length;
|
|
105
|
+
const valOk = outcomes.filter(o => o.postValidationPassed === true).length;
|
|
106
|
+
const valTotal = outcomes.filter(o => o.postValidationPassed !== undefined).length;
|
|
107
|
+
const regOk = outcomes.filter(o => o.regressionTestsPassed === true).length;
|
|
108
|
+
const regTotal = outcomes.filter(o => o.regressionTestsPassed !== undefined).length;
|
|
109
|
+
const qualityScores = outcomes.map(o => computeQualityScore(o));
|
|
110
|
+
const qualityAvg = qualityScores.reduce((s, v) => s + v, 0) / total;
|
|
111
|
+
// Contradictory: accepted but execution failed, OR rejected but execution succeeded
|
|
112
|
+
const contradictory = outcomes.filter(o => o.executionSucceeded !== undefined &&
|
|
113
|
+
o.accepted !== o.executionSucceeded).length;
|
|
114
|
+
return {
|
|
115
|
+
totalOutcomes: total,
|
|
116
|
+
rawAcceptanceRate: total > 0 ? accepted / total : 0,
|
|
117
|
+
executionSuccessRate: execTotal > 0 ? execOk / execTotal : 0,
|
|
118
|
+
validationPassRate: valTotal > 0 ? valOk / valTotal : 0,
|
|
119
|
+
regressionPassRate: regTotal > 0 ? regOk / regTotal : 0,
|
|
120
|
+
qualityScoreAvg: qualityAvg,
|
|
121
|
+
contradictoryOutcomes: contradictory,
|
|
122
|
+
};
|
|
123
|
+
}
|
|
124
|
+
function printQualityReport(report) {
|
|
125
|
+
console.log("\n╔══════════════════════════════════════════╗");
|
|
126
|
+
console.log("║ Data Quality Report ║");
|
|
127
|
+
console.log("╚══════════════════════════════════════════╝\n");
|
|
128
|
+
console.log(`Total Outcomes: ${report.totalOutcomes}`);
|
|
129
|
+
console.log(`Raw Acceptance: ${(report.rawAcceptanceRate * 100).toFixed(1)}%`);
|
|
130
|
+
console.log(`Execution Success: ${(report.executionSuccessRate * 100).toFixed(1)}%`);
|
|
131
|
+
console.log(`Validation Pass: ${(report.validationPassRate * 100).toFixed(1)}%`);
|
|
132
|
+
console.log(`Regression Pass: ${(report.regressionPassRate * 100).toFixed(1)}%`);
|
|
133
|
+
console.log(`Avg Quality Score: ${(report.qualityScoreAvg * 100).toFixed(1)}%`);
|
|
134
|
+
console.log();
|
|
135
|
+
if (report.contradictoryOutcomes > 0) {
|
|
136
|
+
const pct = (report.contradictoryOutcomes / report.totalOutcomes * 100).toFixed(1);
|
|
137
|
+
console.log(`⚠️ Contradictory: ${report.contradictoryOutcomes} (${pct}%)`);
|
|
138
|
+
console.log(" (accepted ≠ execution result — potential data poison)");
|
|
139
|
+
}
|
|
140
|
+
console.log();
|
|
141
|
+
}
|