progmune-runtime 2.1.5 → 3.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +108 -468
- package/dist/ablation-study.js +144 -0
- package/dist/ablation-study.test.js +18 -0
- package/dist/action-runtime.js +3 -1
- package/dist/active-learning.js +211 -0
- package/dist/analytics.js +139 -0
- package/dist/asset-factory.js +309 -0
- package/dist/asset-growth.js +244 -0
- package/dist/asset-promotion.js +382 -0
- package/dist/asset-quality.js +550 -0
- package/dist/audit/business-translator.js +285 -0
- package/dist/audit/cli.js +66 -0
- package/dist/audit/formatters/html.js +379 -0
- package/dist/audit/formatters/json.js +11 -0
- package/dist/audit/formatters/markdown.js +192 -0
- package/dist/audit/formatters/terminal.js +189 -0
- package/dist/audit/index.js +25 -0
- package/dist/audit/report-builder.js +318 -0
- package/dist/audit/types.js +8 -0
- package/dist/audit.js +3 -3
- package/dist/auto-benchmark-generator.js +137 -0
- package/dist/auto-benchmark-generator.test.js +45 -0
- package/dist/auto-protocol-synthesizer.js +362 -0
- package/dist/auto-protocol-synthesizer.test.js +82 -0
- package/dist/autonomous-patch.js +175 -0
- package/dist/autonomous-patch.test.js +128 -0
- package/dist/badge/badge-server.js +98 -0
- package/dist/behavior-miner.js +442 -0
- package/dist/belief-layer.js +475 -0
- package/dist/benchmark-count.js +5 -0
- package/dist/benchmark-generator.js +211 -0
- package/dist/benchmark-harness.js +201 -0
- package/dist/benchmark-pass-rate.js +7 -0
- package/dist/benchmark-report.js +8 -3
- package/dist/benchmark-save.js +14 -1
- package/dist/bootstrap-validation.js +197 -0
- package/dist/bootstrap-validation.test.js +51 -0
- package/dist/branch-ledger.js +1 -1
- package/dist/capability-gap.js +130 -0
- package/dist/certify-html.js +351 -0
- package/dist/certify.js +326 -0
- package/dist/check.js +4 -4
- package/dist/compliance-miner.js +447 -0
- package/dist/continuous-benchmark.js +194 -0
- package/dist/continuous-benchmark.test.js +116 -0
- package/dist/corpus-stats.js +173 -0
- package/dist/counterfactual-engine.js +288 -0
- package/dist/coverage-dashboard.js +109 -0
- package/dist/coverage-system.test.js +205 -0
- package/dist/cross-repo-precision.js +352 -0
- package/dist/cve-benchmark.js +180 -0
- package/dist/cve-benchmark.test.js +28 -0
- package/dist/cve-collector.js +73 -0
- package/dist/data-quality.js +141 -0
- package/dist/decision-engine.js +388 -0
- package/dist/derive-metadata.js +250 -0
- package/dist/difficulty-active.test.js +198 -0
- package/dist/difficulty-map.js +244 -0
- package/dist/discovery-analytics.js +125 -0
- package/dist/discovery-model.js +149 -0
- package/dist/discovery-optimize.test.js +199 -0
- package/dist/discovery-trace.js +276 -0
- package/dist/discovery-trace.test.js +97 -0
- package/dist/emitter.js +83 -1
- package/dist/enterprise-dashboard.js +405 -0
- package/dist/eval-hardening.js +297 -0
- package/dist/eval-hardening.test.js +85 -0
- package/dist/evaluation-campaign.js +359 -0
- package/dist/evaluation-campaign.test.js +181 -0
- package/dist/evidence-growth.js +143 -0
- package/dist/evidence-repository.js +209 -0
- package/dist/evidence-system.js +441 -0
- package/dist/execute.js +15 -7
- package/dist/experimental/software-physics.js +291 -0
- package/dist/experimental/state-inference.js +516 -0
- package/dist/experimental/unsupervised-physics.js +230 -0
- package/dist/extract-ir-python.js +54 -7
- package/dist/extract-ir.js +376 -12
- package/dist/failure-collector.js +2 -2
- package/dist/failure-corpus.js +322 -9
- package/dist/feedback.js +16 -5
- package/dist/feedback.test.js +49 -0
- package/dist/file-lock.js +1 -1
- package/dist/flywheel-batch.js +292 -0
- package/dist/frameworks/express-cli.js +237 -0
- package/dist/frameworks/express-detector.js +445 -0
- package/dist/frameworks/express-detector.test.js +206 -0
- package/dist/frameworks/index.js +30 -0
- package/dist/frameworks/nestjs-detector.js +302 -0
- package/dist/frameworks/trpc-detector.js +161 -0
- package/dist/frameworks/version-awareness.js +179 -0
- package/dist/function-synonyms.js +164 -0
- package/dist/function-synonyms.test.js +68 -0
- package/dist/generalization.test.js +352 -0
- package/dist/goal-annotator.js +113 -0
- package/dist/goal-planner.js +563 -0
- package/dist/gold-cve.js +164 -0
- package/dist/gold-cve.test.js +104 -0
- package/dist/gold-quality.js +206 -0
- package/dist/gold-tiers.js +241 -0
- package/dist/governance-dashboard.js +327 -0
- package/dist/graph-viz.js +240 -0
- package/dist/guided-frontier.js +195 -0
- package/dist/hierarchical-planner.js +148 -0
- package/dist/identifier-parser.js +260 -0
- package/dist/immune-metrics.js +93 -0
- package/dist/immune-receiver.js +158 -0
- package/dist/immune-reporter.js +1 -1
- package/dist/improvement-orchestrator.js +206 -0
- package/dist/inject-p0-vocabulary.js +300 -0
- package/dist/intent-parser.js +218 -0
- package/dist/invariant-algebra.js +476 -0
- package/dist/invariant-calculus.js +533 -0
- package/dist/ir-utils.js +70 -0
- package/dist/ir-utils.test.js +50 -0
- package/dist/knowledge-api.js +312 -0
- package/dist/knowledge-evolution.js +452 -0
- package/dist/knowledge-explorer.js +506 -0
- package/dist/knowledge-flywheel.js +274 -0
- package/dist/knowledge-governance.js +338 -0
- package/dist/knowledge-governance.test.js +150 -0
- package/dist/knowledge-graph.js +181 -0
- package/dist/knowledge-guided-synth.js +246 -0
- package/dist/knowledge-loop.test.js +77 -0
- package/dist/knowledge-object.js +316 -0
- package/dist/knowledge-package.js +98 -0
- package/dist/kpi-dashboard.js +561 -0
- package/dist/l3-cross-function.js +280 -0
- package/dist/learning-ranker.js +148 -0
- package/dist/learning-ranker.test.js +291 -0
- package/dist/ledger/accountability.js +322 -0
- package/dist/ledger/chain-builder.js +185 -0
- package/dist/ledger/cli.js +222 -0
- package/dist/ledger/index.js +13 -0
- package/dist/ledger/signatures.js +193 -0
- package/dist/ledger/types.js +9 -0
- package/dist/llm.js +74 -3
- package/dist/load-benchmarks.js +8 -3
- package/dist/logger.js +66 -0
- package/dist/logger.test.js +37 -0
- package/dist/logistic-reward.js +339 -0
- package/dist/logistic-reward.test.js +180 -0
- package/dist/macro-graph.js +193 -0
- package/dist/macro-repair.js +183 -0
- package/dist/mcp-server.mjs +1202 -483
- package/dist/memory-layer.js +42 -5
- package/dist/multi-repo-precision.js +422 -0
- package/dist/name-free-protocol.js +425 -0
- package/dist/name-free-protocol.test.js +170 -0
- package/dist/name-scrambling.js +138 -0
- package/dist/name-scrambling.test.js +16 -0
- package/dist/p3-observability.test.js +281 -0
- package/dist/p5-orchestrator.test.js +225 -0
- package/dist/pairwise-preference.js +294 -0
- package/dist/pairwise-preference.test.js +140 -0
- package/dist/planner-constraints.js +104 -0
- package/dist/planner-prompts.js +155 -0
- package/dist/planner-telemetry.js +415 -0
- package/dist/planner-trace.js +214 -0
- package/dist/planner.js +162 -167
- package/dist/plsb/artifact.js +116 -0
- package/dist/plsb/cli.js +71 -0
- package/dist/plsb/index.js +19 -0
- package/dist/plsb/leaderboard.js +249 -0
- package/dist/plsb/report-md.js +156 -0
- package/dist/plsb/schema.js +179 -0
- package/dist/plsb-benchmark.js +284 -0
- package/dist/plsb-benchmark.test.js +119 -0
- package/dist/policy/cli.js +134 -0
- package/dist/policy/engine.js +333 -0
- package/dist/policy/index.js +12 -0
- package/dist/policy/types.js +59 -0
- package/dist/policy-miner.js +505 -0
- package/dist/precision-analyze.js +229 -0
- package/dist/precision-benchmark.js +147 -0
- package/dist/precision-label-c.js +134 -0
- package/dist/precision-label.js +193 -0
- package/dist/precision-report-c.js +149 -0
- package/dist/precision-report.js +246 -0
- package/dist/progmune-status.js +108 -0
- package/dist/proof-engine.js +479 -0
- package/dist/proof-provenance.js +315 -0
- package/dist/protocol-coverage.js +294 -0
- package/dist/protocol-detector.js +1189 -0
- package/dist/protocol-embedding-expanded.js +297 -0
- package/dist/protocol-embedding-expanded.test.js +97 -0
- package/dist/protocol-embedding.js +195 -0
- package/dist/protocol-embedding.test.js +82 -0
- package/dist/protocol-extractor-v2.js +354 -0
- package/dist/protocol-extractor-v2.test.js +140 -0
- package/dist/protocol-extractor.js +310 -0
- package/dist/protocol-extractor.test.js +113 -0
- package/dist/protocol-foundation.js +322 -0
- package/dist/protocol-foundation.test.js +163 -0
- package/dist/protocol-frontier.js +243 -0
- package/dist/protocol-frontier.test.js +92 -0
- package/dist/protocol-gap-analyzer.js +228 -0
- package/dist/protocol-gap-analyzer.test.js +49 -0
- package/dist/protocol-invariants.js +276 -0
- package/dist/protocol-invariants.test.js +111 -0
- package/dist/protocol-knowledge.js +464 -0
- package/dist/protocol-miner.js +343 -0
- package/dist/protocol-mining.js +207 -0
- package/dist/protocol-mining.test.js +37 -0
- package/dist/protocol-registry.js +1 -1
- package/dist/protocol-security-benchmark.js +222 -0
- package/dist/protocol-vulnerability.js +257 -0
- package/dist/protocol-vulnerability.test.js +60 -0
- package/dist/python-benchmark.js +120 -0
- package/dist/python-emitter.js +163 -45
- package/dist/python-protocol-extractor.js +187 -0
- package/dist/python-protocol-extractor.test.js +116 -0
- package/dist/realworld-benchmark.js +646 -0
- package/dist/realworld-benchmark.test.js +36 -0
- package/dist/repair-arch.test.js +411 -0
- package/dist/repair-evolution.test.js +454 -0
- package/dist/repair-executor.js +719 -0
- package/dist/repair-proposal.js +4 -4
- package/dist/repair-ranker.js +141 -0
- package/dist/repair-strategies.js +419 -0
- package/dist/repair-taxonomy.js +234 -0
- package/dist/repair-types.js +12 -0
- package/dist/repo-evaluator.js +250 -0
- package/dist/repo-evaluator.test.js +128 -0
- package/dist/resource-abstraction.js +242 -0
- package/dist/resource-detector.js +211 -0
- package/dist/result.test.js +43 -0
- package/dist/reward-system.js +411 -0
- package/dist/reward-system.test.js +175 -0
- package/dist/risk-model.js +215 -0
- package/dist/rule-miner.js +234 -7
- package/dist/rule-specificity.js +254 -0
- package/dist/runtime-types.js +27 -0
- package/dist/scaffold.js +208 -0
- package/dist/scale-collector.test.js +101 -0
- package/dist/scale-trajectory-collector.js +128 -0
- package/dist/sdk.js +250 -0
- package/dist/search-planner.js +4 -41
- package/dist/semantic-snapshot.js +1 -1
- package/dist/semantic-topology.js +121 -0
- package/dist/semantic-trace.js +310 -317
- package/dist/sequence-extractor.js +343 -0
- package/dist/skill-library.js +245 -0
- package/dist/skill-planner.test.js +189 -0
- package/dist/software-physics.js +291 -0
- package/dist/software-physics.test.js +81 -0
- package/dist/ssg-precision.js +478 -0
- package/dist/ssg-validator.js +71 -21
- package/dist/state-inference-doubleblind.test.js +160 -0
- package/dist/state-inference.js +516 -0
- package/dist/state-inference.test.js +115 -0
- package/dist/state-machine-fingerprint.js +345 -0
- package/dist/state-machine-fingerprint.test.js +120 -0
- package/dist/state-miner.js +386 -0
- package/dist/state-name-inference.js +213 -0
- package/dist/state-name-inference.test.js +69 -0
- package/dist/strategy-planner.js +326 -59
- package/dist/strategy-planner.test.js +135 -0
- package/dist/telemetry-analytics.test.js +402 -0
- package/dist/terminal-format.js +68 -0
- package/dist/terminal-format.test.js +83 -0
- package/dist/topology-factory.js +196 -0
- package/dist/topology-representation.js +242 -0
- package/dist/topology-representation.test.js +27 -0
- package/dist/trajectory-augmentation.js +254 -0
- package/dist/trajectory-augmentation.test.js +63 -0
- package/dist/trajectory-corpus.js +440 -0
- package/dist/trajectory-corpus.test.js +32 -0
- package/dist/trajectory-feedback.test.js +116 -0
- package/dist/transition-synthesizer.js +286 -0
- package/dist/transition-synthesizer.test.js +123 -0
- package/dist/trust/api-semantic-mapper.js +809 -0
- package/dist/trust/call-graph-propagator.js +225 -0
- package/dist/trust/cli.js +122 -0
- package/dist/trust/compliance-scorer.js +283 -0
- package/dist/trust/confidence-calculator.js +261 -0
- package/dist/trust/engine.js +1145 -0
- package/dist/trust/explainability.js +85 -0
- package/dist/trust/formatters/ci.js +42 -0
- package/dist/trust/formatters/json.js +11 -0
- package/dist/trust/formatters/terminal.js +152 -0
- package/dist/trust/index.js +39 -0
- package/dist/trust/phase1-verify.js +171 -0
- package/dist/trust/protocol-domain-validator.js +697 -0
- package/dist/trust/score-calculator.js +282 -0
- package/dist/trust/ssg-bridge.js +641 -0
- package/dist/trust/ssg-bridge.test.js +269 -0
- package/dist/trust/types.js +67 -0
- package/dist/trust/violation-trace.js +335 -0
- package/dist/trust-api.js +179 -0
- package/dist/trust-calibration.js +279 -0
- package/dist/unknown-protocol-discovery.js +339 -0
- package/dist/unknown-protocol-discovery.test.js +102 -0
- package/dist/unsupervised-physics.js +230 -0
- package/dist/unsupervised-physics.test.js +95 -0
- package/dist/utils.test.js +37 -0
- package/dist/validator.js +187 -10
- package/dist/verification-intelligence.js +475 -0
- package/dist/verify-api.js +432 -0
- package/dist/vi-impact-report.js +293 -0
- package/dist/wl-fingerprint.js +162 -0
- package/dist/wl-fingerprint.test.js +130 -0
- package/dist/zeroshot-strategy.js +139 -0
- package/docs/Progmune_/346/212/225/350/265/204/344/272/272/347/231/275/347/232/256/344/271/246_v2.0.html +576 -0
- package/docs/Progmune_/351/241/271/347/233/256/345/205/250/350/247/243.html +710 -0
- package/package.json +74 -7
- package/protocols.json +1956 -50
- package/.dockerignore +0 -14
- package/.mcp.json +0 -11
- package/.progmune_allowlist +0 -50
- package/.test_report/test_report.md +0 -87
- package/Dockerfile +0 -9
- package/FAQ.md +0 -167
- package/WHITEPAPER.md +0 -540
- package/demo-project/auth.ts +0 -55
- package/demo-project/tsconfig.json +0 -8
- package/dist/acl-breakdown.js +0 -13
- package/dist/all-sessions.js +0 -11
- package/dist/antibody-stats.js +0 -11
- package/dist/branch-tree-count.js +0 -14
- package/dist/common-fixpath.js +0 -12
- package/dist/constraint-types.js +0 -12
- package/dist/exec-metrics.js +0 -11
- package/dist/failure-report.js +0 -11
- package/dist/fast-path-hits.js +0 -13
- package/dist/fingerprint-list.js +0 -15
- package/dist/gen-history-log.js +0 -13
- package/dist/heatmap-data.js +0 -11
- package/dist/recent-session.js +0 -12
- package/dist/svl-distribution.js +0 -11
- package/dist/terminal-status.js +0 -11
- package/dist/token-savings.js +0 -11
- package/dist/total-repairs.js +0 -12
- package/dist/unresolved-count.js +0 -12
- package/dist/valid-fingerprints.js +0 -13
- package/dist/verify-ledgers.js +0 -11
- package/docs/whitepaper-style.css +0 -77
- package/docs/whitepaper-v2.1.md +0 -609
- package/docs/whitepaper-v2.2.md +0 -1064
- package/docs/whitepaper-v2.2.pdf +0 -0
- package/fly.toml +0 -31
- package/public/dashboard.html +0 -119
- package/server/hub.js +0 -116
- package/test/replay-golden/sess_1780063202050_mgeld.json +0 -9
- package/test/replay-golden/sess_1780064032560_gocld.json +0 -354
- package/test/replay-golden/sess_1780064413331_s2709.json +0 -606
- package/test/replay-golden/sess_1780064792710_y3avo.json +0 -614
- package/test/replay-golden.ts +0 -84
- package/test_benchmark.js +0 -165
- package/test_comprehensive.mjs +0 -638
- package/test_concurrency.js +0 -129
- package/test_ir_robustness.js +0 -85
- package/test_semantic_contracts.js +0 -269
- package/test_ssg_stress.js +0 -156
- package/test_svl3.js +0 -58
- package/tsconfig.json +0 -17
|
@@ -0,0 +1,116 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
/**
|
|
3
|
+
* Phase 9: PLSB Artifact Generator
|
|
4
|
+
*
|
|
5
|
+
* Generates the versioned PLSB v1.0 JSON artifact.
|
|
6
|
+
* This is a standalone, self-describing benchmark file
|
|
7
|
+
* suitable for external consumption (evaluators, CI, papers).
|
|
8
|
+
*/
|
|
9
|
+
var __createBinding = (this && this.__createBinding) || (Object.create ? (function(o, m, k, k2) {
|
|
10
|
+
if (k2 === undefined) k2 = k;
|
|
11
|
+
var desc = Object.getOwnPropertyDescriptor(m, k);
|
|
12
|
+
if (!desc || ("get" in desc ? !m.__esModule : desc.writable || desc.configurable)) {
|
|
13
|
+
desc = { enumerable: true, get: function() { return m[k]; } };
|
|
14
|
+
}
|
|
15
|
+
Object.defineProperty(o, k2, desc);
|
|
16
|
+
}) : (function(o, m, k, k2) {
|
|
17
|
+
if (k2 === undefined) k2 = k;
|
|
18
|
+
o[k2] = m[k];
|
|
19
|
+
}));
|
|
20
|
+
var __setModuleDefault = (this && this.__setModuleDefault) || (Object.create ? (function(o, v) {
|
|
21
|
+
Object.defineProperty(o, "default", { enumerable: true, value: v });
|
|
22
|
+
}) : function(o, v) {
|
|
23
|
+
o["default"] = v;
|
|
24
|
+
});
|
|
25
|
+
var __importStar = (this && this.__importStar) || (function () {
|
|
26
|
+
var ownKeys = function(o) {
|
|
27
|
+
ownKeys = Object.getOwnPropertyNames || function (o) {
|
|
28
|
+
var ar = [];
|
|
29
|
+
for (var k in o) if (Object.prototype.hasOwnProperty.call(o, k)) ar[ar.length] = k;
|
|
30
|
+
return ar;
|
|
31
|
+
};
|
|
32
|
+
return ownKeys(o);
|
|
33
|
+
};
|
|
34
|
+
return function (mod) {
|
|
35
|
+
if (mod && mod.__esModule) return mod;
|
|
36
|
+
var result = {};
|
|
37
|
+
if (mod != null) for (var k = ownKeys(mod), i = 0; i < k.length; i++) if (k[i] !== "default") __createBinding(result, mod, k[i]);
|
|
38
|
+
__setModuleDefault(result, mod);
|
|
39
|
+
return result;
|
|
40
|
+
};
|
|
41
|
+
})();
|
|
42
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
43
|
+
exports.PLSB_ARTIFACT_PATH = exports.PLSB_SCHEMA_URI = exports.PLSB_PUBLIC_URI = void 0;
|
|
44
|
+
exports.generatePLSBArtifact = generatePLSBArtifact;
|
|
45
|
+
const fs = __importStar(require("fs"));
|
|
46
|
+
const path = __importStar(require("path"));
|
|
47
|
+
const crypto = __importStar(require("crypto"));
|
|
48
|
+
/** Public reference URI for PLSB v1.0 — fixed, citable URL */
|
|
49
|
+
exports.PLSB_PUBLIC_URI = "https://progmune.io/plsb/v1.0";
|
|
50
|
+
exports.PLSB_SCHEMA_URI = `${exports.PLSB_PUBLIC_URI}/schema.json`;
|
|
51
|
+
function generatePLSBArtifact(outputPath) {
|
|
52
|
+
const { buildPLSB, PROTOCOL_WEAKNESS_TAXONOMY } = require("../plsb-benchmark");
|
|
53
|
+
const benchmark = buildPLSB();
|
|
54
|
+
// Build a corpus hash from the benchmark content
|
|
55
|
+
const corpusHash = crypto
|
|
56
|
+
.createHash("sha256")
|
|
57
|
+
.update(JSON.stringify(benchmark))
|
|
58
|
+
.digest("hex")
|
|
59
|
+
.slice(0, 16);
|
|
60
|
+
const artifact = {
|
|
61
|
+
$schema: exports.PLSB_SCHEMA_URI,
|
|
62
|
+
"@id": exports.PLSB_PUBLIC_URI,
|
|
63
|
+
"@version": "1.0.0",
|
|
64
|
+
generated: new Date().toISOString(),
|
|
65
|
+
entries: (benchmark.entries || []).map((e) => ({
|
|
66
|
+
id: e.id,
|
|
67
|
+
pls_id: e.pls_id || undefined,
|
|
68
|
+
category: e.category,
|
|
69
|
+
severity: e.severity,
|
|
70
|
+
broken: e.broken || [],
|
|
71
|
+
expected: e.expected || [],
|
|
72
|
+
verified: e.verified || false,
|
|
73
|
+
source: e.source || "unknown",
|
|
74
|
+
cve: e.cve || undefined,
|
|
75
|
+
project: e.project || undefined,
|
|
76
|
+
notes: e.notes || undefined,
|
|
77
|
+
})),
|
|
78
|
+
benchmark: {
|
|
79
|
+
name: benchmark.name || "PLSB-100",
|
|
80
|
+
version: benchmark.version || "1.0",
|
|
81
|
+
taxonomy: PROTOCOL_WEAKNESS_TAXONOMY.map((t) => ({
|
|
82
|
+
id: t.id,
|
|
83
|
+
name: t.name,
|
|
84
|
+
category: t.category,
|
|
85
|
+
description: t.description,
|
|
86
|
+
example_broken: t.example_broken || [],
|
|
87
|
+
example_expected: t.example_expected || [],
|
|
88
|
+
})),
|
|
89
|
+
metadata: {
|
|
90
|
+
total: benchmark.metadata?.total || 0,
|
|
91
|
+
verified: benchmark.metadata?.verified || 0,
|
|
92
|
+
byCategory: benchmark.metadata?.byCategory || {},
|
|
93
|
+
byPLS: benchmark.metadata?.byPLS || {},
|
|
94
|
+
coverage: benchmark.metadata?.coverage || 0,
|
|
95
|
+
recall: benchmark.metadata?.recall || 0,
|
|
96
|
+
precision: benchmark.metadata?.precision || 0,
|
|
97
|
+
},
|
|
98
|
+
},
|
|
99
|
+
provenance: {
|
|
100
|
+
source: "progmune-runtime",
|
|
101
|
+
version: "3.2.0",
|
|
102
|
+
corpusHash,
|
|
103
|
+
},
|
|
104
|
+
};
|
|
105
|
+
// Write to disk
|
|
106
|
+
if (outputPath) {
|
|
107
|
+
const outDir = path.dirname(outputPath);
|
|
108
|
+
if (!fs.existsSync(outDir)) {
|
|
109
|
+
fs.mkdirSync(outDir, { recursive: true });
|
|
110
|
+
}
|
|
111
|
+
fs.writeFileSync(outputPath, JSON.stringify(artifact, null, 2), "utf-8");
|
|
112
|
+
}
|
|
113
|
+
return artifact;
|
|
114
|
+
}
|
|
115
|
+
/** Default output path for the PLSB artifact */
|
|
116
|
+
exports.PLSB_ARTIFACT_PATH = "benchmarks/plsb-v1.0.json";
|
package/dist/plsb/cli.js
ADDED
|
@@ -0,0 +1,71 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
/**
|
|
3
|
+
* Phase 9: PLSB CLI
|
|
4
|
+
*
|
|
5
|
+
* Usage:
|
|
6
|
+
* npx ts-node src/plsb/cli.ts --export write plsb-v1.0.json
|
|
7
|
+
* npx ts-node src/plsb/cli.ts --report write plsb-report.md
|
|
8
|
+
* npx ts-node src/plsb/cli.ts --all both
|
|
9
|
+
* npx ts-node src/plsb/cli.ts --summary terminal summary
|
|
10
|
+
*/
|
|
11
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
12
|
+
const artifact_1 = require("./artifact");
|
|
13
|
+
const schema_1 = require("./schema");
|
|
14
|
+
const report_md_1 = require("./report-md");
|
|
15
|
+
const leaderboard_1 = require("./leaderboard");
|
|
16
|
+
const args = process.argv.slice(2);
|
|
17
|
+
if (args.includes("--help") || args.includes("-h") || args.length === 0) {
|
|
18
|
+
console.log(`
|
|
19
|
+
PLSB CLI — Protocol Lifecycle Security Benchmark
|
|
20
|
+
|
|
21
|
+
Usage:
|
|
22
|
+
npx ts-node src/plsb/cli.ts [options]
|
|
23
|
+
|
|
24
|
+
Options:
|
|
25
|
+
--export Generate ${artifact_1.PLSB_ARTIFACT_PATH}
|
|
26
|
+
--report Generate ${report_md_1.PLSB_REPORT_PATH}
|
|
27
|
+
--all Generate both artifact and report
|
|
28
|
+
--leaderboard Generate PLSB Leaderboard (JSON + Markdown)
|
|
29
|
+
--summary Print terminal summary
|
|
30
|
+
--help, -h Show this help
|
|
31
|
+
`);
|
|
32
|
+
process.exit(0);
|
|
33
|
+
}
|
|
34
|
+
const doExport = args.includes("--export") || args.includes("--all");
|
|
35
|
+
const doReport = args.includes("--report") || args.includes("--all");
|
|
36
|
+
const doSummary = args.includes("--summary") || (!doExport && !doReport);
|
|
37
|
+
if (args.includes("--leaderboard")) {
|
|
38
|
+
(0, leaderboard_1.exportLeaderboard)();
|
|
39
|
+
process.exit(0);
|
|
40
|
+
}
|
|
41
|
+
if (doExport) {
|
|
42
|
+
const artifact = (0, artifact_1.generatePLSBArtifact)(artifact_1.PLSB_ARTIFACT_PATH);
|
|
43
|
+
(0, schema_1.generatePLSBSchema)(schema_1.PLSB_SCHEMA_PATH);
|
|
44
|
+
console.log(`✅ Exported: ${artifact_1.PLSB_ARTIFACT_PATH}`);
|
|
45
|
+
console.log(`✅ Schema: ${schema_1.PLSB_SCHEMA_PATH}`);
|
|
46
|
+
console.log(` Public URI: ${artifact["@id"]}`);
|
|
47
|
+
console.log(` ${artifact.benchmark.metadata.total} entries, ${Object.keys(artifact.benchmark.metadata.byPLS).length}/${artifact.benchmark.taxonomy.length} categories covered`);
|
|
48
|
+
}
|
|
49
|
+
if (doReport) {
|
|
50
|
+
(0, report_md_1.generatePLSBReportMarkdown)(report_md_1.PLSB_REPORT_PATH);
|
|
51
|
+
console.log(`✅ Report: ${report_md_1.PLSB_REPORT_PATH}`);
|
|
52
|
+
}
|
|
53
|
+
if (doSummary) {
|
|
54
|
+
const { buildPLSB, PROTOCOL_WEAKNESS_TAXONOMY } = require("../plsb-benchmark");
|
|
55
|
+
const benchmark = buildPLSB();
|
|
56
|
+
const taxonomy = PROTOCOL_WEAKNESS_TAXONOMY;
|
|
57
|
+
const byPLS = benchmark.metadata?.byPLS || {};
|
|
58
|
+
console.log("");
|
|
59
|
+
console.log("PLSB v1.0 — Protocol Lifecycle Security Benchmark");
|
|
60
|
+
console.log("==================================================");
|
|
61
|
+
console.log(` Entries: ${benchmark.metadata?.total || 0} (${benchmark.metadata?.verified || 0} verified)`);
|
|
62
|
+
console.log(` Recall: ${((benchmark.metadata?.recall || 0) * 100).toFixed(0)}%`);
|
|
63
|
+
console.log(` Precision: ${((benchmark.metadata?.precision || 0) * 100).toFixed(0)}%`);
|
|
64
|
+
console.log("");
|
|
65
|
+
for (const t of taxonomy) {
|
|
66
|
+
const count = byPLS[t.id] || 0;
|
|
67
|
+
const icon = count > 0 ? "✅" : "⚠️";
|
|
68
|
+
console.log(` ${icon} ${t.id} ${t.name.padEnd(22)} ${t.category.padEnd(18)} ${count} entries`);
|
|
69
|
+
}
|
|
70
|
+
console.log("");
|
|
71
|
+
}
|
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
/**
|
|
3
|
+
* Phase 9: PLSB Productization Module
|
|
4
|
+
*
|
|
5
|
+
* Public API for PLSB v1.0 artifact and report generation.
|
|
6
|
+
*/
|
|
7
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
8
|
+
exports.PLSB_REPORT_PATH = exports.generatePLSBReportMarkdown = exports.PLSB_SCHEMA_PATH = exports.generatePLSBSchema = exports.PLSB_SCHEMA_URI = exports.PLSB_PUBLIC_URI = exports.PLSB_ARTIFACT_PATH = exports.generatePLSBArtifact = void 0;
|
|
9
|
+
var artifact_1 = require("./artifact");
|
|
10
|
+
Object.defineProperty(exports, "generatePLSBArtifact", { enumerable: true, get: function () { return artifact_1.generatePLSBArtifact; } });
|
|
11
|
+
Object.defineProperty(exports, "PLSB_ARTIFACT_PATH", { enumerable: true, get: function () { return artifact_1.PLSB_ARTIFACT_PATH; } });
|
|
12
|
+
Object.defineProperty(exports, "PLSB_PUBLIC_URI", { enumerable: true, get: function () { return artifact_1.PLSB_PUBLIC_URI; } });
|
|
13
|
+
Object.defineProperty(exports, "PLSB_SCHEMA_URI", { enumerable: true, get: function () { return artifact_1.PLSB_SCHEMA_URI; } });
|
|
14
|
+
var schema_1 = require("./schema");
|
|
15
|
+
Object.defineProperty(exports, "generatePLSBSchema", { enumerable: true, get: function () { return schema_1.generatePLSBSchema; } });
|
|
16
|
+
Object.defineProperty(exports, "PLSB_SCHEMA_PATH", { enumerable: true, get: function () { return schema_1.PLSB_SCHEMA_PATH; } });
|
|
17
|
+
var report_md_1 = require("./report-md");
|
|
18
|
+
Object.defineProperty(exports, "generatePLSBReportMarkdown", { enumerable: true, get: function () { return report_md_1.generatePLSBReportMarkdown; } });
|
|
19
|
+
Object.defineProperty(exports, "PLSB_REPORT_PATH", { enumerable: true, get: function () { return report_md_1.PLSB_REPORT_PATH; } });
|
|
@@ -0,0 +1,249 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
/**
|
|
3
|
+
* Phase 12: PLSB Leaderboard
|
|
4
|
+
*
|
|
5
|
+
* Public comparison of protocol lifecycle security detection tools.
|
|
6
|
+
* The market believes rankings, not papers.
|
|
7
|
+
*
|
|
8
|
+
* Generates a transparent, reproducible leaderboard:
|
|
9
|
+
* - Progmune's real scores from buildPLSB()
|
|
10
|
+
* - Placeholder rows for other tools with methodology notes
|
|
11
|
+
* - JSON artifact + Markdown + HTML
|
|
12
|
+
*
|
|
13
|
+
* Usage:
|
|
14
|
+
* npx ts-node src/plsb/cli.ts --leaderboard
|
|
15
|
+
*/
|
|
16
|
+
var __createBinding = (this && this.__createBinding) || (Object.create ? (function(o, m, k, k2) {
|
|
17
|
+
if (k2 === undefined) k2 = k;
|
|
18
|
+
var desc = Object.getOwnPropertyDescriptor(m, k);
|
|
19
|
+
if (!desc || ("get" in desc ? !m.__esModule : desc.writable || desc.configurable)) {
|
|
20
|
+
desc = { enumerable: true, get: function() { return m[k]; } };
|
|
21
|
+
}
|
|
22
|
+
Object.defineProperty(o, k2, desc);
|
|
23
|
+
}) : (function(o, m, k, k2) {
|
|
24
|
+
if (k2 === undefined) k2 = k;
|
|
25
|
+
o[k2] = m[k];
|
|
26
|
+
}));
|
|
27
|
+
var __setModuleDefault = (this && this.__setModuleDefault) || (Object.create ? (function(o, v) {
|
|
28
|
+
Object.defineProperty(o, "default", { enumerable: true, value: v });
|
|
29
|
+
}) : function(o, v) {
|
|
30
|
+
o["default"] = v;
|
|
31
|
+
});
|
|
32
|
+
var __importStar = (this && this.__importStar) || (function () {
|
|
33
|
+
var ownKeys = function(o) {
|
|
34
|
+
ownKeys = Object.getOwnPropertyNames || function (o) {
|
|
35
|
+
var ar = [];
|
|
36
|
+
for (var k in o) if (Object.prototype.hasOwnProperty.call(o, k)) ar[ar.length] = k;
|
|
37
|
+
return ar;
|
|
38
|
+
};
|
|
39
|
+
return ownKeys(o);
|
|
40
|
+
};
|
|
41
|
+
return function (mod) {
|
|
42
|
+
if (mod && mod.__esModule) return mod;
|
|
43
|
+
var result = {};
|
|
44
|
+
if (mod != null) for (var k = ownKeys(mod), i = 0; i < k.length; i++) if (k[i] !== "default") __createBinding(result, mod, k[i]);
|
|
45
|
+
__setModuleDefault(result, mod);
|
|
46
|
+
return result;
|
|
47
|
+
};
|
|
48
|
+
})();
|
|
49
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
50
|
+
exports.LEADERBOARD_MD_PATH = exports.LEADERBOARD_JSON_PATH = void 0;
|
|
51
|
+
exports.generateLeaderboard = generateLeaderboard;
|
|
52
|
+
exports.formatLeaderboardMarkdown = formatLeaderboardMarkdown;
|
|
53
|
+
exports.formatLeaderboardJSON = formatLeaderboardJSON;
|
|
54
|
+
exports.exportLeaderboard = exportLeaderboard;
|
|
55
|
+
const fs = __importStar(require("fs"));
|
|
56
|
+
const path = __importStar(require("path"));
|
|
57
|
+
// ═══════════════════════════════════════════════════════════════
|
|
58
|
+
// Generator
|
|
59
|
+
// ═══════════════════════════════════════════════════════════════
|
|
60
|
+
function generateLeaderboard() {
|
|
61
|
+
const { buildPLSB, PROTOCOL_WEAKNESS_TAXONOMY } = require("../plsb-benchmark");
|
|
62
|
+
const benchmark = buildPLSB();
|
|
63
|
+
const taxonomy = PROTOCOL_WEAKNESS_TAXONOMY;
|
|
64
|
+
const byPLS = benchmark.metadata?.byPLS || {};
|
|
65
|
+
const covered = taxonomy.filter((t) => (byPLS[t.id] || 0) > 0).length;
|
|
66
|
+
const totalEntries = benchmark.metadata?.total || 0;
|
|
67
|
+
const verifiedEntries = benchmark.metadata?.verified || 0;
|
|
68
|
+
// Progmune's real scores
|
|
69
|
+
const progmuneEntry = {
|
|
70
|
+
tool: "Progmune Runtime",
|
|
71
|
+
version: "3.2.0",
|
|
72
|
+
vendor: "Progmune (Open Source)",
|
|
73
|
+
score: 0, // F1 requires hand-labeled ground truth — will be filled when .progmune_labels.json exists
|
|
74
|
+
recall: 0,
|
|
75
|
+
precision: 0,
|
|
76
|
+
categoriesCovered: covered,
|
|
77
|
+
categoriesTotal: taxonomy.length,
|
|
78
|
+
entriesTested: totalEntries,
|
|
79
|
+
benchmarkDate: new Date().toISOString().slice(0, 10),
|
|
80
|
+
status: "benchmarked",
|
|
81
|
+
methodology: "SSG state machine validation + auto-discovered protocol rules. Score requires hand-labeled ground truth per repo (.progmune_labels.json).",
|
|
82
|
+
artifacts: "benchmarks/plsb-v1.0.json",
|
|
83
|
+
notes: verifiedEntries > 0
|
|
84
|
+
? `${verifiedEntries} verified gold entries, ${covered}/${taxonomy.length} categories covered. Precision requires annotated labels per target repository.`
|
|
85
|
+
: "Gold dataset ready. Multi-repo labeling in progress.",
|
|
86
|
+
};
|
|
87
|
+
// Comparison tools — not yet independently benchmarked
|
|
88
|
+
const comparisonEntries = [
|
|
89
|
+
{
|
|
90
|
+
tool: "CodeQL",
|
|
91
|
+
version: "—",
|
|
92
|
+
vendor: "GitHub / Microsoft",
|
|
93
|
+
score: 0, recall: 0, precision: 0,
|
|
94
|
+
categoriesCovered: 0, categoriesTotal: taxonomy.length, entriesTested: 0,
|
|
95
|
+
benchmarkDate: "—",
|
|
96
|
+
status: "not_benchmarked",
|
|
97
|
+
methodology: "CodeQL detects data flow and taint patterns but does not model protocol state machines. Protocol lifecycle violations (missing release, double commit, session fixation) fall outside its taint-tracking model. Benchmark integration requires writing custom CodeQL queries for each PLS category.",
|
|
98
|
+
notes: "Custom query pack needed to map PLS-001 through PLS-013 to CodeQL predicates.",
|
|
99
|
+
},
|
|
100
|
+
{
|
|
101
|
+
tool: "Semgrep",
|
|
102
|
+
version: "—",
|
|
103
|
+
vendor: "Semgrep, Inc.",
|
|
104
|
+
score: 0, recall: 0, precision: 0,
|
|
105
|
+
categoriesCovered: 0, categoriesTotal: taxonomy.length, entriesTested: 0,
|
|
106
|
+
benchmarkDate: "—",
|
|
107
|
+
status: "not_benchmarked",
|
|
108
|
+
methodology: "Semgrep matches AST patterns but lacks stateful sequence analysis. Can detect missing function calls only if written as explicit pattern pairs. Protocol lifecycle detection requires cross-function state tracking not available in pattern-matching mode.",
|
|
109
|
+
notes: "Semgrep's taint mode may partially cover PLS-004 (auth bypass) but cannot model acquire→use→release chains.",
|
|
110
|
+
},
|
|
111
|
+
{
|
|
112
|
+
tool: "Snyk Code",
|
|
113
|
+
version: "—",
|
|
114
|
+
vendor: "Snyk Ltd.",
|
|
115
|
+
score: 0, recall: 0, precision: 0,
|
|
116
|
+
categoriesCovered: 0, categoriesTotal: taxonomy.length, entriesTested: 0,
|
|
117
|
+
benchmarkDate: "—",
|
|
118
|
+
status: "not_benchmarked",
|
|
119
|
+
methodology: "Snyk Code uses ML-based vulnerability detection trained on known CVE patterns. May implicitly learn some protocol patterns from training data, but provides no deterministic protocol state machine guarantees.",
|
|
120
|
+
notes: "Black-box evaluation would require running Snyk Code against PLSB benchmark cases and comparing detection results.",
|
|
121
|
+
},
|
|
122
|
+
{
|
|
123
|
+
tool: "Checkmarx SAST",
|
|
124
|
+
version: "—",
|
|
125
|
+
vendor: "Checkmarx Ltd.",
|
|
126
|
+
score: 0, recall: 0, precision: 0,
|
|
127
|
+
categoriesCovered: 0, categoriesTotal: taxonomy.length, entriesTested: 0,
|
|
128
|
+
benchmarkDate: "—",
|
|
129
|
+
status: "not_benchmarked",
|
|
130
|
+
methodology: "Checkmarx provides data flow analysis but does not explicitly model protocol state machines. Requires custom query configuration for cross-functional sequence detection.",
|
|
131
|
+
notes: "Undergoing acquisition by Haveli Investments. Enterprise SAST with custom query support.",
|
|
132
|
+
},
|
|
133
|
+
{
|
|
134
|
+
tool: "Copilot Code Review",
|
|
135
|
+
version: "—",
|
|
136
|
+
vendor: "GitHub / Microsoft",
|
|
137
|
+
score: 0, recall: 0, precision: 0,
|
|
138
|
+
categoriesCovered: 0, categoriesTotal: taxonomy.length, entriesTested: 0,
|
|
139
|
+
benchmarkDate: "—",
|
|
140
|
+
status: "not_benchmarked",
|
|
141
|
+
methodology: "LLM-based code review can identify individual missing function calls but lacks deterministic protocol state machine enforcement. Detection depends on prompt engineering and model behavior, not verifiable rules.",
|
|
142
|
+
notes: "LLM-based review is probabilistic. PLSB requires deterministic detection for governance certification.",
|
|
143
|
+
},
|
|
144
|
+
];
|
|
145
|
+
const entries = [progmuneEntry, ...comparisonEntries];
|
|
146
|
+
return {
|
|
147
|
+
name: "PLSB Leaderboard",
|
|
148
|
+
version: "1.0.0",
|
|
149
|
+
generated: new Date().toISOString(),
|
|
150
|
+
benchmarkVersion: "v1.0",
|
|
151
|
+
plsbUri: "https://progmune.io/plsb/v1.0",
|
|
152
|
+
entries,
|
|
153
|
+
methodology: {
|
|
154
|
+
description: "Protocol Lifecycle Security Benchmark — measures each tool's ability to detect protocol state machine violations (missing states, illegal transitions, lifecycle breaks) that traditional SAST cannot see.",
|
|
155
|
+
benchmark: "13-category Protocol Weakness Taxonomy (PLS-001 through PLS-013). Gold dataset: 25 manually-verified real-world defect cases.",
|
|
156
|
+
metric: "F1 score = harmonic mean of precision and recall. Precision = true positives / (true positives + false positives). Recall = true positives / (true positives + false negatives).",
|
|
157
|
+
categories: "resource_leak, use_after_free, auth_bypass, session_fixation, privilege_escalation, transaction_violation, double_free, race_condition, missing_validation",
|
|
158
|
+
},
|
|
159
|
+
};
|
|
160
|
+
}
|
|
161
|
+
// ═══════════════════════════════════════════════════════════════
|
|
162
|
+
// Formatters
|
|
163
|
+
// ═══════════════════════════════════════════════════════════════
|
|
164
|
+
function formatLeaderboardMarkdown(lb) {
|
|
165
|
+
const lines = [];
|
|
166
|
+
lines.push("# PLSB Leaderboard — Protocol Lifecycle Security Detection");
|
|
167
|
+
lines.push("");
|
|
168
|
+
lines.push(`**Version:** ${lb.benchmarkVersion} | **Generated:** ${lb.generated}`);
|
|
169
|
+
lines.push("");
|
|
170
|
+
lines.push(`> ${lb.methodology.description}`);
|
|
171
|
+
lines.push("");
|
|
172
|
+
lines.push("## Scores");
|
|
173
|
+
lines.push("");
|
|
174
|
+
lines.push(`| # | Tool | Score (F1) | Recall | Precision | Categories | Status |`);
|
|
175
|
+
lines.push(`|---|------|-----------|--------|-----------|------------|--------|`);
|
|
176
|
+
for (let i = 0; i < lb.entries.length; i++) {
|
|
177
|
+
const e = lb.entries[i];
|
|
178
|
+
const scoreStr = e.status === "benchmarked" && e.score > 0
|
|
179
|
+
? `${e.score}%`
|
|
180
|
+
: e.status === "benchmarked"
|
|
181
|
+
? `⚠️ needs labels`
|
|
182
|
+
: "—";
|
|
183
|
+
const recallStr = e.recall > 0 ? `${e.recall}%` : "—";
|
|
184
|
+
const precisionStr = e.precision > 0 ? `${e.precision}%` : "—";
|
|
185
|
+
const catStr = e.categoriesCovered > 0
|
|
186
|
+
? `${e.categoriesCovered}/${e.categoriesTotal}`
|
|
187
|
+
: "—";
|
|
188
|
+
const statusStr = e.status === "benchmarked" ? "✅" : "⏳";
|
|
189
|
+
lines.push(`| ${i + 1} | **${e.tool}** | ${scoreStr} | ${recallStr} | ${precisionStr} | ${catStr} | ${statusStr} |`);
|
|
190
|
+
}
|
|
191
|
+
lines.push("");
|
|
192
|
+
// Methodology per tool
|
|
193
|
+
lines.push("## Methodology Notes");
|
|
194
|
+
lines.push("");
|
|
195
|
+
for (const e of lb.entries) {
|
|
196
|
+
lines.push(`### ${e.tool} (${e.vendor})`);
|
|
197
|
+
lines.push("");
|
|
198
|
+
if (e.status === "benchmarked") {
|
|
199
|
+
lines.push(`- **Status:** Benchmarked on PLSB v1.0 gold dataset`);
|
|
200
|
+
lines.push(`- **Entries tested:** ${e.entriesTested}`);
|
|
201
|
+
lines.push(`- **Method:** ${e.methodology}`);
|
|
202
|
+
}
|
|
203
|
+
else {
|
|
204
|
+
lines.push(`- **Status:** Not yet benchmarked on PLSB`);
|
|
205
|
+
lines.push(`- **Why not:** ${e.methodology}`);
|
|
206
|
+
}
|
|
207
|
+
if (e.notes)
|
|
208
|
+
lines.push(`- **Note:** ${e.notes}`);
|
|
209
|
+
lines.push("");
|
|
210
|
+
}
|
|
211
|
+
// How to contribute
|
|
212
|
+
lines.push("## Contributing");
|
|
213
|
+
lines.push("");
|
|
214
|
+
lines.push("Tool vendors and researchers can submit benchmark results by:");
|
|
215
|
+
lines.push("");
|
|
216
|
+
lines.push("1. Cloning [progmune-runtime](https://github.com/shenlian19831109/progmune-runtime)");
|
|
217
|
+
lines.push("2. Running `npm run plsb:export` to get `benchmarks/plsb-v1.0.json`");
|
|
218
|
+
lines.push("3. Testing their tool against the 25 gold cases (13 categories)");
|
|
219
|
+
lines.push("4. Submitting results as a PR to this leaderboard");
|
|
220
|
+
lines.push("");
|
|
221
|
+
lines.push("All submissions must include reproducible methodology and raw detection logs.");
|
|
222
|
+
lines.push("");
|
|
223
|
+
lines.push("---");
|
|
224
|
+
lines.push(`*Leaderboard generated by [Progmune Runtime](https://github.com/shenlian19831109/progmune-runtime)*`);
|
|
225
|
+
lines.push("");
|
|
226
|
+
return lines.join("\n");
|
|
227
|
+
}
|
|
228
|
+
function formatLeaderboardJSON(lb) {
|
|
229
|
+
return JSON.stringify(lb, null, 2);
|
|
230
|
+
}
|
|
231
|
+
// ═══════════════════════════════════════════════════════════════
|
|
232
|
+
// Export
|
|
233
|
+
// ═══════════════════════════════════════════════════════════════
|
|
234
|
+
exports.LEADERBOARD_JSON_PATH = "benchmarks/plsb-leaderboard.json";
|
|
235
|
+
exports.LEADERBOARD_MD_PATH = "benchmarks/plsb-leaderboard.md";
|
|
236
|
+
function exportLeaderboard(outputDir) {
|
|
237
|
+
const lb = generateLeaderboard();
|
|
238
|
+
const dir = outputDir || "benchmarks";
|
|
239
|
+
if (!fs.existsSync(dir))
|
|
240
|
+
fs.mkdirSync(dir, { recursive: true });
|
|
241
|
+
const jsonPath = path.join(dir, "plsb-leaderboard.json");
|
|
242
|
+
const mdPath = path.join(dir, "plsb-leaderboard.md");
|
|
243
|
+
fs.writeFileSync(jsonPath, formatLeaderboardJSON(lb), "utf-8");
|
|
244
|
+
fs.writeFileSync(mdPath, formatLeaderboardMarkdown(lb), "utf-8");
|
|
245
|
+
console.error(`✅ Leaderboard JSON: ${jsonPath}`);
|
|
246
|
+
console.error(`✅ Leaderboard MD: ${mdPath}`);
|
|
247
|
+
console.error(` Tools: ${lb.entries.length} (1 benchmarked, ${lb.entries.length - 1} awaiting)`);
|
|
248
|
+
return { json: jsonPath, md: mdPath };
|
|
249
|
+
}
|
|
@@ -0,0 +1,156 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
/**
|
|
3
|
+
* Phase 9: PLSB Markdown Report Generator
|
|
4
|
+
*
|
|
5
|
+
* Generates a self-contained markdown report for the PLSB benchmark.
|
|
6
|
+
* Suitable for human stakeholders, CI artifacts, and PDF conversion.
|
|
7
|
+
*/
|
|
8
|
+
var __createBinding = (this && this.__createBinding) || (Object.create ? (function(o, m, k, k2) {
|
|
9
|
+
if (k2 === undefined) k2 = k;
|
|
10
|
+
var desc = Object.getOwnPropertyDescriptor(m, k);
|
|
11
|
+
if (!desc || ("get" in desc ? !m.__esModule : desc.writable || desc.configurable)) {
|
|
12
|
+
desc = { enumerable: true, get: function() { return m[k]; } };
|
|
13
|
+
}
|
|
14
|
+
Object.defineProperty(o, k2, desc);
|
|
15
|
+
}) : (function(o, m, k, k2) {
|
|
16
|
+
if (k2 === undefined) k2 = k;
|
|
17
|
+
o[k2] = m[k];
|
|
18
|
+
}));
|
|
19
|
+
var __setModuleDefault = (this && this.__setModuleDefault) || (Object.create ? (function(o, v) {
|
|
20
|
+
Object.defineProperty(o, "default", { enumerable: true, value: v });
|
|
21
|
+
}) : function(o, v) {
|
|
22
|
+
o["default"] = v;
|
|
23
|
+
});
|
|
24
|
+
var __importStar = (this && this.__importStar) || (function () {
|
|
25
|
+
var ownKeys = function(o) {
|
|
26
|
+
ownKeys = Object.getOwnPropertyNames || function (o) {
|
|
27
|
+
var ar = [];
|
|
28
|
+
for (var k in o) if (Object.prototype.hasOwnProperty.call(o, k)) ar[ar.length] = k;
|
|
29
|
+
return ar;
|
|
30
|
+
};
|
|
31
|
+
return ownKeys(o);
|
|
32
|
+
};
|
|
33
|
+
return function (mod) {
|
|
34
|
+
if (mod && mod.__esModule) return mod;
|
|
35
|
+
var result = {};
|
|
36
|
+
if (mod != null) for (var k = ownKeys(mod), i = 0; i < k.length; i++) if (k[i] !== "default") __createBinding(result, mod, k[i]);
|
|
37
|
+
__setModuleDefault(result, mod);
|
|
38
|
+
return result;
|
|
39
|
+
};
|
|
40
|
+
})();
|
|
41
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
42
|
+
exports.PLSB_REPORT_PATH = void 0;
|
|
43
|
+
exports.generatePLSBReportMarkdown = generatePLSBReportMarkdown;
|
|
44
|
+
const fs = __importStar(require("fs"));
|
|
45
|
+
const path = __importStar(require("path"));
|
|
46
|
+
function generatePLSBReportMarkdown(outputPath) {
|
|
47
|
+
const { buildPLSB, PROTOCOL_WEAKNESS_TAXONOMY } = require("../plsb-benchmark");
|
|
48
|
+
const benchmark = buildPLSB();
|
|
49
|
+
const taxonomy = PROTOCOL_WEAKNESS_TAXONOMY;
|
|
50
|
+
const lines = [];
|
|
51
|
+
lines.push("# PLSB Report: Protocol Lifecycle Security Benchmark");
|
|
52
|
+
lines.push("");
|
|
53
|
+
lines.push(`**Version:** ${benchmark.version || "1.0"}`);
|
|
54
|
+
lines.push(`**Generated:** ${new Date().toISOString()}`);
|
|
55
|
+
lines.push(`**Source:** Progmune Runtime v3.2.0`);
|
|
56
|
+
lines.push("");
|
|
57
|
+
// ── Summary ──
|
|
58
|
+
lines.push("## Summary");
|
|
59
|
+
lines.push("");
|
|
60
|
+
lines.push(`| Metric | Value |`);
|
|
61
|
+
lines.push(`|--------|-------|`);
|
|
62
|
+
lines.push(`| Total Entries | ${benchmark.metadata?.total || 0} |`);
|
|
63
|
+
lines.push(`| Verified Entries | ${benchmark.metadata?.verified || 0} |`);
|
|
64
|
+
lines.push(`| Recall | ${((benchmark.metadata?.recall || 0) * 100).toFixed(0)}% |`);
|
|
65
|
+
lines.push(`| Precision | ${((benchmark.metadata?.precision || 0) * 100).toFixed(0)}% |`);
|
|
66
|
+
lines.push(`| Categories Covered | ${Object.keys(benchmark.metadata?.byPLS || {}).length} / ${taxonomy.length} |`);
|
|
67
|
+
lines.push("");
|
|
68
|
+
// ── Taxonomy Grid ──
|
|
69
|
+
lines.push("## Protocol Weakness Taxonomy");
|
|
70
|
+
lines.push("");
|
|
71
|
+
lines.push(`| PLS-ID | Name | Category | Entries | Status |`);
|
|
72
|
+
lines.push(`|--------|------|----------|---------|--------|`);
|
|
73
|
+
const byPLS = benchmark.metadata?.byPLS || {};
|
|
74
|
+
for (const t of taxonomy) {
|
|
75
|
+
const count = byPLS[t.id] || 0;
|
|
76
|
+
const status = count > 0 ? "✅ Covered" : "⚠️ Uncovered";
|
|
77
|
+
lines.push(`| ${t.id} | ${t.name} | ${t.category} | ${count} | ${status} |`);
|
|
78
|
+
}
|
|
79
|
+
lines.push("");
|
|
80
|
+
// ── Entries Table ──
|
|
81
|
+
const entries = benchmark.entries || [];
|
|
82
|
+
if (entries.length > 0) {
|
|
83
|
+
lines.push("## Entries");
|
|
84
|
+
lines.push("");
|
|
85
|
+
lines.push(`| ID | PLS | Category | Severity | Verified | CVE / Source |`);
|
|
86
|
+
lines.push(`|----|-----|----------|----------|----------|-------------|`);
|
|
87
|
+
for (const e of entries.slice(0, 30)) {
|
|
88
|
+
const cveRef = e.cve ? `CVE-${e.cve}` : e.source || "—";
|
|
89
|
+
lines.push(`| ${e.id} | ${e.pls_id || "—"} | ${e.category} | ${e.severity} | ${e.verified ? "✅" : "—"} | ${cveRef} |`);
|
|
90
|
+
}
|
|
91
|
+
if (entries.length > 30) {
|
|
92
|
+
lines.push(`| ... | ... | ${entries.length - 30} more entries ... |`);
|
|
93
|
+
}
|
|
94
|
+
lines.push("");
|
|
95
|
+
}
|
|
96
|
+
// ── Per-Category Detail ──
|
|
97
|
+
lines.push("## Category Breakdown");
|
|
98
|
+
lines.push("");
|
|
99
|
+
const byCategory = benchmark.metadata?.byCategory || {};
|
|
100
|
+
const categories = [...new Set(taxonomy.map((t) => t.category))];
|
|
101
|
+
for (const cat of categories) {
|
|
102
|
+
const count = byCategory[cat] || 0;
|
|
103
|
+
const plsInCat = taxonomy.filter((t) => t.category === cat).map((t) => t.id);
|
|
104
|
+
lines.push(`### ${cat} (${count} entries)`);
|
|
105
|
+
lines.push(`Covered PLS: ${plsInCat.filter((id) => (byPLS[id] || 0) > 0).join(", ") || "none"}`);
|
|
106
|
+
lines.push("");
|
|
107
|
+
}
|
|
108
|
+
// ── Detector Performance ──
|
|
109
|
+
lines.push("## Detector Performance");
|
|
110
|
+
lines.push("");
|
|
111
|
+
const recall = (benchmark.metadata?.recall || 0) * 100;
|
|
112
|
+
const precision = (benchmark.metadata?.precision || 0) * 100;
|
|
113
|
+
lines.push(`| Metric | Value | Rating |`);
|
|
114
|
+
lines.push(`|--------|-------|--------|`);
|
|
115
|
+
lines.push(`| Recall | ${recall.toFixed(0)}% | ${recall > 85 ? "✅ Excellent" : recall > 70 ? "⚠️ Good" : "❌ Needs Improvement"} |`);
|
|
116
|
+
lines.push(`| Precision | ${precision.toFixed(0)}% | ${precision > 80 ? "✅ Excellent" : precision > 60 ? "⚠️ Good" : "❌ Needs Improvement"} |`);
|
|
117
|
+
lines.push("");
|
|
118
|
+
// ── Certification ──
|
|
119
|
+
lines.push("## Certification");
|
|
120
|
+
lines.push("");
|
|
121
|
+
lines.push("This benchmark was generated by **Progmune Runtime** — an AI-generated software governance system.");
|
|
122
|
+
lines.push("");
|
|
123
|
+
lines.push("- **Detector:** Protocol Lifecycle Security (13-category taxonomy)");
|
|
124
|
+
lines.push("- **Methodology:** State machine fingerprint comparison (structural violation detection)");
|
|
125
|
+
lines.push("- **Gold Standard:** 20 manually-verified real-world defect cases");
|
|
126
|
+
lines.push("");
|
|
127
|
+
if (recall > 85) {
|
|
128
|
+
lines.push("### Verdict: CERTIFIED ✅");
|
|
129
|
+
lines.push("");
|
|
130
|
+
lines.push("The PLSB detector exceeds the 85% recall threshold for protocol lifecycle vulnerability detection.");
|
|
131
|
+
}
|
|
132
|
+
else if (recall > 70) {
|
|
133
|
+
lines.push("### Verdict: PROMISING ⚠️");
|
|
134
|
+
lines.push("");
|
|
135
|
+
lines.push("The PLSB detector exceeds 70% recall. Additional gold cases are needed for uncovered categories.");
|
|
136
|
+
}
|
|
137
|
+
else {
|
|
138
|
+
lines.push("### Verdict: IN DEVELOPMENT 🔬");
|
|
139
|
+
lines.push("");
|
|
140
|
+
lines.push("The PLSB detector is under active development. Recall is below 70% — more protocol rules and gold cases needed.");
|
|
141
|
+
}
|
|
142
|
+
lines.push("");
|
|
143
|
+
lines.push("---");
|
|
144
|
+
lines.push("*Generated by [Progmune Runtime](https://github.com/shenlian19831109/progmune-runtime)*");
|
|
145
|
+
const output = lines.join("\n");
|
|
146
|
+
// Write to disk
|
|
147
|
+
if (outputPath) {
|
|
148
|
+
const outDir = path.dirname(outputPath);
|
|
149
|
+
if (!fs.existsSync(outDir)) {
|
|
150
|
+
fs.mkdirSync(outDir, { recursive: true });
|
|
151
|
+
}
|
|
152
|
+
fs.writeFileSync(outputPath, output, "utf-8");
|
|
153
|
+
}
|
|
154
|
+
return output;
|
|
155
|
+
}
|
|
156
|
+
exports.PLSB_REPORT_PATH = "benchmarks/plsb-report.md";
|