progmune-runtime 2.1.6 → 3.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +108 -468
- package/dist/ablation-study.js +144 -0
- package/dist/ablation-study.test.js +18 -0
- package/dist/action-runtime.js +3 -1
- package/dist/active-learning.js +211 -0
- package/dist/analytics.js +139 -0
- package/dist/asset-factory.js +309 -0
- package/dist/asset-growth.js +244 -0
- package/dist/asset-promotion.js +382 -0
- package/dist/asset-quality.js +550 -0
- package/dist/audit/business-translator.js +285 -0
- package/dist/audit/cli.js +66 -0
- package/dist/audit/formatters/html.js +379 -0
- package/dist/audit/formatters/json.js +11 -0
- package/dist/audit/formatters/markdown.js +192 -0
- package/dist/audit/formatters/terminal.js +189 -0
- package/dist/audit/index.js +25 -0
- package/dist/audit/report-builder.js +318 -0
- package/dist/audit/types.js +8 -0
- package/dist/audit.js +3 -3
- package/dist/auto-benchmark-generator.js +137 -0
- package/dist/auto-benchmark-generator.test.js +45 -0
- package/dist/auto-protocol-synthesizer.js +362 -0
- package/dist/auto-protocol-synthesizer.test.js +82 -0
- package/dist/autonomous-patch.js +175 -0
- package/dist/autonomous-patch.test.js +128 -0
- package/dist/badge/badge-server.js +98 -0
- package/dist/behavior-miner.js +442 -0
- package/dist/belief-layer.js +475 -0
- package/dist/benchmark-count.js +5 -0
- package/dist/benchmark-generator.js +211 -0
- package/dist/benchmark-harness.js +201 -0
- package/dist/benchmark-pass-rate.js +7 -0
- package/dist/benchmark-report.js +8 -3
- package/dist/benchmark-save.js +14 -1
- package/dist/bootstrap-validation.js +197 -0
- package/dist/bootstrap-validation.test.js +51 -0
- package/dist/branch-ledger.js +1 -1
- package/dist/capability-gap.js +130 -0
- package/dist/certify-html.js +351 -0
- package/dist/certify.js +326 -0
- package/dist/check.js +4 -4
- package/dist/compliance-miner.js +447 -0
- package/dist/continuous-benchmark.js +194 -0
- package/dist/continuous-benchmark.test.js +116 -0
- package/dist/corpus-stats.js +173 -0
- package/dist/counterfactual-engine.js +288 -0
- package/dist/coverage-dashboard.js +109 -0
- package/dist/coverage-system.test.js +205 -0
- package/dist/cross-repo-precision.js +352 -0
- package/dist/cve-benchmark.js +180 -0
- package/dist/cve-benchmark.test.js +28 -0
- package/dist/cve-collector.js +73 -0
- package/dist/data-quality.js +141 -0
- package/dist/decision-engine.js +388 -0
- package/dist/derive-metadata.js +250 -0
- package/dist/difficulty-active.test.js +198 -0
- package/dist/difficulty-map.js +244 -0
- package/dist/discovery-analytics.js +125 -0
- package/dist/discovery-model.js +149 -0
- package/dist/discovery-optimize.test.js +199 -0
- package/dist/discovery-trace.js +276 -0
- package/dist/discovery-trace.test.js +97 -0
- package/dist/emitter.js +83 -1
- package/dist/enterprise-dashboard.js +405 -0
- package/dist/eval-hardening.js +297 -0
- package/dist/eval-hardening.test.js +85 -0
- package/dist/evaluation-campaign.js +359 -0
- package/dist/evaluation-campaign.test.js +181 -0
- package/dist/evidence-growth.js +143 -0
- package/dist/evidence-repository.js +209 -0
- package/dist/evidence-system.js +441 -0
- package/dist/execute.js +15 -7
- package/dist/experimental/software-physics.js +291 -0
- package/dist/experimental/state-inference.js +516 -0
- package/dist/experimental/unsupervised-physics.js +230 -0
- package/dist/extract-ir-python.js +54 -7
- package/dist/extract-ir.js +376 -12
- package/dist/failure-collector.js +2 -2
- package/dist/failure-corpus.js +322 -9
- package/dist/feedback.js +16 -5
- package/dist/feedback.test.js +49 -0
- package/dist/file-lock.js +1 -1
- package/dist/flywheel-batch.js +292 -0
- package/dist/frameworks/express-cli.js +237 -0
- package/dist/frameworks/express-detector.js +445 -0
- package/dist/frameworks/express-detector.test.js +206 -0
- package/dist/frameworks/index.js +30 -0
- package/dist/frameworks/nestjs-detector.js +302 -0
- package/dist/frameworks/trpc-detector.js +161 -0
- package/dist/frameworks/version-awareness.js +179 -0
- package/dist/function-synonyms.js +164 -0
- package/dist/function-synonyms.test.js +68 -0
- package/dist/generalization.test.js +352 -0
- package/dist/goal-annotator.js +113 -0
- package/dist/goal-planner.js +563 -0
- package/dist/gold-cve.js +164 -0
- package/dist/gold-cve.test.js +104 -0
- package/dist/gold-quality.js +206 -0
- package/dist/gold-tiers.js +241 -0
- package/dist/governance-dashboard.js +327 -0
- package/dist/graph-viz.js +240 -0
- package/dist/guided-frontier.js +195 -0
- package/dist/hierarchical-planner.js +148 -0
- package/dist/identifier-parser.js +260 -0
- package/dist/immune-metrics.js +93 -0
- package/dist/immune-receiver.js +158 -0
- package/dist/immune-reporter.js +1 -1
- package/dist/improvement-orchestrator.js +206 -0
- package/dist/inject-p0-vocabulary.js +300 -0
- package/dist/intent-parser.js +218 -0
- package/dist/invariant-algebra.js +476 -0
- package/dist/invariant-calculus.js +533 -0
- package/dist/ir-utils.js +70 -0
- package/dist/ir-utils.test.js +50 -0
- package/dist/knowledge-api.js +312 -0
- package/dist/knowledge-evolution.js +452 -0
- package/dist/knowledge-explorer.js +506 -0
- package/dist/knowledge-flywheel.js +274 -0
- package/dist/knowledge-governance.js +338 -0
- package/dist/knowledge-governance.test.js +150 -0
- package/dist/knowledge-graph.js +181 -0
- package/dist/knowledge-guided-synth.js +246 -0
- package/dist/knowledge-loop.test.js +77 -0
- package/dist/knowledge-object.js +316 -0
- package/dist/knowledge-package.js +98 -0
- package/dist/kpi-dashboard.js +561 -0
- package/dist/l3-cross-function.js +280 -0
- package/dist/learning-ranker.js +148 -0
- package/dist/learning-ranker.test.js +291 -0
- package/dist/ledger/accountability.js +322 -0
- package/dist/ledger/chain-builder.js +185 -0
- package/dist/ledger/cli.js +222 -0
- package/dist/ledger/index.js +13 -0
- package/dist/ledger/signatures.js +193 -0
- package/dist/ledger/types.js +9 -0
- package/dist/llm.js +74 -3
- package/dist/load-benchmarks.js +8 -3
- package/dist/logger.js +66 -0
- package/dist/logger.test.js +37 -0
- package/dist/logistic-reward.js +339 -0
- package/dist/logistic-reward.test.js +180 -0
- package/dist/macro-graph.js +193 -0
- package/dist/macro-repair.js +183 -0
- package/dist/mcp-server.mjs +1202 -483
- package/dist/memory-layer.js +42 -5
- package/dist/multi-repo-precision.js +422 -0
- package/dist/name-free-protocol.js +425 -0
- package/dist/name-free-protocol.test.js +170 -0
- package/dist/name-scrambling.js +138 -0
- package/dist/name-scrambling.test.js +16 -0
- package/dist/p3-observability.test.js +281 -0
- package/dist/p5-orchestrator.test.js +225 -0
- package/dist/pairwise-preference.js +294 -0
- package/dist/pairwise-preference.test.js +140 -0
- package/dist/planner-constraints.js +104 -0
- package/dist/planner-prompts.js +155 -0
- package/dist/planner-telemetry.js +415 -0
- package/dist/planner-trace.js +214 -0
- package/dist/planner.js +162 -167
- package/dist/plsb/artifact.js +116 -0
- package/dist/plsb/cli.js +71 -0
- package/dist/plsb/index.js +19 -0
- package/dist/plsb/leaderboard.js +249 -0
- package/dist/plsb/report-md.js +156 -0
- package/dist/plsb/schema.js +179 -0
- package/dist/plsb-benchmark.js +284 -0
- package/dist/plsb-benchmark.test.js +119 -0
- package/dist/policy/cli.js +134 -0
- package/dist/policy/engine.js +333 -0
- package/dist/policy/index.js +12 -0
- package/dist/policy/types.js +59 -0
- package/dist/policy-miner.js +505 -0
- package/dist/precision-analyze.js +229 -0
- package/dist/precision-benchmark.js +147 -0
- package/dist/precision-label-c.js +134 -0
- package/dist/precision-label.js +193 -0
- package/dist/precision-report-c.js +149 -0
- package/dist/precision-report.js +246 -0
- package/dist/progmune-status.js +108 -0
- package/dist/proof-engine.js +479 -0
- package/dist/proof-provenance.js +315 -0
- package/dist/protocol-coverage.js +294 -0
- package/dist/protocol-detector.js +1189 -0
- package/dist/protocol-embedding-expanded.js +297 -0
- package/dist/protocol-embedding-expanded.test.js +97 -0
- package/dist/protocol-embedding.js +195 -0
- package/dist/protocol-embedding.test.js +82 -0
- package/dist/protocol-extractor-v2.js +354 -0
- package/dist/protocol-extractor-v2.test.js +140 -0
- package/dist/protocol-extractor.js +310 -0
- package/dist/protocol-extractor.test.js +113 -0
- package/dist/protocol-foundation.js +322 -0
- package/dist/protocol-foundation.test.js +163 -0
- package/dist/protocol-frontier.js +243 -0
- package/dist/protocol-frontier.test.js +92 -0
- package/dist/protocol-gap-analyzer.js +228 -0
- package/dist/protocol-gap-analyzer.test.js +49 -0
- package/dist/protocol-invariants.js +276 -0
- package/dist/protocol-invariants.test.js +111 -0
- package/dist/protocol-knowledge.js +464 -0
- package/dist/protocol-miner.js +343 -0
- package/dist/protocol-mining.js +207 -0
- package/dist/protocol-mining.test.js +37 -0
- package/dist/protocol-registry.js +1 -1
- package/dist/protocol-security-benchmark.js +222 -0
- package/dist/protocol-vulnerability.js +257 -0
- package/dist/protocol-vulnerability.test.js +60 -0
- package/dist/python-benchmark.js +120 -0
- package/dist/python-emitter.js +163 -45
- package/dist/python-protocol-extractor.js +187 -0
- package/dist/python-protocol-extractor.test.js +116 -0
- package/dist/realworld-benchmark.js +646 -0
- package/dist/realworld-benchmark.test.js +36 -0
- package/dist/repair-arch.test.js +411 -0
- package/dist/repair-evolution.test.js +454 -0
- package/dist/repair-executor.js +719 -0
- package/dist/repair-proposal.js +4 -4
- package/dist/repair-ranker.js +141 -0
- package/dist/repair-strategies.js +419 -0
- package/dist/repair-taxonomy.js +234 -0
- package/dist/repair-types.js +12 -0
- package/dist/repo-evaluator.js +250 -0
- package/dist/repo-evaluator.test.js +128 -0
- package/dist/resource-abstraction.js +242 -0
- package/dist/resource-detector.js +211 -0
- package/dist/result.test.js +43 -0
- package/dist/reward-system.js +411 -0
- package/dist/reward-system.test.js +175 -0
- package/dist/risk-model.js +215 -0
- package/dist/rule-miner.js +234 -7
- package/dist/rule-specificity.js +254 -0
- package/dist/runtime-types.js +27 -0
- package/dist/scaffold.js +208 -0
- package/dist/scale-collector.test.js +101 -0
- package/dist/scale-trajectory-collector.js +128 -0
- package/dist/sdk.js +250 -0
- package/dist/search-planner.js +4 -41
- package/dist/semantic-snapshot.js +1 -1
- package/dist/semantic-topology.js +121 -0
- package/dist/semantic-trace.js +310 -317
- package/dist/sequence-extractor.js +343 -0
- package/dist/skill-library.js +245 -0
- package/dist/skill-planner.test.js +189 -0
- package/dist/software-physics.js +291 -0
- package/dist/software-physics.test.js +81 -0
- package/dist/ssg-precision.js +478 -0
- package/dist/ssg-validator.js +71 -21
- package/dist/state-inference-doubleblind.test.js +160 -0
- package/dist/state-inference.js +516 -0
- package/dist/state-inference.test.js +115 -0
- package/dist/state-machine-fingerprint.js +345 -0
- package/dist/state-machine-fingerprint.test.js +120 -0
- package/dist/state-miner.js +386 -0
- package/dist/state-name-inference.js +213 -0
- package/dist/state-name-inference.test.js +69 -0
- package/dist/strategy-planner.js +262 -96
- package/dist/strategy-planner.test.js +135 -0
- package/dist/telemetry-analytics.test.js +402 -0
- package/dist/terminal-format.js +68 -0
- package/dist/terminal-format.test.js +83 -0
- package/dist/topology-factory.js +196 -0
- package/dist/topology-representation.js +242 -0
- package/dist/topology-representation.test.js +27 -0
- package/dist/trajectory-augmentation.js +254 -0
- package/dist/trajectory-augmentation.test.js +63 -0
- package/dist/trajectory-corpus.js +440 -0
- package/dist/trajectory-corpus.test.js +32 -0
- package/dist/trajectory-feedback.test.js +116 -0
- package/dist/transition-synthesizer.js +286 -0
- package/dist/transition-synthesizer.test.js +123 -0
- package/dist/trust/api-semantic-mapper.js +809 -0
- package/dist/trust/call-graph-propagator.js +225 -0
- package/dist/trust/cli.js +122 -0
- package/dist/trust/compliance-scorer.js +283 -0
- package/dist/trust/confidence-calculator.js +261 -0
- package/dist/trust/engine.js +1145 -0
- package/dist/trust/explainability.js +85 -0
- package/dist/trust/formatters/ci.js +42 -0
- package/dist/trust/formatters/json.js +11 -0
- package/dist/trust/formatters/terminal.js +152 -0
- package/dist/trust/index.js +39 -0
- package/dist/trust/phase1-verify.js +171 -0
- package/dist/trust/protocol-domain-validator.js +697 -0
- package/dist/trust/score-calculator.js +282 -0
- package/dist/trust/ssg-bridge.js +641 -0
- package/dist/trust/ssg-bridge.test.js +269 -0
- package/dist/trust/types.js +67 -0
- package/dist/trust/violation-trace.js +335 -0
- package/dist/trust-api.js +179 -0
- package/dist/trust-calibration.js +279 -0
- package/dist/unknown-protocol-discovery.js +339 -0
- package/dist/unknown-protocol-discovery.test.js +102 -0
- package/dist/unsupervised-physics.js +230 -0
- package/dist/unsupervised-physics.test.js +95 -0
- package/dist/utils.test.js +37 -0
- package/dist/validator.js +187 -10
- package/dist/verification-intelligence.js +475 -0
- package/dist/verify-api.js +432 -0
- package/dist/vi-impact-report.js +293 -0
- package/dist/wl-fingerprint.js +162 -0
- package/dist/wl-fingerprint.test.js +130 -0
- package/dist/zeroshot-strategy.js +139 -0
- package/docs/Progmune_/346/212/225/350/265/204/344/272/272/347/231/275/347/232/256/344/271/246_v2.0.html +576 -0
- package/docs/Progmune_/351/241/271/347/233/256/345/205/250/350/247/243.html +710 -0
- package/package.json +74 -7
- package/protocols.json +1956 -50
- package/.dockerignore +0 -14
- package/.mcp.json +0 -11
- package/.progmune_allowlist +0 -50
- package/.test_report/test_report.md +0 -87
- package/Dockerfile +0 -9
- package/FAQ.md +0 -167
- package/WHITEPAPER.md +0 -540
- package/demo-project/auth.ts +0 -55
- package/demo-project/tsconfig.json +0 -8
- package/dist/acl-breakdown.js +0 -13
- package/dist/all-sessions.js +0 -11
- package/dist/antibody-stats.js +0 -11
- package/dist/branch-tree-count.js +0 -14
- package/dist/common-fixpath.js +0 -12
- package/dist/constraint-types.js +0 -12
- package/dist/exec-metrics.js +0 -11
- package/dist/failure-report.js +0 -11
- package/dist/fast-path-hits.js +0 -13
- package/dist/fingerprint-list.js +0 -15
- package/dist/gen-history-log.js +0 -13
- package/dist/heatmap-data.js +0 -11
- package/dist/recent-session.js +0 -12
- package/dist/svl-distribution.js +0 -11
- package/dist/terminal-status.js +0 -11
- package/dist/token-savings.js +0 -11
- package/dist/total-repairs.js +0 -12
- package/dist/unresolved-count.js +0 -12
- package/dist/valid-fingerprints.js +0 -13
- package/dist/verify-ledgers.js +0 -11
- package/docs/whitepaper-style.css +0 -77
- package/docs/whitepaper-v2.1.md +0 -609
- package/docs/whitepaper-v2.2.md +0 -1064
- package/docs/whitepaper-v2.2.pdf +0 -0
- package/fly.toml +0 -31
- package/public/dashboard.html +0 -119
- package/server/hub.js +0 -116
- package/test/replay-golden/sess_1780063202050_mgeld.json +0 -9
- package/test/replay-golden/sess_1780064032560_gocld.json +0 -354
- package/test/replay-golden/sess_1780064413331_s2709.json +0 -606
- package/test/replay-golden/sess_1780064792710_y3avo.json +0 -614
- package/test/replay-golden.ts +0 -84
- package/test_benchmark.js +0 -165
- package/test_comprehensive.mjs +0 -638
- package/test_concurrency.js +0 -129
- package/test_ir_robustness.js +0 -85
- package/test_semantic_contracts.js +0 -269
- package/test_ssg_stress.js +0 -156
- package/test_svl3.js +0 -58
- package/tsconfig.json +0 -17
|
@@ -0,0 +1,109 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
/**
|
|
3
|
+
* P3.6: Coverage Dashboard
|
|
4
|
+
*
|
|
5
|
+
* Visualizes protocol coverage gaps, ranks protocols by risk,
|
|
6
|
+
* and generates acquisition priorities.
|
|
7
|
+
*
|
|
8
|
+
* The dashboard answers:
|
|
9
|
+
* - Which protocols are fully observed? Which are data-poor?
|
|
10
|
+
* - Where should we collect more data next?
|
|
11
|
+
* - What transitions should we benchmark first?
|
|
12
|
+
*/
|
|
13
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
14
|
+
exports.assessRisk = assessRisk;
|
|
15
|
+
exports.generateCoverageDashboard = generateCoverageDashboard;
|
|
16
|
+
exports.printCoverageDashboard = printCoverageDashboard;
|
|
17
|
+
const protocol_coverage_1 = require("./protocol-coverage");
|
|
18
|
+
const failure_corpus_1 = require("./failure-corpus");
|
|
19
|
+
function assessRisk(report) {
|
|
20
|
+
const sc = report.stateCoverage.stateCoverage;
|
|
21
|
+
const tc = report.transitionCoverage.transitionCoverage;
|
|
22
|
+
const avg = (sc + tc) / 2;
|
|
23
|
+
let risk;
|
|
24
|
+
let recommendation;
|
|
25
|
+
if (avg < 0.25) {
|
|
26
|
+
risk = "critical";
|
|
27
|
+
recommendation = `Immediate: add ${report.transitionCoverage.missingTransitions.length} benchmark cases for uncovered transitions`;
|
|
28
|
+
}
|
|
29
|
+
else if (avg < 0.50) {
|
|
30
|
+
risk = "high";
|
|
31
|
+
recommendation = `Priority: focus on missing ${report.stateCoverage.missingStates.length} states and ${report.transitionCoverage.missingTransitions.length} transitions`;
|
|
32
|
+
}
|
|
33
|
+
else if (avg < 0.75) {
|
|
34
|
+
risk = "medium";
|
|
35
|
+
recommendation = `Fill remaining gaps: ${report.transitionCoverage.missingTransitions.length} transitions uncovered`;
|
|
36
|
+
}
|
|
37
|
+
else {
|
|
38
|
+
risk = "low";
|
|
39
|
+
recommendation = "Well-covered. Monitor for regressions.";
|
|
40
|
+
}
|
|
41
|
+
return {
|
|
42
|
+
protocol: report.protocol,
|
|
43
|
+
stateCoverage: sc,
|
|
44
|
+
transitionCoverage: tc,
|
|
45
|
+
trajectoryCount: report.trajectoryCount,
|
|
46
|
+
risk,
|
|
47
|
+
missingTransitionCount: report.transitionCoverage.missingTransitions.length,
|
|
48
|
+
recommendation,
|
|
49
|
+
};
|
|
50
|
+
}
|
|
51
|
+
function generateCoverageDashboard(trajectories) {
|
|
52
|
+
const trajs = trajectories || (0, failure_corpus_1.loadTrajectories)();
|
|
53
|
+
const protocols = (0, protocol_coverage_1.loadDefaultProtocolDefinitions)();
|
|
54
|
+
const reports = (0, protocol_coverage_1.analyzeAllCoverage)(protocols, trajs);
|
|
55
|
+
const risks = reports.map(assessRisk).sort((a, b) => a.transitionCoverage - b.transitionCoverage);
|
|
56
|
+
const totalStates = reports.reduce((s, r) => s + r.stateCoverage.totalStates, 0);
|
|
57
|
+
const visitedStates = reports.reduce((s, r) => s + r.stateCoverage.visitedStates, 0);
|
|
58
|
+
const totalTrans = reports.reduce((s, r) => s + r.transitionCoverage.totalTransitions, 0);
|
|
59
|
+
const visitedTrans = reports.reduce((s, r) => s + r.transitionCoverage.visitedTransitions, 0);
|
|
60
|
+
return {
|
|
61
|
+
reports,
|
|
62
|
+
riskRanking: risks,
|
|
63
|
+
overallStateCoverage: totalStates > 0 ? visitedStates / totalStates : 0,
|
|
64
|
+
overallTransitionCoverage: totalTrans > 0 ? visitedTrans / totalTrans : 0,
|
|
65
|
+
totalTrajectories: trajs.length,
|
|
66
|
+
criticalProtocols: risks.filter(r => r.risk === "critical" || r.risk === "high").length,
|
|
67
|
+
};
|
|
68
|
+
}
|
|
69
|
+
function printCoverageDashboard(dashboard) {
|
|
70
|
+
console.log("\n╔════════════════════════════════════════════════════╗");
|
|
71
|
+
console.log("║ Protocol Coverage Dashboard ║");
|
|
72
|
+
console.log("╚════════════════════════════════════════════════════╝\n");
|
|
73
|
+
console.log(`Trajectories Analyzed: ${dashboard.totalTrajectories}`);
|
|
74
|
+
console.log(`Overall State Coverage: ${(dashboard.overallStateCoverage * 100).toFixed(0)}%`);
|
|
75
|
+
console.log(`Overall Transition Coverage: ${(dashboard.overallTransitionCoverage * 100).toFixed(0)}%`);
|
|
76
|
+
console.log(`Critical/High Risk Protocols: ${dashboard.criticalProtocols}\n`);
|
|
77
|
+
console.log("─── By Protocol ───");
|
|
78
|
+
console.log("Protocol State Trans Trajs Risk");
|
|
79
|
+
console.log("────────────────────────────────────────────────");
|
|
80
|
+
for (const r of dashboard.riskRanking) {
|
|
81
|
+
const riskIcon = r.risk === "critical" ? "🔴" : r.risk === "high" ? "🟠" : r.risk === "medium" ? "🟡" : "🟢";
|
|
82
|
+
const sc = (r.stateCoverage * 100).toFixed(0).padStart(3);
|
|
83
|
+
const tc = (r.transitionCoverage * 100).toFixed(0).padStart(3);
|
|
84
|
+
console.log(` ${r.protocol.padEnd(16)} ${sc}% ${tc}% ${String(r.trajectoryCount).padStart(4)} ${riskIcon} ${r.risk}`);
|
|
85
|
+
}
|
|
86
|
+
console.log();
|
|
87
|
+
// Missing transitions detail for high-risk protocols
|
|
88
|
+
const criticalReports = dashboard.reports.filter(r => assessRisk(r).risk === "critical" || assessRisk(r).risk === "high");
|
|
89
|
+
if (criticalReports.length > 0) {
|
|
90
|
+
console.log("─── Highest Risk: Missing Transitions ───");
|
|
91
|
+
for (const r of criticalReports) {
|
|
92
|
+
const missing = r.transitionCoverage.missingTransitions.slice(0, 5);
|
|
93
|
+
if (missing.length === 0)
|
|
94
|
+
continue;
|
|
95
|
+
console.log(`\n ${r.protocol} (${r.transitionCoverage.missingTransitions.length} missing):`);
|
|
96
|
+
for (const m of missing) {
|
|
97
|
+
console.log(` ${m.from} → ${m.to} (via ${m.rule})`);
|
|
98
|
+
}
|
|
99
|
+
}
|
|
100
|
+
console.log();
|
|
101
|
+
}
|
|
102
|
+
console.log("─── Recommendations ───");
|
|
103
|
+
for (const r of dashboard.riskRanking) {
|
|
104
|
+
if (r.risk === "critical" || r.risk === "high") {
|
|
105
|
+
console.log(` ${r.protocol}: ${r.recommendation}`);
|
|
106
|
+
}
|
|
107
|
+
}
|
|
108
|
+
console.log();
|
|
109
|
+
}
|
|
@@ -0,0 +1,205 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
/**
|
|
3
|
+
* P3.6: Coverage System Integration Tests
|
|
4
|
+
*
|
|
5
|
+
* Verifying:
|
|
6
|
+
* 1. Coverage engine correctly computes state/transition coverage
|
|
7
|
+
* 2. Dashboard visualizes gaps and risk ranking
|
|
8
|
+
* 3. Benchmark generator produces cases for uncovered transitions
|
|
9
|
+
* 4. End-to-end: analyze → generate → new cases
|
|
10
|
+
*/
|
|
11
|
+
var __createBinding = (this && this.__createBinding) || (Object.create ? (function(o, m, k, k2) {
|
|
12
|
+
if (k2 === undefined) k2 = k;
|
|
13
|
+
var desc = Object.getOwnPropertyDescriptor(m, k);
|
|
14
|
+
if (!desc || ("get" in desc ? !m.__esModule : desc.writable || desc.configurable)) {
|
|
15
|
+
desc = { enumerable: true, get: function() { return m[k]; } };
|
|
16
|
+
}
|
|
17
|
+
Object.defineProperty(o, k2, desc);
|
|
18
|
+
}) : (function(o, m, k, k2) {
|
|
19
|
+
if (k2 === undefined) k2 = k;
|
|
20
|
+
o[k2] = m[k];
|
|
21
|
+
}));
|
|
22
|
+
var __setModuleDefault = (this && this.__setModuleDefault) || (Object.create ? (function(o, v) {
|
|
23
|
+
Object.defineProperty(o, "default", { enumerable: true, value: v });
|
|
24
|
+
}) : function(o, v) {
|
|
25
|
+
o["default"] = v;
|
|
26
|
+
});
|
|
27
|
+
var __importStar = (this && this.__importStar) || (function () {
|
|
28
|
+
var ownKeys = function(o) {
|
|
29
|
+
ownKeys = Object.getOwnPropertyNames || function (o) {
|
|
30
|
+
var ar = [];
|
|
31
|
+
for (var k in o) if (Object.prototype.hasOwnProperty.call(o, k)) ar[ar.length] = k;
|
|
32
|
+
return ar;
|
|
33
|
+
};
|
|
34
|
+
return ownKeys(o);
|
|
35
|
+
};
|
|
36
|
+
return function (mod) {
|
|
37
|
+
if (mod && mod.__esModule) return mod;
|
|
38
|
+
var result = {};
|
|
39
|
+
if (mod != null) for (var k = ownKeys(mod), i = 0; i < k.length; i++) if (k[i] !== "default") __createBinding(result, mod, k[i]);
|
|
40
|
+
__setModuleDefault(result, mod);
|
|
41
|
+
return result;
|
|
42
|
+
};
|
|
43
|
+
})();
|
|
44
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
45
|
+
const vitest_1 = require("vitest");
|
|
46
|
+
const fs = __importStar(require("fs"));
|
|
47
|
+
const path = __importStar(require("path"));
|
|
48
|
+
const protocol_coverage_1 = require("./protocol-coverage");
|
|
49
|
+
const coverage_dashboard_1 = require("./coverage-dashboard");
|
|
50
|
+
const benchmark_generator_1 = require("./benchmark-generator");
|
|
51
|
+
// ═══════════════════════════════════════════════════════════════
|
|
52
|
+
// Coverage Engine
|
|
53
|
+
// ═══════════════════════════════════════════════════════════════
|
|
54
|
+
(0, vitest_1.describe)("Coverage Engine", () => {
|
|
55
|
+
function makeFileRules() {
|
|
56
|
+
return new Map([
|
|
57
|
+
["open_file", { pre_states: [], post_states: ["FILE_OPEN"] }],
|
|
58
|
+
["write_file", { pre_states: ["FILE_OPEN"], post_states: ["FILE_DIRTY"] }],
|
|
59
|
+
["close_file", { pre_states: ["FILE_OPEN", "FILE_DIRTY"], post_states: [], invalidate: ["FILE_OPEN", "FILE_DIRTY"] }],
|
|
60
|
+
]);
|
|
61
|
+
}
|
|
62
|
+
const fileProto = (0, protocol_coverage_1.parseProtocolDefinition)("FileProtocol", makeFileRules(), "INIT");
|
|
63
|
+
(0, vitest_1.it)("computes full coverage when all transitions visited", () => {
|
|
64
|
+
const trajectories = [{
|
|
65
|
+
id: "t1", timestamp: new Date().toISOString(),
|
|
66
|
+
protocol: "FileProtocol", initialState: ["INIT"], finalState: [],
|
|
67
|
+
trajectory: ["open_file", "write_file", "close_file"],
|
|
68
|
+
result: "success", context: { nestingDepth: 0, exceptionHandled: false, insideLoop: false, branchCount: 0, asyncContext: false },
|
|
69
|
+
successRate: 1.0, metadata: { source: "human" },
|
|
70
|
+
}];
|
|
71
|
+
const report = (0, protocol_coverage_1.analyzeCoverage)(fileProto, trajectories);
|
|
72
|
+
(0, vitest_1.expect)(report.transitionCoverage.transitionCoverage).toBeGreaterThan(0.5);
|
|
73
|
+
(0, vitest_1.expect)(report.stateCoverage.stateCoverage).toBeGreaterThan(0.5);
|
|
74
|
+
});
|
|
75
|
+
(0, vitest_1.it)("detects uncovered transitions", () => {
|
|
76
|
+
const trajectories = [{
|
|
77
|
+
id: "t2", timestamp: new Date().toISOString(),
|
|
78
|
+
protocol: "FileProtocol", initialState: ["INIT"], finalState: [],
|
|
79
|
+
trajectory: ["open_file", "close_file"], // missing write_file
|
|
80
|
+
result: "success", context: { nestingDepth: 0, exceptionHandled: false, insideLoop: false, branchCount: 0, asyncContext: false },
|
|
81
|
+
successRate: 1.0, metadata: { source: "human" },
|
|
82
|
+
}];
|
|
83
|
+
const report = (0, protocol_coverage_1.analyzeCoverage)(fileProto, trajectories);
|
|
84
|
+
(0, vitest_1.expect)(report.transitionCoverage.missingTransitions.length).toBeGreaterThan(0);
|
|
85
|
+
});
|
|
86
|
+
(0, vitest_1.it)("empty trajectories = zero coverage", () => {
|
|
87
|
+
const report = (0, protocol_coverage_1.analyzeCoverage)(fileProto, []);
|
|
88
|
+
(0, vitest_1.expect)(report.transitionCoverage.transitionCoverage).toBe(0);
|
|
89
|
+
(0, vitest_1.expect)(report.stateCoverage.stateCoverage).toBe(0);
|
|
90
|
+
(0, vitest_1.expect)(report.trajectoryCount).toBe(0);
|
|
91
|
+
});
|
|
92
|
+
});
|
|
93
|
+
// ═══════════════════════════════════════════════════════════════
|
|
94
|
+
// Default Protocol Definitions
|
|
95
|
+
// ═══════════════════════════════════════════════════════════════
|
|
96
|
+
(0, vitest_1.describe)("Default Protocol Definitions", () => {
|
|
97
|
+
(0, vitest_1.it)("loads all 9 protocol groups", () => {
|
|
98
|
+
const protocols = (0, protocol_coverage_1.loadDefaultProtocolDefinitions)();
|
|
99
|
+
(0, vitest_1.expect)(protocols.length).toBeGreaterThanOrEqual(9);
|
|
100
|
+
const names = protocols.map(p => p.name);
|
|
101
|
+
(0, vitest_1.expect)(names).toContain("FileProtocol");
|
|
102
|
+
(0, vitest_1.expect)(names).toContain("AuthProtocol");
|
|
103
|
+
(0, vitest_1.expect)(names).toContain("DBProtocol");
|
|
104
|
+
(0, vitest_1.expect)(names).toContain("IRProtocol");
|
|
105
|
+
(0, vitest_1.expect)(names).toContain("StatelessProtocol");
|
|
106
|
+
(0, vitest_1.expect)(names).toContain("TransactionProtocol");
|
|
107
|
+
(0, vitest_1.expect)(names).toContain("ConditionalProtocol");
|
|
108
|
+
(0, vitest_1.expect)(names).toContain("LoopProtocol");
|
|
109
|
+
(0, vitest_1.expect)(names).toContain("CrossProtocol");
|
|
110
|
+
});
|
|
111
|
+
(0, vitest_1.it)("each protocol (except stateless) has states and transitions", () => {
|
|
112
|
+
for (const p of (0, protocol_coverage_1.loadDefaultProtocolDefinitions)()) {
|
|
113
|
+
(0, vitest_1.expect)(p.states.length).toBeGreaterThan(0);
|
|
114
|
+
// StatelessProtocol has empty pre/post states → 0 acquire transitions
|
|
115
|
+
if (p.name === "StatelessProtocol") {
|
|
116
|
+
(0, vitest_1.expect)(p.transitions.length).toBeGreaterThanOrEqual(0);
|
|
117
|
+
}
|
|
118
|
+
else {
|
|
119
|
+
(0, vitest_1.expect)(p.transitions.length).toBeGreaterThan(0);
|
|
120
|
+
}
|
|
121
|
+
}
|
|
122
|
+
});
|
|
123
|
+
(0, vitest_1.it)("FileProtocol has open/write/close transitions", () => {
|
|
124
|
+
const file = (0, protocol_coverage_1.loadDefaultProtocolDefinitions)().find(p => p.name === "FileProtocol");
|
|
125
|
+
const tKeys = file.transitions.map(t => `${t.from}→${t.to}`);
|
|
126
|
+
(0, vitest_1.expect)(tKeys).toContain("INIT→FILE_OPEN"); // open_file
|
|
127
|
+
(0, vitest_1.expect)(tKeys).toContain("FILE_OPEN→∅"); // close_file invalidates FILE_OPEN
|
|
128
|
+
});
|
|
129
|
+
(0, vitest_1.it)("AuthProtocol has auth lifecycle transitions", () => {
|
|
130
|
+
const auth = (0, protocol_coverage_1.loadDefaultProtocolDefinitions)().find(p => p.name === "AuthProtocol");
|
|
131
|
+
const tKeys = auth.transitions.map(t => `${t.from}→${t.to}`);
|
|
132
|
+
(0, vitest_1.expect)(tKeys).toContain("UNAUTHENTICATED→PASSWORD_VERIFIED"); // verify_password
|
|
133
|
+
(0, vitest_1.expect)(tKeys).toContain("PASSWORD_VERIFIED→TOKEN_ISSUED"); // generate_jwt
|
|
134
|
+
(0, vitest_1.expect)(tKeys).toContain("TOKEN_ISSUED→SESSION_ACTIVE"); // create_session
|
|
135
|
+
(0, vitest_1.expect)(tKeys).toContain("SESSION_ACTIVE→UNAUTHENTICATED"); // logout
|
|
136
|
+
});
|
|
137
|
+
});
|
|
138
|
+
// ═══════════════════════════════════════════════════════════════
|
|
139
|
+
// Coverage Dashboard
|
|
140
|
+
// ═══════════════════════════════════════════════════════════════
|
|
141
|
+
const GEN_DIR = path.resolve(__dirname, "..", "test-coverage-gen");
|
|
142
|
+
process.env.PROGMUNE_PROJECT_DIR = GEN_DIR;
|
|
143
|
+
fs.mkdirSync(GEN_DIR, { recursive: true });
|
|
144
|
+
fs.mkdirSync(path.join(GEN_DIR, ".progmune_corpus", "trajectories"), { recursive: true });
|
|
145
|
+
(0, vitest_1.describe)("Coverage Dashboard", () => {
|
|
146
|
+
(0, vitest_1.it)("generates dashboard from current trajectories", () => {
|
|
147
|
+
const dashboard = (0, coverage_dashboard_1.generateCoverageDashboard)([]);
|
|
148
|
+
(0, vitest_1.expect)(dashboard.reports.length).toBeGreaterThanOrEqual(9);
|
|
149
|
+
(0, vitest_1.expect)(dashboard.riskRanking.length).toBeGreaterThanOrEqual(9);
|
|
150
|
+
(0, vitest_1.expect)(dashboard.overallTransitionCoverage).toBeGreaterThanOrEqual(0);
|
|
151
|
+
(0, vitest_1.expect)(dashboard.overallTransitionCoverage).toBeLessThanOrEqual(1);
|
|
152
|
+
(0, vitest_1.expect)(dashboard.criticalProtocols).toBeGreaterThanOrEqual(0);
|
|
153
|
+
(0, coverage_dashboard_1.printCoverageDashboard)(dashboard);
|
|
154
|
+
});
|
|
155
|
+
(0, vitest_1.it)("correctly ranks empty protocols as critical", () => {
|
|
156
|
+
const dashboard = (0, coverage_dashboard_1.generateCoverageDashboard)([]);
|
|
157
|
+
// With zero trajectories, all protocols should be critical or high risk
|
|
158
|
+
const emptyProtocols = dashboard.riskRanking.filter(r => r.trajectoryCount === 0);
|
|
159
|
+
for (const r of emptyProtocols) {
|
|
160
|
+
(0, vitest_1.expect)(r.stateCoverage).toBe(0);
|
|
161
|
+
(0, vitest_1.expect)(r.transitionCoverage).toBe(0);
|
|
162
|
+
(0, vitest_1.expect)(r.risk).toBe("critical");
|
|
163
|
+
}
|
|
164
|
+
});
|
|
165
|
+
});
|
|
166
|
+
// ═══════════════════════════════════════════════════════════════
|
|
167
|
+
// Benchmark Generator
|
|
168
|
+
// ═══════════════════════════════════════════════════════════════
|
|
169
|
+
(0, vitest_1.describe)("Benchmark Generator", () => {
|
|
170
|
+
(0, vitest_1.it)("generates cases for uncovered transitions", () => {
|
|
171
|
+
const generated = (0, benchmark_generator_1.generateMissingBenchmarks)([]);
|
|
172
|
+
// With zero trajectories, all protocols have uncovered transitions
|
|
173
|
+
(0, vitest_1.expect)(Object.keys(generated).length).toBeGreaterThanOrEqual(3);
|
|
174
|
+
// Each protocol should have generated cases
|
|
175
|
+
for (const [protocol, cases] of Object.entries(generated)) {
|
|
176
|
+
(0, vitest_1.expect)(cases.length).toBeGreaterThan(0);
|
|
177
|
+
for (const c of cases) {
|
|
178
|
+
(0, vitest_1.expect)(c.broken.length).toBeGreaterThan(0);
|
|
179
|
+
(0, vitest_1.expect)(c.expected.length).toBeGreaterThan(0);
|
|
180
|
+
(0, vitest_1.expect)(c.expected.length).toBeGreaterThan(c.broken.length);
|
|
181
|
+
(0, vitest_1.expect)(["resource_leak", "missing_prerequisite"]).toContain(c.violationType);
|
|
182
|
+
}
|
|
183
|
+
}
|
|
184
|
+
});
|
|
185
|
+
(0, vitest_1.it)("writes generated benchmarks to disk", () => {
|
|
186
|
+
const generated = (0, benchmark_generator_1.generateMissingBenchmarks)([]);
|
|
187
|
+
const outDir = path.resolve(GEN_DIR, "generated-benchmarks");
|
|
188
|
+
const written = (0, benchmark_generator_1.writeGeneratedBenchmarks)(generated, outDir);
|
|
189
|
+
(0, vitest_1.expect)(written.length).toBeGreaterThanOrEqual(3);
|
|
190
|
+
// Verify files exist and are valid JSON
|
|
191
|
+
for (const filepath of written) {
|
|
192
|
+
(0, vitest_1.expect)(fs.existsSync(filepath)).toBe(true);
|
|
193
|
+
const content = JSON.parse(fs.readFileSync(filepath, "utf-8"));
|
|
194
|
+
(0, vitest_1.expect)(content.cases.length).toBeGreaterThan(0);
|
|
195
|
+
(0, vitest_1.expect)(content.source).toBe("coverage-gap");
|
|
196
|
+
}
|
|
197
|
+
});
|
|
198
|
+
(0, vitest_1.it)("runs the full coverage→generation pipeline", () => {
|
|
199
|
+
const result = (0, benchmark_generator_1.runCoverageDrivenGeneration)();
|
|
200
|
+
(0, vitest_1.expect)(result.existingCases).toBeGreaterThanOrEqual(1); // from previous test writes
|
|
201
|
+
(0, vitest_1.expect)(result.generatedCases).toBeGreaterThanOrEqual(10);
|
|
202
|
+
(0, vitest_1.expect)(result.writtenFiles.length).toBeGreaterThanOrEqual(3);
|
|
203
|
+
console.log(`\nCoverage-Driven Generation: ${result.summary}`);
|
|
204
|
+
});
|
|
205
|
+
});
|
|
@@ -0,0 +1,352 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
/**
|
|
3
|
+
* P1: Cross-Repository Precision Runner
|
|
4
|
+
*
|
|
5
|
+
* Runs the full SSG precision pipeline (discover → validate → measure)
|
|
6
|
+
* on every benchmark repo that has labeled sequences.
|
|
7
|
+
*
|
|
8
|
+
* Produces:
|
|
9
|
+
* benchmarks/reports/cross-repo-precision-<date>.json
|
|
10
|
+
* benchmarks/reports/cross-repo-precision-latest.json
|
|
11
|
+
*
|
|
12
|
+
* Usage:
|
|
13
|
+
* npx ts-node --transpile-only src/cross-repo-precision.ts
|
|
14
|
+
* npx ts-node --transpile-only src/cross-repo-precision.ts --repos curl,libssh
|
|
15
|
+
*/
|
|
16
|
+
var __createBinding = (this && this.__createBinding) || (Object.create ? (function(o, m, k, k2) {
|
|
17
|
+
if (k2 === undefined) k2 = k;
|
|
18
|
+
var desc = Object.getOwnPropertyDescriptor(m, k);
|
|
19
|
+
if (!desc || ("get" in desc ? !m.__esModule : desc.writable || desc.configurable)) {
|
|
20
|
+
desc = { enumerable: true, get: function() { return m[k]; } };
|
|
21
|
+
}
|
|
22
|
+
Object.defineProperty(o, k2, desc);
|
|
23
|
+
}) : (function(o, m, k, k2) {
|
|
24
|
+
if (k2 === undefined) k2 = k;
|
|
25
|
+
o[k2] = m[k];
|
|
26
|
+
}));
|
|
27
|
+
var __setModuleDefault = (this && this.__setModuleDefault) || (Object.create ? (function(o, v) {
|
|
28
|
+
Object.defineProperty(o, "default", { enumerable: true, value: v });
|
|
29
|
+
}) : function(o, v) {
|
|
30
|
+
o["default"] = v;
|
|
31
|
+
});
|
|
32
|
+
var __importStar = (this && this.__importStar) || (function () {
|
|
33
|
+
var ownKeys = function(o) {
|
|
34
|
+
ownKeys = Object.getOwnPropertyNames || function (o) {
|
|
35
|
+
var ar = [];
|
|
36
|
+
for (var k in o) if (Object.prototype.hasOwnProperty.call(o, k)) ar[ar.length] = k;
|
|
37
|
+
return ar;
|
|
38
|
+
};
|
|
39
|
+
return ownKeys(o);
|
|
40
|
+
};
|
|
41
|
+
return function (mod) {
|
|
42
|
+
if (mod && mod.__esModule) return mod;
|
|
43
|
+
var result = {};
|
|
44
|
+
if (mod != null) for (var k = ownKeys(mod), i = 0; i < k.length; i++) if (k[i] !== "default") __createBinding(result, mod, k[i]);
|
|
45
|
+
__setModuleDefault(result, mod);
|
|
46
|
+
return result;
|
|
47
|
+
};
|
|
48
|
+
})();
|
|
49
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
50
|
+
const fs = __importStar(require("fs"));
|
|
51
|
+
const path = __importStar(require("path"));
|
|
52
|
+
// ═══════════════════════════════════════════════════════════════
|
|
53
|
+
// Per-Repo Precision Runner
|
|
54
|
+
// ═══════════════════════════════════════════════════════════════
|
|
55
|
+
function runPrecisionForRepo(repoName) {
|
|
56
|
+
const benchmarksDir = path.resolve(process.cwd(), "benchmarks");
|
|
57
|
+
const labelFile = path.join(benchmarksDir, `${repoName}-labels.json`);
|
|
58
|
+
const empty = {
|
|
59
|
+
repo: repoName,
|
|
60
|
+
status: "no_labels",
|
|
61
|
+
total: 0,
|
|
62
|
+
cleanLabels: 0,
|
|
63
|
+
violationLabels: 0,
|
|
64
|
+
tp: 0, fp: 0, tn: 0, fn: 0,
|
|
65
|
+
precision: 0, recall: 0, f1: 0,
|
|
66
|
+
falsePositiveRate: 0, falseNegativeRate: 0,
|
|
67
|
+
rulesDiscovered: 0,
|
|
68
|
+
mismatches: [],
|
|
69
|
+
};
|
|
70
|
+
if (!fs.existsSync(labelFile)) {
|
|
71
|
+
empty.error = `No labels file: ${labelFile}`;
|
|
72
|
+
return empty;
|
|
73
|
+
}
|
|
74
|
+
let data;
|
|
75
|
+
try {
|
|
76
|
+
data = JSON.parse(fs.readFileSync(labelFile, "utf-8"));
|
|
77
|
+
}
|
|
78
|
+
catch (e) {
|
|
79
|
+
empty.status = "error";
|
|
80
|
+
empty.error = `Parse error: ${e}`;
|
|
81
|
+
return empty;
|
|
82
|
+
}
|
|
83
|
+
const labels = data.labels || {};
|
|
84
|
+
const sequences = data.sequences || {};
|
|
85
|
+
const labeledIndices = Object.keys(labels).map(Number);
|
|
86
|
+
if (labeledIndices.length === 0) {
|
|
87
|
+
empty.error = "No labeled sequences";
|
|
88
|
+
return empty;
|
|
89
|
+
}
|
|
90
|
+
// Count label distribution
|
|
91
|
+
let cleanLabels = 0;
|
|
92
|
+
let violationLabels = 0;
|
|
93
|
+
const cleanSeqs = [];
|
|
94
|
+
for (const idx of labeledIndices) {
|
|
95
|
+
if (labels[idx] === "clean") {
|
|
96
|
+
cleanLabels++;
|
|
97
|
+
if (sequences[idx])
|
|
98
|
+
cleanSeqs.push(sequences[idx]);
|
|
99
|
+
}
|
|
100
|
+
else if (labels[idx] === "violation") {
|
|
101
|
+
violationLabels++;
|
|
102
|
+
}
|
|
103
|
+
}
|
|
104
|
+
// Discover SSG rules from clean sequences
|
|
105
|
+
let rules;
|
|
106
|
+
let nsInit;
|
|
107
|
+
try {
|
|
108
|
+
const { discoverRulesFromSequences, validateSequenceWithSSG } = require("./ssg-precision");
|
|
109
|
+
const result = discoverRulesFromSequences(cleanSeqs);
|
|
110
|
+
rules = result.rules;
|
|
111
|
+
nsInit = result.nsInit;
|
|
112
|
+
}
|
|
113
|
+
catch (e) {
|
|
114
|
+
empty.status = "error";
|
|
115
|
+
empty.error = `Rule discovery failed: ${e}`;
|
|
116
|
+
return empty;
|
|
117
|
+
}
|
|
118
|
+
// Validate all labeled sequences
|
|
119
|
+
let tp = 0, fp = 0, tn = 0, fn = 0;
|
|
120
|
+
const mismatches = [];
|
|
121
|
+
for (const idx of labeledIndices) {
|
|
122
|
+
const expected = labels[idx];
|
|
123
|
+
const calls = sequences[idx] || [];
|
|
124
|
+
let detected;
|
|
125
|
+
try {
|
|
126
|
+
const { validateSequenceWithSSG } = require("./ssg-precision");
|
|
127
|
+
const result = validateSequenceWithSSG(calls, rules, nsInit);
|
|
128
|
+
detected = result.valid ? "clean" : "violation";
|
|
129
|
+
}
|
|
130
|
+
catch {
|
|
131
|
+
detected = "clean"; // Can't validate → assume clean (conservative)
|
|
132
|
+
}
|
|
133
|
+
if (expected === "violation" && detected === "violation")
|
|
134
|
+
tp++;
|
|
135
|
+
else if (expected === "clean" && detected === "violation")
|
|
136
|
+
fp++;
|
|
137
|
+
else if (expected === "clean" && detected === "clean")
|
|
138
|
+
tn++;
|
|
139
|
+
else if (expected === "violation" && detected === "clean")
|
|
140
|
+
fn++;
|
|
141
|
+
if (expected !== detected) {
|
|
142
|
+
mismatches.push({ index: idx, expected, got: detected, calls });
|
|
143
|
+
}
|
|
144
|
+
}
|
|
145
|
+
const total = labeledIndices.length;
|
|
146
|
+
const precision = tp + fp > 0 ? tp / (tp + fp) : 0;
|
|
147
|
+
const recall = tp + fn > 0 ? tp / (tp + fn) : 0;
|
|
148
|
+
const f1 = precision + recall > 0
|
|
149
|
+
? 2 * precision * recall / (precision + recall)
|
|
150
|
+
: 0;
|
|
151
|
+
const fpr = fp + tn > 0 ? fp / (fp + tn) : 0;
|
|
152
|
+
const fnr = fn + tp > 0 ? fn / (fn + tp) : 0;
|
|
153
|
+
return {
|
|
154
|
+
repo: repoName,
|
|
155
|
+
status: "measured",
|
|
156
|
+
total,
|
|
157
|
+
cleanLabels,
|
|
158
|
+
violationLabels,
|
|
159
|
+
tp, fp, tn, fn,
|
|
160
|
+
precision, recall, f1,
|
|
161
|
+
falsePositiveRate: fpr,
|
|
162
|
+
falseNegativeRate: fnr,
|
|
163
|
+
rulesDiscovered: rules.size,
|
|
164
|
+
mismatches,
|
|
165
|
+
};
|
|
166
|
+
}
|
|
167
|
+
// ═══════════════════════════════════════════════════════════════
|
|
168
|
+
// Report Generator
|
|
169
|
+
// ═══════════════════════════════════════════════════════════════
|
|
170
|
+
function generateReport(repoNames) {
|
|
171
|
+
const repos = repoNames.map(runPrecisionForRepo);
|
|
172
|
+
const measured = repos.filter(r => r.status === "measured");
|
|
173
|
+
let totalTP = 0, totalFP = 0, totalFN = 0, totalSamples = 0;
|
|
174
|
+
for (const r of measured) {
|
|
175
|
+
totalTP += r.tp;
|
|
176
|
+
totalFP += r.fp;
|
|
177
|
+
totalFN += r.fn;
|
|
178
|
+
totalSamples += r.total;
|
|
179
|
+
}
|
|
180
|
+
const macroF1 = measured.length > 0
|
|
181
|
+
? measured.reduce((s, r) => s + r.f1, 0) / measured.length
|
|
182
|
+
: 0;
|
|
183
|
+
const microPrecision = totalTP + totalFP > 0
|
|
184
|
+
? totalTP / (totalTP + totalFP)
|
|
185
|
+
: 0;
|
|
186
|
+
const microRecall = totalTP + totalFN > 0
|
|
187
|
+
? totalTP / (totalTP + totalFN)
|
|
188
|
+
: 0;
|
|
189
|
+
const microF1 = microPrecision + microRecall > 0
|
|
190
|
+
? 2 * microPrecision * microRecall / (microPrecision + microRecall)
|
|
191
|
+
: 0;
|
|
192
|
+
const sortedByF1 = [...measured].sort((a, b) => b.f1 - a.f1);
|
|
193
|
+
const bestRepo = sortedByF1.length > 0 ? sortedByF1[0].repo : "N/A";
|
|
194
|
+
const worstRepo = sortedByF1.length > 1
|
|
195
|
+
? sortedByF1[sortedByF1.length - 1].repo
|
|
196
|
+
: "N/A";
|
|
197
|
+
const avgFPR = measured.length > 0
|
|
198
|
+
? measured.reduce((s, r) => s + r.falsePositiveRate, 0) / measured.length
|
|
199
|
+
: 0;
|
|
200
|
+
const avgFNR = measured.length > 0
|
|
201
|
+
? measured.reduce((s, r) => s + r.falseNegativeRate, 0) / measured.length
|
|
202
|
+
: 0;
|
|
203
|
+
let assessment = "INSUFFICIENT DATA";
|
|
204
|
+
if (measured.length >= 3) {
|
|
205
|
+
if (microF1 >= 0.80)
|
|
206
|
+
assessment = "PRODUCTION READY";
|
|
207
|
+
else if (microF1 >= 0.65)
|
|
208
|
+
assessment = "BETA QUALITY";
|
|
209
|
+
else if (microF1 >= 0.50)
|
|
210
|
+
assessment = "ALPHA — NEEDS MORE DATA";
|
|
211
|
+
else
|
|
212
|
+
assessment = "EARLY STAGE";
|
|
213
|
+
}
|
|
214
|
+
else if (measured.length >= 1) {
|
|
215
|
+
assessment = "PILOT — EXPAND LABELING";
|
|
216
|
+
}
|
|
217
|
+
return {
|
|
218
|
+
generated: new Date().toISOString(),
|
|
219
|
+
version: "3.2.0",
|
|
220
|
+
repos,
|
|
221
|
+
overall: {
|
|
222
|
+
reposMeasured: measured.length,
|
|
223
|
+
totalSamples,
|
|
224
|
+
totalTP,
|
|
225
|
+
totalFP,
|
|
226
|
+
totalFN,
|
|
227
|
+
macroF1,
|
|
228
|
+
microPrecision,
|
|
229
|
+
microRecall,
|
|
230
|
+
microF1,
|
|
231
|
+
bestRepo,
|
|
232
|
+
worstRepo,
|
|
233
|
+
assessment,
|
|
234
|
+
avgFPRate: avgFPR,
|
|
235
|
+
avgFNRate: avgFNR,
|
|
236
|
+
},
|
|
237
|
+
};
|
|
238
|
+
}
|
|
239
|
+
// ═══════════════════════════════════════════════════════════════
|
|
240
|
+
// Formatting
|
|
241
|
+
// ═══════════════════════════════════════════════════════════════
|
|
242
|
+
function formatReport(report) {
|
|
243
|
+
const lines = [];
|
|
244
|
+
const C = { bold: "", dim: "", green: "", red: "", yellow: "", cyan: "", reset: "" };
|
|
245
|
+
lines.push("");
|
|
246
|
+
lines.push("╔══════════════════════════════════════════════════════════════╗");
|
|
247
|
+
lines.push("║ Progmune Cross-Repository Precision Benchmark ║");
|
|
248
|
+
lines.push("╠══════════════════════════════════════════════════════════════╣");
|
|
249
|
+
lines.push(`║ Generated: ${report.generated} ║`);
|
|
250
|
+
lines.push(`║ Version: v${report.version} ║`);
|
|
251
|
+
lines.push("╚══════════════════════════════════════════════════════════════╝");
|
|
252
|
+
lines.push("");
|
|
253
|
+
// Per-repo table
|
|
254
|
+
const header = "┌─────────────────┬───────┬───────┬───────┬───────┬───────┬───────┬───────┐";
|
|
255
|
+
const sep = "├─────────────────┼───────┼───────┼───────┼───────┼───────┼───────┼───────┤";
|
|
256
|
+
const footer = "└─────────────────┴───────┴───────┴───────┴───────┴───────┴───────┴───────┘";
|
|
257
|
+
lines.push(header);
|
|
258
|
+
lines.push("│ Repo │ P │ R │ F1 │ FP% │ FN% │ N │ Rules │");
|
|
259
|
+
lines.push(sep);
|
|
260
|
+
for (const repo of report.repos) {
|
|
261
|
+
if (repo.status !== "measured") {
|
|
262
|
+
const status = repo.status === "no_labels" ? "no labels" : "error";
|
|
263
|
+
lines.push(`│ ${repo.repo.padEnd(15)} │ ${"-".padStart(3)} │ ${"-".padStart(3)} │ ${"-".padStart(3)} │ ${"-".padStart(3)} │ ${"-".padStart(3)} │ ${String(repo.total || 0).padStart(4)} │ ${"-".padStart(3)} │`);
|
|
264
|
+
continue;
|
|
265
|
+
}
|
|
266
|
+
const p = (repo.precision * 100).toFixed(0);
|
|
267
|
+
const r = (repo.recall * 100).toFixed(0);
|
|
268
|
+
const f = (repo.f1 * 100).toFixed(0);
|
|
269
|
+
const fpr = (repo.falsePositiveRate * 100).toFixed(0);
|
|
270
|
+
const fnr = (repo.falseNegativeRate * 100).toFixed(0);
|
|
271
|
+
// Color-code F1
|
|
272
|
+
let fDisplay = `${f}%`;
|
|
273
|
+
if (repo.f1 >= 0.7)
|
|
274
|
+
fDisplay = `${f}% ★`;
|
|
275
|
+
else if (repo.f1 >= 0.5)
|
|
276
|
+
fDisplay = `${f}%`;
|
|
277
|
+
lines.push(`│ ${repo.repo.padEnd(15)} │ ${p.padStart(3)}% │ ${r.padStart(3)}% │ ${fDisplay.padStart(5)} │ ${fpr.padStart(3)}% │ ${fnr.padStart(3)}% │ ${String(repo.total).padStart(4)} │ ${String(repo.rulesDiscovered).padStart(4)} │`);
|
|
278
|
+
}
|
|
279
|
+
lines.push(sep);
|
|
280
|
+
const o = report.overall;
|
|
281
|
+
const op = (o.microPrecision * 100).toFixed(0);
|
|
282
|
+
const or_ = (o.microRecall * 100).toFixed(0);
|
|
283
|
+
const of1 = (o.microF1 * 100).toFixed(0);
|
|
284
|
+
const maF1 = (o.macroF1 * 100).toFixed(0);
|
|
285
|
+
lines.push(`│ OVERALL (micro) │ ${op.padStart(3)}% │ ${or_.padStart(3)}% │ ${of1.padStart(3)}% │ ${(o.avgFPRate * 100).toFixed(0).padStart(3)}% │ ${(o.avgFNRate * 100).toFixed(0).padStart(3)}% │ ${String(o.totalSamples).padStart(4)} │ │`);
|
|
286
|
+
lines.push(`│ OVERALL (macro) │ │ │ ${maF1.padStart(3)}% │ │ │ │ │`);
|
|
287
|
+
lines.push(footer);
|
|
288
|
+
lines.push("");
|
|
289
|
+
// Summary
|
|
290
|
+
lines.push(`Repos measured: ${o.reposMeasured}/${report.repos.length}`);
|
|
291
|
+
lines.push(`Best repo: ${o.bestRepo}`);
|
|
292
|
+
lines.push(`Worst repo: ${o.worstRepo}`);
|
|
293
|
+
lines.push(`Avg FP Rate: ${(o.avgFPRate * 100).toFixed(1)}%`);
|
|
294
|
+
lines.push(`Avg FN Rate: ${(o.avgFNRate * 100).toFixed(1)}%`);
|
|
295
|
+
lines.push(`Assessment: ${o.assessment}`);
|
|
296
|
+
lines.push("");
|
|
297
|
+
// Mismatches detail
|
|
298
|
+
const reposWithMismatches = report.repos.filter(r => r.mismatches.length > 0);
|
|
299
|
+
if (reposWithMismatches.length > 0) {
|
|
300
|
+
lines.push("── Mismatch Details ──");
|
|
301
|
+
for (const repo of reposWithMismatches) {
|
|
302
|
+
lines.push(` ${repo.repo}: ${repo.mismatches.length} mismatches`);
|
|
303
|
+
for (const m of repo.mismatches.slice(0, 5)) {
|
|
304
|
+
const tag = m.expected === "clean" ? "FP" : "FN";
|
|
305
|
+
lines.push(` [${tag}] #${m.index}: expected ${m.expected}, got ${m.got}`);
|
|
306
|
+
lines.push(` ${m.calls.join(" → ")}`);
|
|
307
|
+
}
|
|
308
|
+
if (repo.mismatches.length > 5) {
|
|
309
|
+
lines.push(` ... and ${repo.mismatches.length - 5} more`);
|
|
310
|
+
}
|
|
311
|
+
}
|
|
312
|
+
lines.push("");
|
|
313
|
+
}
|
|
314
|
+
return lines.join("\n");
|
|
315
|
+
}
|
|
316
|
+
// ═══════════════════════════════════════════════════════════════
|
|
317
|
+
// Save
|
|
318
|
+
// ═══════════════════════════════════════════════════════════════
|
|
319
|
+
function saveReport(report) {
|
|
320
|
+
const reportsDir = path.resolve(process.cwd(), "benchmarks", "reports");
|
|
321
|
+
if (!fs.existsSync(reportsDir))
|
|
322
|
+
fs.mkdirSync(reportsDir, { recursive: true });
|
|
323
|
+
const date = new Date().toISOString().slice(0, 10);
|
|
324
|
+
const filePath = path.join(reportsDir, `cross-repo-precision-${date}.json`);
|
|
325
|
+
fs.writeFileSync(filePath, JSON.stringify(report, null, 2));
|
|
326
|
+
const latestPath = path.join(reportsDir, "cross-repo-precision-latest.json");
|
|
327
|
+
fs.writeFileSync(latestPath, JSON.stringify(report, null, 2));
|
|
328
|
+
console.log(`Reports saved:`);
|
|
329
|
+
console.log(` ${filePath}`);
|
|
330
|
+
console.log(` ${latestPath}`);
|
|
331
|
+
}
|
|
332
|
+
// ═══════════════════════════════════════════════════════════════
|
|
333
|
+
// Main
|
|
334
|
+
// ═══════════════════════════════════════════════════════════════
|
|
335
|
+
function main() {
|
|
336
|
+
const args = process.argv.slice(2);
|
|
337
|
+
const repoArg = args.find(a => a.startsWith("--repos="));
|
|
338
|
+
const repoNames = repoArg
|
|
339
|
+
? repoArg.replace("--repos=", "").split(",")
|
|
340
|
+
: [
|
|
341
|
+
"curl",
|
|
342
|
+
"libssh",
|
|
343
|
+
"nginx",
|
|
344
|
+
"redis",
|
|
345
|
+
// Future: add "nghttp2", "apache", "openssl" when labels are available
|
|
346
|
+
];
|
|
347
|
+
console.log(`Running precision measurement on ${repoNames.length} repos...`);
|
|
348
|
+
const report = generateReport(repoNames);
|
|
349
|
+
console.log(formatReport(report));
|
|
350
|
+
saveReport(report);
|
|
351
|
+
}
|
|
352
|
+
main();
|