progmune-runtime 2.1.6 → 3.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (356) hide show
  1. package/README.md +108 -468
  2. package/dist/ablation-study.js +144 -0
  3. package/dist/ablation-study.test.js +18 -0
  4. package/dist/action-runtime.js +3 -1
  5. package/dist/active-learning.js +211 -0
  6. package/dist/analytics.js +139 -0
  7. package/dist/asset-factory.js +309 -0
  8. package/dist/asset-growth.js +244 -0
  9. package/dist/asset-promotion.js +382 -0
  10. package/dist/asset-quality.js +550 -0
  11. package/dist/audit/business-translator.js +285 -0
  12. package/dist/audit/cli.js +66 -0
  13. package/dist/audit/formatters/html.js +379 -0
  14. package/dist/audit/formatters/json.js +11 -0
  15. package/dist/audit/formatters/markdown.js +192 -0
  16. package/dist/audit/formatters/terminal.js +189 -0
  17. package/dist/audit/index.js +25 -0
  18. package/dist/audit/report-builder.js +318 -0
  19. package/dist/audit/types.js +8 -0
  20. package/dist/audit.js +3 -3
  21. package/dist/auto-benchmark-generator.js +137 -0
  22. package/dist/auto-benchmark-generator.test.js +45 -0
  23. package/dist/auto-protocol-synthesizer.js +362 -0
  24. package/dist/auto-protocol-synthesizer.test.js +82 -0
  25. package/dist/autonomous-patch.js +175 -0
  26. package/dist/autonomous-patch.test.js +128 -0
  27. package/dist/badge/badge-server.js +98 -0
  28. package/dist/behavior-miner.js +442 -0
  29. package/dist/belief-layer.js +475 -0
  30. package/dist/benchmark-count.js +5 -0
  31. package/dist/benchmark-generator.js +211 -0
  32. package/dist/benchmark-harness.js +201 -0
  33. package/dist/benchmark-pass-rate.js +7 -0
  34. package/dist/benchmark-report.js +8 -3
  35. package/dist/benchmark-save.js +14 -1
  36. package/dist/bootstrap-validation.js +197 -0
  37. package/dist/bootstrap-validation.test.js +51 -0
  38. package/dist/branch-ledger.js +1 -1
  39. package/dist/capability-gap.js +130 -0
  40. package/dist/certify-html.js +351 -0
  41. package/dist/certify.js +326 -0
  42. package/dist/check.js +4 -4
  43. package/dist/compliance-miner.js +447 -0
  44. package/dist/continuous-benchmark.js +194 -0
  45. package/dist/continuous-benchmark.test.js +116 -0
  46. package/dist/corpus-stats.js +173 -0
  47. package/dist/counterfactual-engine.js +288 -0
  48. package/dist/coverage-dashboard.js +109 -0
  49. package/dist/coverage-system.test.js +205 -0
  50. package/dist/cross-repo-precision.js +352 -0
  51. package/dist/cve-benchmark.js +180 -0
  52. package/dist/cve-benchmark.test.js +28 -0
  53. package/dist/cve-collector.js +73 -0
  54. package/dist/data-quality.js +141 -0
  55. package/dist/decision-engine.js +388 -0
  56. package/dist/derive-metadata.js +250 -0
  57. package/dist/difficulty-active.test.js +198 -0
  58. package/dist/difficulty-map.js +244 -0
  59. package/dist/discovery-analytics.js +125 -0
  60. package/dist/discovery-model.js +149 -0
  61. package/dist/discovery-optimize.test.js +199 -0
  62. package/dist/discovery-trace.js +276 -0
  63. package/dist/discovery-trace.test.js +97 -0
  64. package/dist/emitter.js +83 -1
  65. package/dist/enterprise-dashboard.js +405 -0
  66. package/dist/eval-hardening.js +297 -0
  67. package/dist/eval-hardening.test.js +85 -0
  68. package/dist/evaluation-campaign.js +359 -0
  69. package/dist/evaluation-campaign.test.js +181 -0
  70. package/dist/evidence-growth.js +143 -0
  71. package/dist/evidence-repository.js +209 -0
  72. package/dist/evidence-system.js +441 -0
  73. package/dist/execute.js +15 -7
  74. package/dist/experimental/software-physics.js +291 -0
  75. package/dist/experimental/state-inference.js +516 -0
  76. package/dist/experimental/unsupervised-physics.js +230 -0
  77. package/dist/extract-ir-python.js +54 -7
  78. package/dist/extract-ir.js +376 -12
  79. package/dist/failure-collector.js +2 -2
  80. package/dist/failure-corpus.js +322 -9
  81. package/dist/feedback.js +16 -5
  82. package/dist/feedback.test.js +49 -0
  83. package/dist/file-lock.js +1 -1
  84. package/dist/flywheel-batch.js +292 -0
  85. package/dist/frameworks/express-cli.js +237 -0
  86. package/dist/frameworks/express-detector.js +445 -0
  87. package/dist/frameworks/express-detector.test.js +206 -0
  88. package/dist/frameworks/index.js +30 -0
  89. package/dist/frameworks/nestjs-detector.js +302 -0
  90. package/dist/frameworks/trpc-detector.js +161 -0
  91. package/dist/frameworks/version-awareness.js +179 -0
  92. package/dist/function-synonyms.js +164 -0
  93. package/dist/function-synonyms.test.js +68 -0
  94. package/dist/generalization.test.js +352 -0
  95. package/dist/goal-annotator.js +113 -0
  96. package/dist/goal-planner.js +563 -0
  97. package/dist/gold-cve.js +164 -0
  98. package/dist/gold-cve.test.js +104 -0
  99. package/dist/gold-quality.js +206 -0
  100. package/dist/gold-tiers.js +241 -0
  101. package/dist/governance-dashboard.js +327 -0
  102. package/dist/graph-viz.js +240 -0
  103. package/dist/guided-frontier.js +195 -0
  104. package/dist/hierarchical-planner.js +148 -0
  105. package/dist/identifier-parser.js +260 -0
  106. package/dist/immune-metrics.js +93 -0
  107. package/dist/immune-receiver.js +158 -0
  108. package/dist/immune-reporter.js +1 -1
  109. package/dist/improvement-orchestrator.js +206 -0
  110. package/dist/inject-p0-vocabulary.js +300 -0
  111. package/dist/intent-parser.js +218 -0
  112. package/dist/invariant-algebra.js +476 -0
  113. package/dist/invariant-calculus.js +533 -0
  114. package/dist/ir-utils.js +70 -0
  115. package/dist/ir-utils.test.js +50 -0
  116. package/dist/knowledge-api.js +312 -0
  117. package/dist/knowledge-evolution.js +452 -0
  118. package/dist/knowledge-explorer.js +506 -0
  119. package/dist/knowledge-flywheel.js +274 -0
  120. package/dist/knowledge-governance.js +338 -0
  121. package/dist/knowledge-governance.test.js +150 -0
  122. package/dist/knowledge-graph.js +181 -0
  123. package/dist/knowledge-guided-synth.js +246 -0
  124. package/dist/knowledge-loop.test.js +77 -0
  125. package/dist/knowledge-object.js +316 -0
  126. package/dist/knowledge-package.js +98 -0
  127. package/dist/kpi-dashboard.js +561 -0
  128. package/dist/l3-cross-function.js +280 -0
  129. package/dist/learning-ranker.js +148 -0
  130. package/dist/learning-ranker.test.js +291 -0
  131. package/dist/ledger/accountability.js +322 -0
  132. package/dist/ledger/chain-builder.js +185 -0
  133. package/dist/ledger/cli.js +222 -0
  134. package/dist/ledger/index.js +13 -0
  135. package/dist/ledger/signatures.js +193 -0
  136. package/dist/ledger/types.js +9 -0
  137. package/dist/llm.js +74 -3
  138. package/dist/load-benchmarks.js +8 -3
  139. package/dist/logger.js +66 -0
  140. package/dist/logger.test.js +37 -0
  141. package/dist/logistic-reward.js +339 -0
  142. package/dist/logistic-reward.test.js +180 -0
  143. package/dist/macro-graph.js +193 -0
  144. package/dist/macro-repair.js +183 -0
  145. package/dist/mcp-server.mjs +1202 -483
  146. package/dist/memory-layer.js +42 -5
  147. package/dist/multi-repo-precision.js +422 -0
  148. package/dist/name-free-protocol.js +425 -0
  149. package/dist/name-free-protocol.test.js +170 -0
  150. package/dist/name-scrambling.js +138 -0
  151. package/dist/name-scrambling.test.js +16 -0
  152. package/dist/p3-observability.test.js +281 -0
  153. package/dist/p5-orchestrator.test.js +225 -0
  154. package/dist/pairwise-preference.js +294 -0
  155. package/dist/pairwise-preference.test.js +140 -0
  156. package/dist/planner-constraints.js +104 -0
  157. package/dist/planner-prompts.js +155 -0
  158. package/dist/planner-telemetry.js +415 -0
  159. package/dist/planner-trace.js +214 -0
  160. package/dist/planner.js +162 -167
  161. package/dist/plsb/artifact.js +116 -0
  162. package/dist/plsb/cli.js +71 -0
  163. package/dist/plsb/index.js +19 -0
  164. package/dist/plsb/leaderboard.js +249 -0
  165. package/dist/plsb/report-md.js +156 -0
  166. package/dist/plsb/schema.js +179 -0
  167. package/dist/plsb-benchmark.js +284 -0
  168. package/dist/plsb-benchmark.test.js +119 -0
  169. package/dist/policy/cli.js +134 -0
  170. package/dist/policy/engine.js +333 -0
  171. package/dist/policy/index.js +12 -0
  172. package/dist/policy/types.js +59 -0
  173. package/dist/policy-miner.js +505 -0
  174. package/dist/precision-analyze.js +229 -0
  175. package/dist/precision-benchmark.js +147 -0
  176. package/dist/precision-label-c.js +134 -0
  177. package/dist/precision-label.js +193 -0
  178. package/dist/precision-report-c.js +149 -0
  179. package/dist/precision-report.js +246 -0
  180. package/dist/progmune-status.js +108 -0
  181. package/dist/proof-engine.js +479 -0
  182. package/dist/proof-provenance.js +315 -0
  183. package/dist/protocol-coverage.js +294 -0
  184. package/dist/protocol-detector.js +1189 -0
  185. package/dist/protocol-embedding-expanded.js +297 -0
  186. package/dist/protocol-embedding-expanded.test.js +97 -0
  187. package/dist/protocol-embedding.js +195 -0
  188. package/dist/protocol-embedding.test.js +82 -0
  189. package/dist/protocol-extractor-v2.js +354 -0
  190. package/dist/protocol-extractor-v2.test.js +140 -0
  191. package/dist/protocol-extractor.js +310 -0
  192. package/dist/protocol-extractor.test.js +113 -0
  193. package/dist/protocol-foundation.js +322 -0
  194. package/dist/protocol-foundation.test.js +163 -0
  195. package/dist/protocol-frontier.js +243 -0
  196. package/dist/protocol-frontier.test.js +92 -0
  197. package/dist/protocol-gap-analyzer.js +228 -0
  198. package/dist/protocol-gap-analyzer.test.js +49 -0
  199. package/dist/protocol-invariants.js +276 -0
  200. package/dist/protocol-invariants.test.js +111 -0
  201. package/dist/protocol-knowledge.js +464 -0
  202. package/dist/protocol-miner.js +343 -0
  203. package/dist/protocol-mining.js +207 -0
  204. package/dist/protocol-mining.test.js +37 -0
  205. package/dist/protocol-registry.js +1 -1
  206. package/dist/protocol-security-benchmark.js +222 -0
  207. package/dist/protocol-vulnerability.js +257 -0
  208. package/dist/protocol-vulnerability.test.js +60 -0
  209. package/dist/python-benchmark.js +120 -0
  210. package/dist/python-emitter.js +163 -45
  211. package/dist/python-protocol-extractor.js +187 -0
  212. package/dist/python-protocol-extractor.test.js +116 -0
  213. package/dist/realworld-benchmark.js +646 -0
  214. package/dist/realworld-benchmark.test.js +36 -0
  215. package/dist/repair-arch.test.js +411 -0
  216. package/dist/repair-evolution.test.js +454 -0
  217. package/dist/repair-executor.js +719 -0
  218. package/dist/repair-proposal.js +4 -4
  219. package/dist/repair-ranker.js +141 -0
  220. package/dist/repair-strategies.js +419 -0
  221. package/dist/repair-taxonomy.js +234 -0
  222. package/dist/repair-types.js +12 -0
  223. package/dist/repo-evaluator.js +250 -0
  224. package/dist/repo-evaluator.test.js +128 -0
  225. package/dist/resource-abstraction.js +242 -0
  226. package/dist/resource-detector.js +211 -0
  227. package/dist/result.test.js +43 -0
  228. package/dist/reward-system.js +411 -0
  229. package/dist/reward-system.test.js +175 -0
  230. package/dist/risk-model.js +215 -0
  231. package/dist/rule-miner.js +234 -7
  232. package/dist/rule-specificity.js +254 -0
  233. package/dist/runtime-types.js +27 -0
  234. package/dist/scaffold.js +208 -0
  235. package/dist/scale-collector.test.js +101 -0
  236. package/dist/scale-trajectory-collector.js +128 -0
  237. package/dist/sdk.js +250 -0
  238. package/dist/search-planner.js +4 -41
  239. package/dist/semantic-snapshot.js +1 -1
  240. package/dist/semantic-topology.js +121 -0
  241. package/dist/semantic-trace.js +310 -317
  242. package/dist/sequence-extractor.js +343 -0
  243. package/dist/skill-library.js +245 -0
  244. package/dist/skill-planner.test.js +189 -0
  245. package/dist/software-physics.js +291 -0
  246. package/dist/software-physics.test.js +81 -0
  247. package/dist/ssg-precision.js +478 -0
  248. package/dist/ssg-validator.js +71 -21
  249. package/dist/state-inference-doubleblind.test.js +160 -0
  250. package/dist/state-inference.js +516 -0
  251. package/dist/state-inference.test.js +115 -0
  252. package/dist/state-machine-fingerprint.js +345 -0
  253. package/dist/state-machine-fingerprint.test.js +120 -0
  254. package/dist/state-miner.js +386 -0
  255. package/dist/state-name-inference.js +213 -0
  256. package/dist/state-name-inference.test.js +69 -0
  257. package/dist/strategy-planner.js +262 -96
  258. package/dist/strategy-planner.test.js +135 -0
  259. package/dist/telemetry-analytics.test.js +402 -0
  260. package/dist/terminal-format.js +68 -0
  261. package/dist/terminal-format.test.js +83 -0
  262. package/dist/topology-factory.js +196 -0
  263. package/dist/topology-representation.js +242 -0
  264. package/dist/topology-representation.test.js +27 -0
  265. package/dist/trajectory-augmentation.js +254 -0
  266. package/dist/trajectory-augmentation.test.js +63 -0
  267. package/dist/trajectory-corpus.js +440 -0
  268. package/dist/trajectory-corpus.test.js +32 -0
  269. package/dist/trajectory-feedback.test.js +116 -0
  270. package/dist/transition-synthesizer.js +286 -0
  271. package/dist/transition-synthesizer.test.js +123 -0
  272. package/dist/trust/api-semantic-mapper.js +809 -0
  273. package/dist/trust/call-graph-propagator.js +225 -0
  274. package/dist/trust/cli.js +122 -0
  275. package/dist/trust/compliance-scorer.js +283 -0
  276. package/dist/trust/confidence-calculator.js +261 -0
  277. package/dist/trust/engine.js +1145 -0
  278. package/dist/trust/explainability.js +85 -0
  279. package/dist/trust/formatters/ci.js +42 -0
  280. package/dist/trust/formatters/json.js +11 -0
  281. package/dist/trust/formatters/terminal.js +152 -0
  282. package/dist/trust/index.js +39 -0
  283. package/dist/trust/phase1-verify.js +171 -0
  284. package/dist/trust/protocol-domain-validator.js +697 -0
  285. package/dist/trust/score-calculator.js +282 -0
  286. package/dist/trust/ssg-bridge.js +641 -0
  287. package/dist/trust/ssg-bridge.test.js +269 -0
  288. package/dist/trust/types.js +67 -0
  289. package/dist/trust/violation-trace.js +335 -0
  290. package/dist/trust-api.js +179 -0
  291. package/dist/trust-calibration.js +279 -0
  292. package/dist/unknown-protocol-discovery.js +339 -0
  293. package/dist/unknown-protocol-discovery.test.js +102 -0
  294. package/dist/unsupervised-physics.js +230 -0
  295. package/dist/unsupervised-physics.test.js +95 -0
  296. package/dist/utils.test.js +37 -0
  297. package/dist/validator.js +187 -10
  298. package/dist/verification-intelligence.js +475 -0
  299. package/dist/verify-api.js +432 -0
  300. package/dist/vi-impact-report.js +293 -0
  301. package/dist/wl-fingerprint.js +162 -0
  302. package/dist/wl-fingerprint.test.js +130 -0
  303. package/dist/zeroshot-strategy.js +139 -0
  304. package/docs/Progmune_/346/212/225/350/265/204/344/272/272/347/231/275/347/232/256/344/271/246_v2.0.html +576 -0
  305. package/docs/Progmune_/351/241/271/347/233/256/345/205/250/350/247/243.html +710 -0
  306. package/package.json +74 -7
  307. package/protocols.json +1956 -50
  308. package/.dockerignore +0 -14
  309. package/.mcp.json +0 -11
  310. package/.progmune_allowlist +0 -50
  311. package/.test_report/test_report.md +0 -87
  312. package/Dockerfile +0 -9
  313. package/FAQ.md +0 -167
  314. package/WHITEPAPER.md +0 -540
  315. package/demo-project/auth.ts +0 -55
  316. package/demo-project/tsconfig.json +0 -8
  317. package/dist/acl-breakdown.js +0 -13
  318. package/dist/all-sessions.js +0 -11
  319. package/dist/antibody-stats.js +0 -11
  320. package/dist/branch-tree-count.js +0 -14
  321. package/dist/common-fixpath.js +0 -12
  322. package/dist/constraint-types.js +0 -12
  323. package/dist/exec-metrics.js +0 -11
  324. package/dist/failure-report.js +0 -11
  325. package/dist/fast-path-hits.js +0 -13
  326. package/dist/fingerprint-list.js +0 -15
  327. package/dist/gen-history-log.js +0 -13
  328. package/dist/heatmap-data.js +0 -11
  329. package/dist/recent-session.js +0 -12
  330. package/dist/svl-distribution.js +0 -11
  331. package/dist/terminal-status.js +0 -11
  332. package/dist/token-savings.js +0 -11
  333. package/dist/total-repairs.js +0 -12
  334. package/dist/unresolved-count.js +0 -12
  335. package/dist/valid-fingerprints.js +0 -13
  336. package/dist/verify-ledgers.js +0 -11
  337. package/docs/whitepaper-style.css +0 -77
  338. package/docs/whitepaper-v2.1.md +0 -609
  339. package/docs/whitepaper-v2.2.md +0 -1064
  340. package/docs/whitepaper-v2.2.pdf +0 -0
  341. package/fly.toml +0 -31
  342. package/public/dashboard.html +0 -119
  343. package/server/hub.js +0 -116
  344. package/test/replay-golden/sess_1780063202050_mgeld.json +0 -9
  345. package/test/replay-golden/sess_1780064032560_gocld.json +0 -354
  346. package/test/replay-golden/sess_1780064413331_s2709.json +0 -606
  347. package/test/replay-golden/sess_1780064792710_y3avo.json +0 -614
  348. package/test/replay-golden.ts +0 -84
  349. package/test_benchmark.js +0 -165
  350. package/test_comprehensive.mjs +0 -638
  351. package/test_concurrency.js +0 -129
  352. package/test_ir_robustness.js +0 -85
  353. package/test_semantic_contracts.js +0 -269
  354. package/test_ssg_stress.js +0 -156
  355. package/test_svl3.js +0 -58
  356. package/tsconfig.json +0 -17
@@ -0,0 +1,109 @@
1
+ "use strict";
2
+ /**
3
+ * P3.6: Coverage Dashboard
4
+ *
5
+ * Visualizes protocol coverage gaps, ranks protocols by risk,
6
+ * and generates acquisition priorities.
7
+ *
8
+ * The dashboard answers:
9
+ * - Which protocols are fully observed? Which are data-poor?
10
+ * - Where should we collect more data next?
11
+ * - What transitions should we benchmark first?
12
+ */
13
+ Object.defineProperty(exports, "__esModule", { value: true });
14
+ exports.assessRisk = assessRisk;
15
+ exports.generateCoverageDashboard = generateCoverageDashboard;
16
+ exports.printCoverageDashboard = printCoverageDashboard;
17
+ const protocol_coverage_1 = require("./protocol-coverage");
18
+ const failure_corpus_1 = require("./failure-corpus");
19
+ function assessRisk(report) {
20
+ const sc = report.stateCoverage.stateCoverage;
21
+ const tc = report.transitionCoverage.transitionCoverage;
22
+ const avg = (sc + tc) / 2;
23
+ let risk;
24
+ let recommendation;
25
+ if (avg < 0.25) {
26
+ risk = "critical";
27
+ recommendation = `Immediate: add ${report.transitionCoverage.missingTransitions.length} benchmark cases for uncovered transitions`;
28
+ }
29
+ else if (avg < 0.50) {
30
+ risk = "high";
31
+ recommendation = `Priority: focus on missing ${report.stateCoverage.missingStates.length} states and ${report.transitionCoverage.missingTransitions.length} transitions`;
32
+ }
33
+ else if (avg < 0.75) {
34
+ risk = "medium";
35
+ recommendation = `Fill remaining gaps: ${report.transitionCoverage.missingTransitions.length} transitions uncovered`;
36
+ }
37
+ else {
38
+ risk = "low";
39
+ recommendation = "Well-covered. Monitor for regressions.";
40
+ }
41
+ return {
42
+ protocol: report.protocol,
43
+ stateCoverage: sc,
44
+ transitionCoverage: tc,
45
+ trajectoryCount: report.trajectoryCount,
46
+ risk,
47
+ missingTransitionCount: report.transitionCoverage.missingTransitions.length,
48
+ recommendation,
49
+ };
50
+ }
51
+ function generateCoverageDashboard(trajectories) {
52
+ const trajs = trajectories || (0, failure_corpus_1.loadTrajectories)();
53
+ const protocols = (0, protocol_coverage_1.loadDefaultProtocolDefinitions)();
54
+ const reports = (0, protocol_coverage_1.analyzeAllCoverage)(protocols, trajs);
55
+ const risks = reports.map(assessRisk).sort((a, b) => a.transitionCoverage - b.transitionCoverage);
56
+ const totalStates = reports.reduce((s, r) => s + r.stateCoverage.totalStates, 0);
57
+ const visitedStates = reports.reduce((s, r) => s + r.stateCoverage.visitedStates, 0);
58
+ const totalTrans = reports.reduce((s, r) => s + r.transitionCoverage.totalTransitions, 0);
59
+ const visitedTrans = reports.reduce((s, r) => s + r.transitionCoverage.visitedTransitions, 0);
60
+ return {
61
+ reports,
62
+ riskRanking: risks,
63
+ overallStateCoverage: totalStates > 0 ? visitedStates / totalStates : 0,
64
+ overallTransitionCoverage: totalTrans > 0 ? visitedTrans / totalTrans : 0,
65
+ totalTrajectories: trajs.length,
66
+ criticalProtocols: risks.filter(r => r.risk === "critical" || r.risk === "high").length,
67
+ };
68
+ }
69
+ function printCoverageDashboard(dashboard) {
70
+ console.log("\n╔════════════════════════════════════════════════════╗");
71
+ console.log("║ Protocol Coverage Dashboard ║");
72
+ console.log("╚════════════════════════════════════════════════════╝\n");
73
+ console.log(`Trajectories Analyzed: ${dashboard.totalTrajectories}`);
74
+ console.log(`Overall State Coverage: ${(dashboard.overallStateCoverage * 100).toFixed(0)}%`);
75
+ console.log(`Overall Transition Coverage: ${(dashboard.overallTransitionCoverage * 100).toFixed(0)}%`);
76
+ console.log(`Critical/High Risk Protocols: ${dashboard.criticalProtocols}\n`);
77
+ console.log("─── By Protocol ───");
78
+ console.log("Protocol State Trans Trajs Risk");
79
+ console.log("────────────────────────────────────────────────");
80
+ for (const r of dashboard.riskRanking) {
81
+ const riskIcon = r.risk === "critical" ? "🔴" : r.risk === "high" ? "🟠" : r.risk === "medium" ? "🟡" : "🟢";
82
+ const sc = (r.stateCoverage * 100).toFixed(0).padStart(3);
83
+ const tc = (r.transitionCoverage * 100).toFixed(0).padStart(3);
84
+ console.log(` ${r.protocol.padEnd(16)} ${sc}% ${tc}% ${String(r.trajectoryCount).padStart(4)} ${riskIcon} ${r.risk}`);
85
+ }
86
+ console.log();
87
+ // Missing transitions detail for high-risk protocols
88
+ const criticalReports = dashboard.reports.filter(r => assessRisk(r).risk === "critical" || assessRisk(r).risk === "high");
89
+ if (criticalReports.length > 0) {
90
+ console.log("─── Highest Risk: Missing Transitions ───");
91
+ for (const r of criticalReports) {
92
+ const missing = r.transitionCoverage.missingTransitions.slice(0, 5);
93
+ if (missing.length === 0)
94
+ continue;
95
+ console.log(`\n ${r.protocol} (${r.transitionCoverage.missingTransitions.length} missing):`);
96
+ for (const m of missing) {
97
+ console.log(` ${m.from} → ${m.to} (via ${m.rule})`);
98
+ }
99
+ }
100
+ console.log();
101
+ }
102
+ console.log("─── Recommendations ───");
103
+ for (const r of dashboard.riskRanking) {
104
+ if (r.risk === "critical" || r.risk === "high") {
105
+ console.log(` ${r.protocol}: ${r.recommendation}`);
106
+ }
107
+ }
108
+ console.log();
109
+ }
@@ -0,0 +1,205 @@
1
+ "use strict";
2
+ /**
3
+ * P3.6: Coverage System Integration Tests
4
+ *
5
+ * Verifying:
6
+ * 1. Coverage engine correctly computes state/transition coverage
7
+ * 2. Dashboard visualizes gaps and risk ranking
8
+ * 3. Benchmark generator produces cases for uncovered transitions
9
+ * 4. End-to-end: analyze → generate → new cases
10
+ */
11
+ var __createBinding = (this && this.__createBinding) || (Object.create ? (function(o, m, k, k2) {
12
+ if (k2 === undefined) k2 = k;
13
+ var desc = Object.getOwnPropertyDescriptor(m, k);
14
+ if (!desc || ("get" in desc ? !m.__esModule : desc.writable || desc.configurable)) {
15
+ desc = { enumerable: true, get: function() { return m[k]; } };
16
+ }
17
+ Object.defineProperty(o, k2, desc);
18
+ }) : (function(o, m, k, k2) {
19
+ if (k2 === undefined) k2 = k;
20
+ o[k2] = m[k];
21
+ }));
22
+ var __setModuleDefault = (this && this.__setModuleDefault) || (Object.create ? (function(o, v) {
23
+ Object.defineProperty(o, "default", { enumerable: true, value: v });
24
+ }) : function(o, v) {
25
+ o["default"] = v;
26
+ });
27
+ var __importStar = (this && this.__importStar) || (function () {
28
+ var ownKeys = function(o) {
29
+ ownKeys = Object.getOwnPropertyNames || function (o) {
30
+ var ar = [];
31
+ for (var k in o) if (Object.prototype.hasOwnProperty.call(o, k)) ar[ar.length] = k;
32
+ return ar;
33
+ };
34
+ return ownKeys(o);
35
+ };
36
+ return function (mod) {
37
+ if (mod && mod.__esModule) return mod;
38
+ var result = {};
39
+ if (mod != null) for (var k = ownKeys(mod), i = 0; i < k.length; i++) if (k[i] !== "default") __createBinding(result, mod, k[i]);
40
+ __setModuleDefault(result, mod);
41
+ return result;
42
+ };
43
+ })();
44
+ Object.defineProperty(exports, "__esModule", { value: true });
45
+ const vitest_1 = require("vitest");
46
+ const fs = __importStar(require("fs"));
47
+ const path = __importStar(require("path"));
48
+ const protocol_coverage_1 = require("./protocol-coverage");
49
+ const coverage_dashboard_1 = require("./coverage-dashboard");
50
+ const benchmark_generator_1 = require("./benchmark-generator");
51
+ // ═══════════════════════════════════════════════════════════════
52
+ // Coverage Engine
53
+ // ═══════════════════════════════════════════════════════════════
54
+ (0, vitest_1.describe)("Coverage Engine", () => {
55
+ function makeFileRules() {
56
+ return new Map([
57
+ ["open_file", { pre_states: [], post_states: ["FILE_OPEN"] }],
58
+ ["write_file", { pre_states: ["FILE_OPEN"], post_states: ["FILE_DIRTY"] }],
59
+ ["close_file", { pre_states: ["FILE_OPEN", "FILE_DIRTY"], post_states: [], invalidate: ["FILE_OPEN", "FILE_DIRTY"] }],
60
+ ]);
61
+ }
62
+ const fileProto = (0, protocol_coverage_1.parseProtocolDefinition)("FileProtocol", makeFileRules(), "INIT");
63
+ (0, vitest_1.it)("computes full coverage when all transitions visited", () => {
64
+ const trajectories = [{
65
+ id: "t1", timestamp: new Date().toISOString(),
66
+ protocol: "FileProtocol", initialState: ["INIT"], finalState: [],
67
+ trajectory: ["open_file", "write_file", "close_file"],
68
+ result: "success", context: { nestingDepth: 0, exceptionHandled: false, insideLoop: false, branchCount: 0, asyncContext: false },
69
+ successRate: 1.0, metadata: { source: "human" },
70
+ }];
71
+ const report = (0, protocol_coverage_1.analyzeCoverage)(fileProto, trajectories);
72
+ (0, vitest_1.expect)(report.transitionCoverage.transitionCoverage).toBeGreaterThan(0.5);
73
+ (0, vitest_1.expect)(report.stateCoverage.stateCoverage).toBeGreaterThan(0.5);
74
+ });
75
+ (0, vitest_1.it)("detects uncovered transitions", () => {
76
+ const trajectories = [{
77
+ id: "t2", timestamp: new Date().toISOString(),
78
+ protocol: "FileProtocol", initialState: ["INIT"], finalState: [],
79
+ trajectory: ["open_file", "close_file"], // missing write_file
80
+ result: "success", context: { nestingDepth: 0, exceptionHandled: false, insideLoop: false, branchCount: 0, asyncContext: false },
81
+ successRate: 1.0, metadata: { source: "human" },
82
+ }];
83
+ const report = (0, protocol_coverage_1.analyzeCoverage)(fileProto, trajectories);
84
+ (0, vitest_1.expect)(report.transitionCoverage.missingTransitions.length).toBeGreaterThan(0);
85
+ });
86
+ (0, vitest_1.it)("empty trajectories = zero coverage", () => {
87
+ const report = (0, protocol_coverage_1.analyzeCoverage)(fileProto, []);
88
+ (0, vitest_1.expect)(report.transitionCoverage.transitionCoverage).toBe(0);
89
+ (0, vitest_1.expect)(report.stateCoverage.stateCoverage).toBe(0);
90
+ (0, vitest_1.expect)(report.trajectoryCount).toBe(0);
91
+ });
92
+ });
93
+ // ═══════════════════════════════════════════════════════════════
94
+ // Default Protocol Definitions
95
+ // ═══════════════════════════════════════════════════════════════
96
+ (0, vitest_1.describe)("Default Protocol Definitions", () => {
97
+ (0, vitest_1.it)("loads all 9 protocol groups", () => {
98
+ const protocols = (0, protocol_coverage_1.loadDefaultProtocolDefinitions)();
99
+ (0, vitest_1.expect)(protocols.length).toBeGreaterThanOrEqual(9);
100
+ const names = protocols.map(p => p.name);
101
+ (0, vitest_1.expect)(names).toContain("FileProtocol");
102
+ (0, vitest_1.expect)(names).toContain("AuthProtocol");
103
+ (0, vitest_1.expect)(names).toContain("DBProtocol");
104
+ (0, vitest_1.expect)(names).toContain("IRProtocol");
105
+ (0, vitest_1.expect)(names).toContain("StatelessProtocol");
106
+ (0, vitest_1.expect)(names).toContain("TransactionProtocol");
107
+ (0, vitest_1.expect)(names).toContain("ConditionalProtocol");
108
+ (0, vitest_1.expect)(names).toContain("LoopProtocol");
109
+ (0, vitest_1.expect)(names).toContain("CrossProtocol");
110
+ });
111
+ (0, vitest_1.it)("each protocol (except stateless) has states and transitions", () => {
112
+ for (const p of (0, protocol_coverage_1.loadDefaultProtocolDefinitions)()) {
113
+ (0, vitest_1.expect)(p.states.length).toBeGreaterThan(0);
114
+ // StatelessProtocol has empty pre/post states → 0 acquire transitions
115
+ if (p.name === "StatelessProtocol") {
116
+ (0, vitest_1.expect)(p.transitions.length).toBeGreaterThanOrEqual(0);
117
+ }
118
+ else {
119
+ (0, vitest_1.expect)(p.transitions.length).toBeGreaterThan(0);
120
+ }
121
+ }
122
+ });
123
+ (0, vitest_1.it)("FileProtocol has open/write/close transitions", () => {
124
+ const file = (0, protocol_coverage_1.loadDefaultProtocolDefinitions)().find(p => p.name === "FileProtocol");
125
+ const tKeys = file.transitions.map(t => `${t.from}→${t.to}`);
126
+ (0, vitest_1.expect)(tKeys).toContain("INIT→FILE_OPEN"); // open_file
127
+ (0, vitest_1.expect)(tKeys).toContain("FILE_OPEN→∅"); // close_file invalidates FILE_OPEN
128
+ });
129
+ (0, vitest_1.it)("AuthProtocol has auth lifecycle transitions", () => {
130
+ const auth = (0, protocol_coverage_1.loadDefaultProtocolDefinitions)().find(p => p.name === "AuthProtocol");
131
+ const tKeys = auth.transitions.map(t => `${t.from}→${t.to}`);
132
+ (0, vitest_1.expect)(tKeys).toContain("UNAUTHENTICATED→PASSWORD_VERIFIED"); // verify_password
133
+ (0, vitest_1.expect)(tKeys).toContain("PASSWORD_VERIFIED→TOKEN_ISSUED"); // generate_jwt
134
+ (0, vitest_1.expect)(tKeys).toContain("TOKEN_ISSUED→SESSION_ACTIVE"); // create_session
135
+ (0, vitest_1.expect)(tKeys).toContain("SESSION_ACTIVE→UNAUTHENTICATED"); // logout
136
+ });
137
+ });
138
+ // ═══════════════════════════════════════════════════════════════
139
+ // Coverage Dashboard
140
+ // ═══════════════════════════════════════════════════════════════
141
+ const GEN_DIR = path.resolve(__dirname, "..", "test-coverage-gen");
142
+ process.env.PROGMUNE_PROJECT_DIR = GEN_DIR;
143
+ fs.mkdirSync(GEN_DIR, { recursive: true });
144
+ fs.mkdirSync(path.join(GEN_DIR, ".progmune_corpus", "trajectories"), { recursive: true });
145
+ (0, vitest_1.describe)("Coverage Dashboard", () => {
146
+ (0, vitest_1.it)("generates dashboard from current trajectories", () => {
147
+ const dashboard = (0, coverage_dashboard_1.generateCoverageDashboard)([]);
148
+ (0, vitest_1.expect)(dashboard.reports.length).toBeGreaterThanOrEqual(9);
149
+ (0, vitest_1.expect)(dashboard.riskRanking.length).toBeGreaterThanOrEqual(9);
150
+ (0, vitest_1.expect)(dashboard.overallTransitionCoverage).toBeGreaterThanOrEqual(0);
151
+ (0, vitest_1.expect)(dashboard.overallTransitionCoverage).toBeLessThanOrEqual(1);
152
+ (0, vitest_1.expect)(dashboard.criticalProtocols).toBeGreaterThanOrEqual(0);
153
+ (0, coverage_dashboard_1.printCoverageDashboard)(dashboard);
154
+ });
155
+ (0, vitest_1.it)("correctly ranks empty protocols as critical", () => {
156
+ const dashboard = (0, coverage_dashboard_1.generateCoverageDashboard)([]);
157
+ // With zero trajectories, all protocols should be critical or high risk
158
+ const emptyProtocols = dashboard.riskRanking.filter(r => r.trajectoryCount === 0);
159
+ for (const r of emptyProtocols) {
160
+ (0, vitest_1.expect)(r.stateCoverage).toBe(0);
161
+ (0, vitest_1.expect)(r.transitionCoverage).toBe(0);
162
+ (0, vitest_1.expect)(r.risk).toBe("critical");
163
+ }
164
+ });
165
+ });
166
+ // ═══════════════════════════════════════════════════════════════
167
+ // Benchmark Generator
168
+ // ═══════════════════════════════════════════════════════════════
169
+ (0, vitest_1.describe)("Benchmark Generator", () => {
170
+ (0, vitest_1.it)("generates cases for uncovered transitions", () => {
171
+ const generated = (0, benchmark_generator_1.generateMissingBenchmarks)([]);
172
+ // With zero trajectories, all protocols have uncovered transitions
173
+ (0, vitest_1.expect)(Object.keys(generated).length).toBeGreaterThanOrEqual(3);
174
+ // Each protocol should have generated cases
175
+ for (const [protocol, cases] of Object.entries(generated)) {
176
+ (0, vitest_1.expect)(cases.length).toBeGreaterThan(0);
177
+ for (const c of cases) {
178
+ (0, vitest_1.expect)(c.broken.length).toBeGreaterThan(0);
179
+ (0, vitest_1.expect)(c.expected.length).toBeGreaterThan(0);
180
+ (0, vitest_1.expect)(c.expected.length).toBeGreaterThan(c.broken.length);
181
+ (0, vitest_1.expect)(["resource_leak", "missing_prerequisite"]).toContain(c.violationType);
182
+ }
183
+ }
184
+ });
185
+ (0, vitest_1.it)("writes generated benchmarks to disk", () => {
186
+ const generated = (0, benchmark_generator_1.generateMissingBenchmarks)([]);
187
+ const outDir = path.resolve(GEN_DIR, "generated-benchmarks");
188
+ const written = (0, benchmark_generator_1.writeGeneratedBenchmarks)(generated, outDir);
189
+ (0, vitest_1.expect)(written.length).toBeGreaterThanOrEqual(3);
190
+ // Verify files exist and are valid JSON
191
+ for (const filepath of written) {
192
+ (0, vitest_1.expect)(fs.existsSync(filepath)).toBe(true);
193
+ const content = JSON.parse(fs.readFileSync(filepath, "utf-8"));
194
+ (0, vitest_1.expect)(content.cases.length).toBeGreaterThan(0);
195
+ (0, vitest_1.expect)(content.source).toBe("coverage-gap");
196
+ }
197
+ });
198
+ (0, vitest_1.it)("runs the full coverage→generation pipeline", () => {
199
+ const result = (0, benchmark_generator_1.runCoverageDrivenGeneration)();
200
+ (0, vitest_1.expect)(result.existingCases).toBeGreaterThanOrEqual(1); // from previous test writes
201
+ (0, vitest_1.expect)(result.generatedCases).toBeGreaterThanOrEqual(10);
202
+ (0, vitest_1.expect)(result.writtenFiles.length).toBeGreaterThanOrEqual(3);
203
+ console.log(`\nCoverage-Driven Generation: ${result.summary}`);
204
+ });
205
+ });
@@ -0,0 +1,352 @@
1
+ "use strict";
2
+ /**
3
+ * P1: Cross-Repository Precision Runner
4
+ *
5
+ * Runs the full SSG precision pipeline (discover → validate → measure)
6
+ * on every benchmark repo that has labeled sequences.
7
+ *
8
+ * Produces:
9
+ * benchmarks/reports/cross-repo-precision-<date>.json
10
+ * benchmarks/reports/cross-repo-precision-latest.json
11
+ *
12
+ * Usage:
13
+ * npx ts-node --transpile-only src/cross-repo-precision.ts
14
+ * npx ts-node --transpile-only src/cross-repo-precision.ts --repos curl,libssh
15
+ */
16
+ var __createBinding = (this && this.__createBinding) || (Object.create ? (function(o, m, k, k2) {
17
+ if (k2 === undefined) k2 = k;
18
+ var desc = Object.getOwnPropertyDescriptor(m, k);
19
+ if (!desc || ("get" in desc ? !m.__esModule : desc.writable || desc.configurable)) {
20
+ desc = { enumerable: true, get: function() { return m[k]; } };
21
+ }
22
+ Object.defineProperty(o, k2, desc);
23
+ }) : (function(o, m, k, k2) {
24
+ if (k2 === undefined) k2 = k;
25
+ o[k2] = m[k];
26
+ }));
27
+ var __setModuleDefault = (this && this.__setModuleDefault) || (Object.create ? (function(o, v) {
28
+ Object.defineProperty(o, "default", { enumerable: true, value: v });
29
+ }) : function(o, v) {
30
+ o["default"] = v;
31
+ });
32
+ var __importStar = (this && this.__importStar) || (function () {
33
+ var ownKeys = function(o) {
34
+ ownKeys = Object.getOwnPropertyNames || function (o) {
35
+ var ar = [];
36
+ for (var k in o) if (Object.prototype.hasOwnProperty.call(o, k)) ar[ar.length] = k;
37
+ return ar;
38
+ };
39
+ return ownKeys(o);
40
+ };
41
+ return function (mod) {
42
+ if (mod && mod.__esModule) return mod;
43
+ var result = {};
44
+ if (mod != null) for (var k = ownKeys(mod), i = 0; i < k.length; i++) if (k[i] !== "default") __createBinding(result, mod, k[i]);
45
+ __setModuleDefault(result, mod);
46
+ return result;
47
+ };
48
+ })();
49
+ Object.defineProperty(exports, "__esModule", { value: true });
50
+ const fs = __importStar(require("fs"));
51
+ const path = __importStar(require("path"));
52
+ // ═══════════════════════════════════════════════════════════════
53
+ // Per-Repo Precision Runner
54
+ // ═══════════════════════════════════════════════════════════════
55
+ function runPrecisionForRepo(repoName) {
56
+ const benchmarksDir = path.resolve(process.cwd(), "benchmarks");
57
+ const labelFile = path.join(benchmarksDir, `${repoName}-labels.json`);
58
+ const empty = {
59
+ repo: repoName,
60
+ status: "no_labels",
61
+ total: 0,
62
+ cleanLabels: 0,
63
+ violationLabels: 0,
64
+ tp: 0, fp: 0, tn: 0, fn: 0,
65
+ precision: 0, recall: 0, f1: 0,
66
+ falsePositiveRate: 0, falseNegativeRate: 0,
67
+ rulesDiscovered: 0,
68
+ mismatches: [],
69
+ };
70
+ if (!fs.existsSync(labelFile)) {
71
+ empty.error = `No labels file: ${labelFile}`;
72
+ return empty;
73
+ }
74
+ let data;
75
+ try {
76
+ data = JSON.parse(fs.readFileSync(labelFile, "utf-8"));
77
+ }
78
+ catch (e) {
79
+ empty.status = "error";
80
+ empty.error = `Parse error: ${e}`;
81
+ return empty;
82
+ }
83
+ const labels = data.labels || {};
84
+ const sequences = data.sequences || {};
85
+ const labeledIndices = Object.keys(labels).map(Number);
86
+ if (labeledIndices.length === 0) {
87
+ empty.error = "No labeled sequences";
88
+ return empty;
89
+ }
90
+ // Count label distribution
91
+ let cleanLabels = 0;
92
+ let violationLabels = 0;
93
+ const cleanSeqs = [];
94
+ for (const idx of labeledIndices) {
95
+ if (labels[idx] === "clean") {
96
+ cleanLabels++;
97
+ if (sequences[idx])
98
+ cleanSeqs.push(sequences[idx]);
99
+ }
100
+ else if (labels[idx] === "violation") {
101
+ violationLabels++;
102
+ }
103
+ }
104
+ // Discover SSG rules from clean sequences
105
+ let rules;
106
+ let nsInit;
107
+ try {
108
+ const { discoverRulesFromSequences, validateSequenceWithSSG } = require("./ssg-precision");
109
+ const result = discoverRulesFromSequences(cleanSeqs);
110
+ rules = result.rules;
111
+ nsInit = result.nsInit;
112
+ }
113
+ catch (e) {
114
+ empty.status = "error";
115
+ empty.error = `Rule discovery failed: ${e}`;
116
+ return empty;
117
+ }
118
+ // Validate all labeled sequences
119
+ let tp = 0, fp = 0, tn = 0, fn = 0;
120
+ const mismatches = [];
121
+ for (const idx of labeledIndices) {
122
+ const expected = labels[idx];
123
+ const calls = sequences[idx] || [];
124
+ let detected;
125
+ try {
126
+ const { validateSequenceWithSSG } = require("./ssg-precision");
127
+ const result = validateSequenceWithSSG(calls, rules, nsInit);
128
+ detected = result.valid ? "clean" : "violation";
129
+ }
130
+ catch {
131
+ detected = "clean"; // Can't validate → assume clean (conservative)
132
+ }
133
+ if (expected === "violation" && detected === "violation")
134
+ tp++;
135
+ else if (expected === "clean" && detected === "violation")
136
+ fp++;
137
+ else if (expected === "clean" && detected === "clean")
138
+ tn++;
139
+ else if (expected === "violation" && detected === "clean")
140
+ fn++;
141
+ if (expected !== detected) {
142
+ mismatches.push({ index: idx, expected, got: detected, calls });
143
+ }
144
+ }
145
+ const total = labeledIndices.length;
146
+ const precision = tp + fp > 0 ? tp / (tp + fp) : 0;
147
+ const recall = tp + fn > 0 ? tp / (tp + fn) : 0;
148
+ const f1 = precision + recall > 0
149
+ ? 2 * precision * recall / (precision + recall)
150
+ : 0;
151
+ const fpr = fp + tn > 0 ? fp / (fp + tn) : 0;
152
+ const fnr = fn + tp > 0 ? fn / (fn + tp) : 0;
153
+ return {
154
+ repo: repoName,
155
+ status: "measured",
156
+ total,
157
+ cleanLabels,
158
+ violationLabels,
159
+ tp, fp, tn, fn,
160
+ precision, recall, f1,
161
+ falsePositiveRate: fpr,
162
+ falseNegativeRate: fnr,
163
+ rulesDiscovered: rules.size,
164
+ mismatches,
165
+ };
166
+ }
167
+ // ═══════════════════════════════════════════════════════════════
168
+ // Report Generator
169
+ // ═══════════════════════════════════════════════════════════════
170
+ function generateReport(repoNames) {
171
+ const repos = repoNames.map(runPrecisionForRepo);
172
+ const measured = repos.filter(r => r.status === "measured");
173
+ let totalTP = 0, totalFP = 0, totalFN = 0, totalSamples = 0;
174
+ for (const r of measured) {
175
+ totalTP += r.tp;
176
+ totalFP += r.fp;
177
+ totalFN += r.fn;
178
+ totalSamples += r.total;
179
+ }
180
+ const macroF1 = measured.length > 0
181
+ ? measured.reduce((s, r) => s + r.f1, 0) / measured.length
182
+ : 0;
183
+ const microPrecision = totalTP + totalFP > 0
184
+ ? totalTP / (totalTP + totalFP)
185
+ : 0;
186
+ const microRecall = totalTP + totalFN > 0
187
+ ? totalTP / (totalTP + totalFN)
188
+ : 0;
189
+ const microF1 = microPrecision + microRecall > 0
190
+ ? 2 * microPrecision * microRecall / (microPrecision + microRecall)
191
+ : 0;
192
+ const sortedByF1 = [...measured].sort((a, b) => b.f1 - a.f1);
193
+ const bestRepo = sortedByF1.length > 0 ? sortedByF1[0].repo : "N/A";
194
+ const worstRepo = sortedByF1.length > 1
195
+ ? sortedByF1[sortedByF1.length - 1].repo
196
+ : "N/A";
197
+ const avgFPR = measured.length > 0
198
+ ? measured.reduce((s, r) => s + r.falsePositiveRate, 0) / measured.length
199
+ : 0;
200
+ const avgFNR = measured.length > 0
201
+ ? measured.reduce((s, r) => s + r.falseNegativeRate, 0) / measured.length
202
+ : 0;
203
+ let assessment = "INSUFFICIENT DATA";
204
+ if (measured.length >= 3) {
205
+ if (microF1 >= 0.80)
206
+ assessment = "PRODUCTION READY";
207
+ else if (microF1 >= 0.65)
208
+ assessment = "BETA QUALITY";
209
+ else if (microF1 >= 0.50)
210
+ assessment = "ALPHA — NEEDS MORE DATA";
211
+ else
212
+ assessment = "EARLY STAGE";
213
+ }
214
+ else if (measured.length >= 1) {
215
+ assessment = "PILOT — EXPAND LABELING";
216
+ }
217
+ return {
218
+ generated: new Date().toISOString(),
219
+ version: "3.2.0",
220
+ repos,
221
+ overall: {
222
+ reposMeasured: measured.length,
223
+ totalSamples,
224
+ totalTP,
225
+ totalFP,
226
+ totalFN,
227
+ macroF1,
228
+ microPrecision,
229
+ microRecall,
230
+ microF1,
231
+ bestRepo,
232
+ worstRepo,
233
+ assessment,
234
+ avgFPRate: avgFPR,
235
+ avgFNRate: avgFNR,
236
+ },
237
+ };
238
+ }
239
+ // ═══════════════════════════════════════════════════════════════
240
+ // Formatting
241
+ // ═══════════════════════════════════════════════════════════════
242
+ function formatReport(report) {
243
+ const lines = [];
244
+ const C = { bold: "", dim: "", green: "", red: "", yellow: "", cyan: "", reset: "" };
245
+ lines.push("");
246
+ lines.push("╔══════════════════════════════════════════════════════════════╗");
247
+ lines.push("║ Progmune Cross-Repository Precision Benchmark ║");
248
+ lines.push("╠══════════════════════════════════════════════════════════════╣");
249
+ lines.push(`║ Generated: ${report.generated} ║`);
250
+ lines.push(`║ Version: v${report.version} ║`);
251
+ lines.push("╚══════════════════════════════════════════════════════════════╝");
252
+ lines.push("");
253
+ // Per-repo table
254
+ const header = "┌─────────────────┬───────┬───────┬───────┬───────┬───────┬───────┬───────┐";
255
+ const sep = "├─────────────────┼───────┼───────┼───────┼───────┼───────┼───────┼───────┤";
256
+ const footer = "└─────────────────┴───────┴───────┴───────┴───────┴───────┴───────┴───────┘";
257
+ lines.push(header);
258
+ lines.push("│ Repo │ P │ R │ F1 │ FP% │ FN% │ N │ Rules │");
259
+ lines.push(sep);
260
+ for (const repo of report.repos) {
261
+ if (repo.status !== "measured") {
262
+ const status = repo.status === "no_labels" ? "no labels" : "error";
263
+ lines.push(`│ ${repo.repo.padEnd(15)} │ ${"-".padStart(3)} │ ${"-".padStart(3)} │ ${"-".padStart(3)} │ ${"-".padStart(3)} │ ${"-".padStart(3)} │ ${String(repo.total || 0).padStart(4)} │ ${"-".padStart(3)} │`);
264
+ continue;
265
+ }
266
+ const p = (repo.precision * 100).toFixed(0);
267
+ const r = (repo.recall * 100).toFixed(0);
268
+ const f = (repo.f1 * 100).toFixed(0);
269
+ const fpr = (repo.falsePositiveRate * 100).toFixed(0);
270
+ const fnr = (repo.falseNegativeRate * 100).toFixed(0);
271
+ // Color-code F1
272
+ let fDisplay = `${f}%`;
273
+ if (repo.f1 >= 0.7)
274
+ fDisplay = `${f}% ★`;
275
+ else if (repo.f1 >= 0.5)
276
+ fDisplay = `${f}%`;
277
+ lines.push(`│ ${repo.repo.padEnd(15)} │ ${p.padStart(3)}% │ ${r.padStart(3)}% │ ${fDisplay.padStart(5)} │ ${fpr.padStart(3)}% │ ${fnr.padStart(3)}% │ ${String(repo.total).padStart(4)} │ ${String(repo.rulesDiscovered).padStart(4)} │`);
278
+ }
279
+ lines.push(sep);
280
+ const o = report.overall;
281
+ const op = (o.microPrecision * 100).toFixed(0);
282
+ const or_ = (o.microRecall * 100).toFixed(0);
283
+ const of1 = (o.microF1 * 100).toFixed(0);
284
+ const maF1 = (o.macroF1 * 100).toFixed(0);
285
+ lines.push(`│ OVERALL (micro) │ ${op.padStart(3)}% │ ${or_.padStart(3)}% │ ${of1.padStart(3)}% │ ${(o.avgFPRate * 100).toFixed(0).padStart(3)}% │ ${(o.avgFNRate * 100).toFixed(0).padStart(3)}% │ ${String(o.totalSamples).padStart(4)} │ │`);
286
+ lines.push(`│ OVERALL (macro) │ │ │ ${maF1.padStart(3)}% │ │ │ │ │`);
287
+ lines.push(footer);
288
+ lines.push("");
289
+ // Summary
290
+ lines.push(`Repos measured: ${o.reposMeasured}/${report.repos.length}`);
291
+ lines.push(`Best repo: ${o.bestRepo}`);
292
+ lines.push(`Worst repo: ${o.worstRepo}`);
293
+ lines.push(`Avg FP Rate: ${(o.avgFPRate * 100).toFixed(1)}%`);
294
+ lines.push(`Avg FN Rate: ${(o.avgFNRate * 100).toFixed(1)}%`);
295
+ lines.push(`Assessment: ${o.assessment}`);
296
+ lines.push("");
297
+ // Mismatches detail
298
+ const reposWithMismatches = report.repos.filter(r => r.mismatches.length > 0);
299
+ if (reposWithMismatches.length > 0) {
300
+ lines.push("── Mismatch Details ──");
301
+ for (const repo of reposWithMismatches) {
302
+ lines.push(` ${repo.repo}: ${repo.mismatches.length} mismatches`);
303
+ for (const m of repo.mismatches.slice(0, 5)) {
304
+ const tag = m.expected === "clean" ? "FP" : "FN";
305
+ lines.push(` [${tag}] #${m.index}: expected ${m.expected}, got ${m.got}`);
306
+ lines.push(` ${m.calls.join(" → ")}`);
307
+ }
308
+ if (repo.mismatches.length > 5) {
309
+ lines.push(` ... and ${repo.mismatches.length - 5} more`);
310
+ }
311
+ }
312
+ lines.push("");
313
+ }
314
+ return lines.join("\n");
315
+ }
316
+ // ═══════════════════════════════════════════════════════════════
317
+ // Save
318
+ // ═══════════════════════════════════════════════════════════════
319
+ function saveReport(report) {
320
+ const reportsDir = path.resolve(process.cwd(), "benchmarks", "reports");
321
+ if (!fs.existsSync(reportsDir))
322
+ fs.mkdirSync(reportsDir, { recursive: true });
323
+ const date = new Date().toISOString().slice(0, 10);
324
+ const filePath = path.join(reportsDir, `cross-repo-precision-${date}.json`);
325
+ fs.writeFileSync(filePath, JSON.stringify(report, null, 2));
326
+ const latestPath = path.join(reportsDir, "cross-repo-precision-latest.json");
327
+ fs.writeFileSync(latestPath, JSON.stringify(report, null, 2));
328
+ console.log(`Reports saved:`);
329
+ console.log(` ${filePath}`);
330
+ console.log(` ${latestPath}`);
331
+ }
332
+ // ═══════════════════════════════════════════════════════════════
333
+ // Main
334
+ // ═══════════════════════════════════════════════════════════════
335
+ function main() {
336
+ const args = process.argv.slice(2);
337
+ const repoArg = args.find(a => a.startsWith("--repos="));
338
+ const repoNames = repoArg
339
+ ? repoArg.replace("--repos=", "").split(",")
340
+ : [
341
+ "curl",
342
+ "libssh",
343
+ "nginx",
344
+ "redis",
345
+ // Future: add "nghttp2", "apache", "openssl" when labels are available
346
+ ];
347
+ console.log(`Running precision measurement on ${repoNames.length} repos...`);
348
+ const report = generateReport(repoNames);
349
+ console.log(formatReport(report));
350
+ saveReport(report);
351
+ }
352
+ main();