progmune-runtime 2.1.6 → 3.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (356) hide show
  1. package/README.md +108 -468
  2. package/dist/ablation-study.js +144 -0
  3. package/dist/ablation-study.test.js +18 -0
  4. package/dist/action-runtime.js +3 -1
  5. package/dist/active-learning.js +211 -0
  6. package/dist/analytics.js +139 -0
  7. package/dist/asset-factory.js +309 -0
  8. package/dist/asset-growth.js +244 -0
  9. package/dist/asset-promotion.js +382 -0
  10. package/dist/asset-quality.js +550 -0
  11. package/dist/audit/business-translator.js +285 -0
  12. package/dist/audit/cli.js +66 -0
  13. package/dist/audit/formatters/html.js +379 -0
  14. package/dist/audit/formatters/json.js +11 -0
  15. package/dist/audit/formatters/markdown.js +192 -0
  16. package/dist/audit/formatters/terminal.js +189 -0
  17. package/dist/audit/index.js +25 -0
  18. package/dist/audit/report-builder.js +318 -0
  19. package/dist/audit/types.js +8 -0
  20. package/dist/audit.js +3 -3
  21. package/dist/auto-benchmark-generator.js +137 -0
  22. package/dist/auto-benchmark-generator.test.js +45 -0
  23. package/dist/auto-protocol-synthesizer.js +362 -0
  24. package/dist/auto-protocol-synthesizer.test.js +82 -0
  25. package/dist/autonomous-patch.js +175 -0
  26. package/dist/autonomous-patch.test.js +128 -0
  27. package/dist/badge/badge-server.js +98 -0
  28. package/dist/behavior-miner.js +442 -0
  29. package/dist/belief-layer.js +475 -0
  30. package/dist/benchmark-count.js +5 -0
  31. package/dist/benchmark-generator.js +211 -0
  32. package/dist/benchmark-harness.js +201 -0
  33. package/dist/benchmark-pass-rate.js +7 -0
  34. package/dist/benchmark-report.js +8 -3
  35. package/dist/benchmark-save.js +14 -1
  36. package/dist/bootstrap-validation.js +197 -0
  37. package/dist/bootstrap-validation.test.js +51 -0
  38. package/dist/branch-ledger.js +1 -1
  39. package/dist/capability-gap.js +130 -0
  40. package/dist/certify-html.js +351 -0
  41. package/dist/certify.js +326 -0
  42. package/dist/check.js +4 -4
  43. package/dist/compliance-miner.js +447 -0
  44. package/dist/continuous-benchmark.js +194 -0
  45. package/dist/continuous-benchmark.test.js +116 -0
  46. package/dist/corpus-stats.js +173 -0
  47. package/dist/counterfactual-engine.js +288 -0
  48. package/dist/coverage-dashboard.js +109 -0
  49. package/dist/coverage-system.test.js +205 -0
  50. package/dist/cross-repo-precision.js +352 -0
  51. package/dist/cve-benchmark.js +180 -0
  52. package/dist/cve-benchmark.test.js +28 -0
  53. package/dist/cve-collector.js +73 -0
  54. package/dist/data-quality.js +141 -0
  55. package/dist/decision-engine.js +388 -0
  56. package/dist/derive-metadata.js +250 -0
  57. package/dist/difficulty-active.test.js +198 -0
  58. package/dist/difficulty-map.js +244 -0
  59. package/dist/discovery-analytics.js +125 -0
  60. package/dist/discovery-model.js +149 -0
  61. package/dist/discovery-optimize.test.js +199 -0
  62. package/dist/discovery-trace.js +276 -0
  63. package/dist/discovery-trace.test.js +97 -0
  64. package/dist/emitter.js +83 -1
  65. package/dist/enterprise-dashboard.js +405 -0
  66. package/dist/eval-hardening.js +297 -0
  67. package/dist/eval-hardening.test.js +85 -0
  68. package/dist/evaluation-campaign.js +359 -0
  69. package/dist/evaluation-campaign.test.js +181 -0
  70. package/dist/evidence-growth.js +143 -0
  71. package/dist/evidence-repository.js +209 -0
  72. package/dist/evidence-system.js +441 -0
  73. package/dist/execute.js +15 -7
  74. package/dist/experimental/software-physics.js +291 -0
  75. package/dist/experimental/state-inference.js +516 -0
  76. package/dist/experimental/unsupervised-physics.js +230 -0
  77. package/dist/extract-ir-python.js +54 -7
  78. package/dist/extract-ir.js +376 -12
  79. package/dist/failure-collector.js +2 -2
  80. package/dist/failure-corpus.js +322 -9
  81. package/dist/feedback.js +16 -5
  82. package/dist/feedback.test.js +49 -0
  83. package/dist/file-lock.js +1 -1
  84. package/dist/flywheel-batch.js +292 -0
  85. package/dist/frameworks/express-cli.js +237 -0
  86. package/dist/frameworks/express-detector.js +445 -0
  87. package/dist/frameworks/express-detector.test.js +206 -0
  88. package/dist/frameworks/index.js +30 -0
  89. package/dist/frameworks/nestjs-detector.js +302 -0
  90. package/dist/frameworks/trpc-detector.js +161 -0
  91. package/dist/frameworks/version-awareness.js +179 -0
  92. package/dist/function-synonyms.js +164 -0
  93. package/dist/function-synonyms.test.js +68 -0
  94. package/dist/generalization.test.js +352 -0
  95. package/dist/goal-annotator.js +113 -0
  96. package/dist/goal-planner.js +563 -0
  97. package/dist/gold-cve.js +164 -0
  98. package/dist/gold-cve.test.js +104 -0
  99. package/dist/gold-quality.js +206 -0
  100. package/dist/gold-tiers.js +241 -0
  101. package/dist/governance-dashboard.js +327 -0
  102. package/dist/graph-viz.js +240 -0
  103. package/dist/guided-frontier.js +195 -0
  104. package/dist/hierarchical-planner.js +148 -0
  105. package/dist/identifier-parser.js +260 -0
  106. package/dist/immune-metrics.js +93 -0
  107. package/dist/immune-receiver.js +158 -0
  108. package/dist/immune-reporter.js +1 -1
  109. package/dist/improvement-orchestrator.js +206 -0
  110. package/dist/inject-p0-vocabulary.js +300 -0
  111. package/dist/intent-parser.js +218 -0
  112. package/dist/invariant-algebra.js +476 -0
  113. package/dist/invariant-calculus.js +533 -0
  114. package/dist/ir-utils.js +70 -0
  115. package/dist/ir-utils.test.js +50 -0
  116. package/dist/knowledge-api.js +312 -0
  117. package/dist/knowledge-evolution.js +452 -0
  118. package/dist/knowledge-explorer.js +506 -0
  119. package/dist/knowledge-flywheel.js +274 -0
  120. package/dist/knowledge-governance.js +338 -0
  121. package/dist/knowledge-governance.test.js +150 -0
  122. package/dist/knowledge-graph.js +181 -0
  123. package/dist/knowledge-guided-synth.js +246 -0
  124. package/dist/knowledge-loop.test.js +77 -0
  125. package/dist/knowledge-object.js +316 -0
  126. package/dist/knowledge-package.js +98 -0
  127. package/dist/kpi-dashboard.js +561 -0
  128. package/dist/l3-cross-function.js +280 -0
  129. package/dist/learning-ranker.js +148 -0
  130. package/dist/learning-ranker.test.js +291 -0
  131. package/dist/ledger/accountability.js +322 -0
  132. package/dist/ledger/chain-builder.js +185 -0
  133. package/dist/ledger/cli.js +222 -0
  134. package/dist/ledger/index.js +13 -0
  135. package/dist/ledger/signatures.js +193 -0
  136. package/dist/ledger/types.js +9 -0
  137. package/dist/llm.js +74 -3
  138. package/dist/load-benchmarks.js +8 -3
  139. package/dist/logger.js +66 -0
  140. package/dist/logger.test.js +37 -0
  141. package/dist/logistic-reward.js +339 -0
  142. package/dist/logistic-reward.test.js +180 -0
  143. package/dist/macro-graph.js +193 -0
  144. package/dist/macro-repair.js +183 -0
  145. package/dist/mcp-server.mjs +1202 -483
  146. package/dist/memory-layer.js +42 -5
  147. package/dist/multi-repo-precision.js +422 -0
  148. package/dist/name-free-protocol.js +425 -0
  149. package/dist/name-free-protocol.test.js +170 -0
  150. package/dist/name-scrambling.js +138 -0
  151. package/dist/name-scrambling.test.js +16 -0
  152. package/dist/p3-observability.test.js +281 -0
  153. package/dist/p5-orchestrator.test.js +225 -0
  154. package/dist/pairwise-preference.js +294 -0
  155. package/dist/pairwise-preference.test.js +140 -0
  156. package/dist/planner-constraints.js +104 -0
  157. package/dist/planner-prompts.js +155 -0
  158. package/dist/planner-telemetry.js +415 -0
  159. package/dist/planner-trace.js +214 -0
  160. package/dist/planner.js +162 -167
  161. package/dist/plsb/artifact.js +116 -0
  162. package/dist/plsb/cli.js +71 -0
  163. package/dist/plsb/index.js +19 -0
  164. package/dist/plsb/leaderboard.js +249 -0
  165. package/dist/plsb/report-md.js +156 -0
  166. package/dist/plsb/schema.js +179 -0
  167. package/dist/plsb-benchmark.js +284 -0
  168. package/dist/plsb-benchmark.test.js +119 -0
  169. package/dist/policy/cli.js +134 -0
  170. package/dist/policy/engine.js +333 -0
  171. package/dist/policy/index.js +12 -0
  172. package/dist/policy/types.js +59 -0
  173. package/dist/policy-miner.js +505 -0
  174. package/dist/precision-analyze.js +229 -0
  175. package/dist/precision-benchmark.js +147 -0
  176. package/dist/precision-label-c.js +134 -0
  177. package/dist/precision-label.js +193 -0
  178. package/dist/precision-report-c.js +149 -0
  179. package/dist/precision-report.js +246 -0
  180. package/dist/progmune-status.js +108 -0
  181. package/dist/proof-engine.js +479 -0
  182. package/dist/proof-provenance.js +315 -0
  183. package/dist/protocol-coverage.js +294 -0
  184. package/dist/protocol-detector.js +1189 -0
  185. package/dist/protocol-embedding-expanded.js +297 -0
  186. package/dist/protocol-embedding-expanded.test.js +97 -0
  187. package/dist/protocol-embedding.js +195 -0
  188. package/dist/protocol-embedding.test.js +82 -0
  189. package/dist/protocol-extractor-v2.js +354 -0
  190. package/dist/protocol-extractor-v2.test.js +140 -0
  191. package/dist/protocol-extractor.js +310 -0
  192. package/dist/protocol-extractor.test.js +113 -0
  193. package/dist/protocol-foundation.js +322 -0
  194. package/dist/protocol-foundation.test.js +163 -0
  195. package/dist/protocol-frontier.js +243 -0
  196. package/dist/protocol-frontier.test.js +92 -0
  197. package/dist/protocol-gap-analyzer.js +228 -0
  198. package/dist/protocol-gap-analyzer.test.js +49 -0
  199. package/dist/protocol-invariants.js +276 -0
  200. package/dist/protocol-invariants.test.js +111 -0
  201. package/dist/protocol-knowledge.js +464 -0
  202. package/dist/protocol-miner.js +343 -0
  203. package/dist/protocol-mining.js +207 -0
  204. package/dist/protocol-mining.test.js +37 -0
  205. package/dist/protocol-registry.js +1 -1
  206. package/dist/protocol-security-benchmark.js +222 -0
  207. package/dist/protocol-vulnerability.js +257 -0
  208. package/dist/protocol-vulnerability.test.js +60 -0
  209. package/dist/python-benchmark.js +120 -0
  210. package/dist/python-emitter.js +163 -45
  211. package/dist/python-protocol-extractor.js +187 -0
  212. package/dist/python-protocol-extractor.test.js +116 -0
  213. package/dist/realworld-benchmark.js +646 -0
  214. package/dist/realworld-benchmark.test.js +36 -0
  215. package/dist/repair-arch.test.js +411 -0
  216. package/dist/repair-evolution.test.js +454 -0
  217. package/dist/repair-executor.js +719 -0
  218. package/dist/repair-proposal.js +4 -4
  219. package/dist/repair-ranker.js +141 -0
  220. package/dist/repair-strategies.js +419 -0
  221. package/dist/repair-taxonomy.js +234 -0
  222. package/dist/repair-types.js +12 -0
  223. package/dist/repo-evaluator.js +250 -0
  224. package/dist/repo-evaluator.test.js +128 -0
  225. package/dist/resource-abstraction.js +242 -0
  226. package/dist/resource-detector.js +211 -0
  227. package/dist/result.test.js +43 -0
  228. package/dist/reward-system.js +411 -0
  229. package/dist/reward-system.test.js +175 -0
  230. package/dist/risk-model.js +215 -0
  231. package/dist/rule-miner.js +234 -7
  232. package/dist/rule-specificity.js +254 -0
  233. package/dist/runtime-types.js +27 -0
  234. package/dist/scaffold.js +208 -0
  235. package/dist/scale-collector.test.js +101 -0
  236. package/dist/scale-trajectory-collector.js +128 -0
  237. package/dist/sdk.js +250 -0
  238. package/dist/search-planner.js +4 -41
  239. package/dist/semantic-snapshot.js +1 -1
  240. package/dist/semantic-topology.js +121 -0
  241. package/dist/semantic-trace.js +310 -317
  242. package/dist/sequence-extractor.js +343 -0
  243. package/dist/skill-library.js +245 -0
  244. package/dist/skill-planner.test.js +189 -0
  245. package/dist/software-physics.js +291 -0
  246. package/dist/software-physics.test.js +81 -0
  247. package/dist/ssg-precision.js +478 -0
  248. package/dist/ssg-validator.js +71 -21
  249. package/dist/state-inference-doubleblind.test.js +160 -0
  250. package/dist/state-inference.js +516 -0
  251. package/dist/state-inference.test.js +115 -0
  252. package/dist/state-machine-fingerprint.js +345 -0
  253. package/dist/state-machine-fingerprint.test.js +120 -0
  254. package/dist/state-miner.js +386 -0
  255. package/dist/state-name-inference.js +213 -0
  256. package/dist/state-name-inference.test.js +69 -0
  257. package/dist/strategy-planner.js +262 -96
  258. package/dist/strategy-planner.test.js +135 -0
  259. package/dist/telemetry-analytics.test.js +402 -0
  260. package/dist/terminal-format.js +68 -0
  261. package/dist/terminal-format.test.js +83 -0
  262. package/dist/topology-factory.js +196 -0
  263. package/dist/topology-representation.js +242 -0
  264. package/dist/topology-representation.test.js +27 -0
  265. package/dist/trajectory-augmentation.js +254 -0
  266. package/dist/trajectory-augmentation.test.js +63 -0
  267. package/dist/trajectory-corpus.js +440 -0
  268. package/dist/trajectory-corpus.test.js +32 -0
  269. package/dist/trajectory-feedback.test.js +116 -0
  270. package/dist/transition-synthesizer.js +286 -0
  271. package/dist/transition-synthesizer.test.js +123 -0
  272. package/dist/trust/api-semantic-mapper.js +809 -0
  273. package/dist/trust/call-graph-propagator.js +225 -0
  274. package/dist/trust/cli.js +122 -0
  275. package/dist/trust/compliance-scorer.js +283 -0
  276. package/dist/trust/confidence-calculator.js +261 -0
  277. package/dist/trust/engine.js +1145 -0
  278. package/dist/trust/explainability.js +85 -0
  279. package/dist/trust/formatters/ci.js +42 -0
  280. package/dist/trust/formatters/json.js +11 -0
  281. package/dist/trust/formatters/terminal.js +152 -0
  282. package/dist/trust/index.js +39 -0
  283. package/dist/trust/phase1-verify.js +171 -0
  284. package/dist/trust/protocol-domain-validator.js +697 -0
  285. package/dist/trust/score-calculator.js +282 -0
  286. package/dist/trust/ssg-bridge.js +641 -0
  287. package/dist/trust/ssg-bridge.test.js +269 -0
  288. package/dist/trust/types.js +67 -0
  289. package/dist/trust/violation-trace.js +335 -0
  290. package/dist/trust-api.js +179 -0
  291. package/dist/trust-calibration.js +279 -0
  292. package/dist/unknown-protocol-discovery.js +339 -0
  293. package/dist/unknown-protocol-discovery.test.js +102 -0
  294. package/dist/unsupervised-physics.js +230 -0
  295. package/dist/unsupervised-physics.test.js +95 -0
  296. package/dist/utils.test.js +37 -0
  297. package/dist/validator.js +187 -10
  298. package/dist/verification-intelligence.js +475 -0
  299. package/dist/verify-api.js +432 -0
  300. package/dist/vi-impact-report.js +293 -0
  301. package/dist/wl-fingerprint.js +162 -0
  302. package/dist/wl-fingerprint.test.js +130 -0
  303. package/dist/zeroshot-strategy.js +139 -0
  304. package/docs/Progmune_/346/212/225/350/265/204/344/272/272/347/231/275/347/232/256/344/271/246_v2.0.html +576 -0
  305. package/docs/Progmune_/351/241/271/347/233/256/345/205/250/350/247/243.html +710 -0
  306. package/package.json +74 -7
  307. package/protocols.json +1956 -50
  308. package/.dockerignore +0 -14
  309. package/.mcp.json +0 -11
  310. package/.progmune_allowlist +0 -50
  311. package/.test_report/test_report.md +0 -87
  312. package/Dockerfile +0 -9
  313. package/FAQ.md +0 -167
  314. package/WHITEPAPER.md +0 -540
  315. package/demo-project/auth.ts +0 -55
  316. package/demo-project/tsconfig.json +0 -8
  317. package/dist/acl-breakdown.js +0 -13
  318. package/dist/all-sessions.js +0 -11
  319. package/dist/antibody-stats.js +0 -11
  320. package/dist/branch-tree-count.js +0 -14
  321. package/dist/common-fixpath.js +0 -12
  322. package/dist/constraint-types.js +0 -12
  323. package/dist/exec-metrics.js +0 -11
  324. package/dist/failure-report.js +0 -11
  325. package/dist/fast-path-hits.js +0 -13
  326. package/dist/fingerprint-list.js +0 -15
  327. package/dist/gen-history-log.js +0 -13
  328. package/dist/heatmap-data.js +0 -11
  329. package/dist/recent-session.js +0 -12
  330. package/dist/svl-distribution.js +0 -11
  331. package/dist/terminal-status.js +0 -11
  332. package/dist/token-savings.js +0 -11
  333. package/dist/total-repairs.js +0 -12
  334. package/dist/unresolved-count.js +0 -12
  335. package/dist/valid-fingerprints.js +0 -13
  336. package/dist/verify-ledgers.js +0 -11
  337. package/docs/whitepaper-style.css +0 -77
  338. package/docs/whitepaper-v2.1.md +0 -609
  339. package/docs/whitepaper-v2.2.md +0 -1064
  340. package/docs/whitepaper-v2.2.pdf +0 -0
  341. package/fly.toml +0 -31
  342. package/public/dashboard.html +0 -119
  343. package/server/hub.js +0 -116
  344. package/test/replay-golden/sess_1780063202050_mgeld.json +0 -9
  345. package/test/replay-golden/sess_1780064032560_gocld.json +0 -354
  346. package/test/replay-golden/sess_1780064413331_s2709.json +0 -606
  347. package/test/replay-golden/sess_1780064792710_y3avo.json +0 -614
  348. package/test/replay-golden.ts +0 -84
  349. package/test_benchmark.js +0 -165
  350. package/test_comprehensive.mjs +0 -638
  351. package/test_concurrency.js +0 -129
  352. package/test_ir_robustness.js +0 -85
  353. package/test_semantic_contracts.js +0 -269
  354. package/test_ssg_stress.js +0 -156
  355. package/test_svl3.js +0 -58
  356. package/tsconfig.json +0 -17
@@ -0,0 +1,359 @@
1
+ "use strict";
2
+ /**
3
+ * P3.9: Evaluation Campaign
4
+ *
5
+ * Shifts from "building modules" to "validating hypotheses".
6
+ *
7
+ * Three tools:
8
+ * 1. Failure Attribution — classify WHY each benchmark case fails
9
+ * 2. Error Budget Dashboard — aggregate failure reasons
10
+ * 3. Offline Replay Engine — replay history with new rankers, compute accuracy
11
+ *
12
+ * Key metric: Replay Accuracy — how often would the ranker have matched
13
+ * what the user actually chose? If LearningRanker > LinearRanker by 10%+,
14
+ * the Telemetry→Feedback→Learning loop is proven effective.
15
+ */
16
+ var __createBinding = (this && this.__createBinding) || (Object.create ? (function(o, m, k, k2) {
17
+ if (k2 === undefined) k2 = k;
18
+ var desc = Object.getOwnPropertyDescriptor(m, k);
19
+ if (!desc || ("get" in desc ? !m.__esModule : desc.writable || desc.configurable)) {
20
+ desc = { enumerable: true, get: function() { return m[k]; } };
21
+ }
22
+ Object.defineProperty(o, k2, desc);
23
+ }) : (function(o, m, k, k2) {
24
+ if (k2 === undefined) k2 = k;
25
+ o[k2] = m[k];
26
+ }));
27
+ var __setModuleDefault = (this && this.__setModuleDefault) || (Object.create ? (function(o, v) {
28
+ Object.defineProperty(o, "default", { enumerable: true, value: v });
29
+ }) : function(o, v) {
30
+ o["default"] = v;
31
+ });
32
+ var __importStar = (this && this.__importStar) || (function () {
33
+ var ownKeys = function(o) {
34
+ ownKeys = Object.getOwnPropertyNames || function (o) {
35
+ var ar = [];
36
+ for (var k in o) if (Object.prototype.hasOwnProperty.call(o, k)) ar[ar.length] = k;
37
+ return ar;
38
+ };
39
+ return ownKeys(o);
40
+ };
41
+ return function (mod) {
42
+ if (mod && mod.__esModule) return mod;
43
+ var result = {};
44
+ if (mod != null) for (var k = ownKeys(mod), i = 0; i < k.length; i++) if (k[i] !== "default") __createBinding(result, mod, k[i]);
45
+ __setModuleDefault(result, mod);
46
+ return result;
47
+ };
48
+ })();
49
+ Object.defineProperty(exports, "__esModule", { value: true });
50
+ exports.runFailureAttribution = runFailureAttribution;
51
+ exports.computeErrorBudget = computeErrorBudget;
52
+ exports.printErrorBudget = printErrorBudget;
53
+ exports.replayDecisions = replayDecisions;
54
+ exports.compareRankers = compareRankers;
55
+ exports.printReplayReport = printReplayReport;
56
+ exports.printRankerComparison = printRankerComparison;
57
+ const fs = __importStar(require("fs"));
58
+ const path = __importStar(require("path"));
59
+ const counterfactual_engine_1 = require("./counterfactual-engine");
60
+ const ssg_validator_1 = require("./ssg-validator");
61
+ function expectedSignature(expected) {
62
+ return [...expected].sort().join("→");
63
+ }
64
+ function resultSignature(fixPath) {
65
+ return [...fixPath].sort().join("→");
66
+ }
67
+ async function classifyFailure(tc, result, alts) {
68
+ if (result.top1Hit)
69
+ return "success";
70
+ const expSig = expectedSignature(tc.expected);
71
+ // No candidates at all → protocol model can't find the path
72
+ if (alts.length === 0) {
73
+ // Check if it's a resource leak (ProtocolStrategy handles these)
74
+ if (tc.violationType === "resource_leak")
75
+ return "bad_protocol_model";
76
+ return "bad_protocol_model";
77
+ }
78
+ // Check if ANY candidate has the expected repair (even if ranked wrong)
79
+ const anyMatch = alts.some(a => resultSignature(a.fixPath) === expSig);
80
+ if (anyMatch)
81
+ return "bad_ranking";
82
+ // Check if it's a corpus-dependent scenario
83
+ if (tc.violationType === "missing_prerequisite" && alts.length <= 2) {
84
+ return "insufficient_history";
85
+ }
86
+ // If expected includes functions not in protocol rules, goal mismatch
87
+ const protoDef = JSON.parse(fs.readFileSync(path.resolve(__dirname, "..", "protocols.json"), "utf-8"));
88
+ const allRules = new Set(Object.keys(protoDef.rules));
89
+ const unknownFns = tc.expected.filter(fn => !allRules.has(fn));
90
+ if (unknownFns.length > 0)
91
+ return "goal_mismatch";
92
+ // Otherwise: candidate simply wasn't found
93
+ return "missing_candidate";
94
+ }
95
+ async function runFailureAttribution(suitePath) {
96
+ const benchmarksDir = suitePath || path.resolve(__dirname, "..", "benchmarks");
97
+ const files = fs.readdirSync(benchmarksDir).filter(f => f.endsWith(".json") && !f.includes("generated") && !f.includes("priority"));
98
+ const protoDef = JSON.parse(fs.readFileSync(path.resolve(__dirname, "..", "protocols.json"), "utf-8"));
99
+ const protocols = (0, ssg_validator_1.parseProtocolsFromJSON)(protoDef);
100
+ const rules = new Map();
101
+ for (const p of protocols)
102
+ rules.set(p.function, p.protocol);
103
+ const attributed = [];
104
+ for (const file of files) {
105
+ const cases = JSON.parse(fs.readFileSync(path.join(benchmarksDir, file), "utf-8"));
106
+ for (const tc of cases) {
107
+ const currentStates = new Set();
108
+ for (const fn of tc.broken) {
109
+ const rule = rules.get(fn);
110
+ if (rule) {
111
+ for (const post of rule.post_states)
112
+ currentStates.add(post);
113
+ if (rule.invalidate)
114
+ rule.invalidate.forEach(s => currentStates.delete(s));
115
+ }
116
+ }
117
+ const alts = await (0, counterfactual_engine_1.suggestAlternatives)({
118
+ violation: {
119
+ svl: 4,
120
+ violatedConstraint: tc.violationType,
121
+ actionIndex: tc.broken.length,
122
+ currentStates: [...currentStates],
123
+ requiredStates: [],
124
+ description: `Benchmark: ${tc.goal}`,
125
+ },
126
+ protocol: tc.protocol,
127
+ currentState: [...currentStates],
128
+ targetState: [],
129
+ constraints: [],
130
+ rules,
131
+ goal: tc.goal,
132
+ });
133
+ const expSig = expectedSignature(tc.expected);
134
+ let top1Hit = false;
135
+ let rank = null;
136
+ for (let i = 0; i < alts.length; i++) {
137
+ const fullSig = expectedSignature([...tc.broken, ...alts[i].fixPath]);
138
+ if (fullSig === expSig) {
139
+ if (rank === null)
140
+ rank = i + 1;
141
+ if (i === 0)
142
+ top1Hit = true;
143
+ }
144
+ }
145
+ const caseResult = {
146
+ goal: tc.goal, top1Hit, top3Hit: rank !== null && rank <= 3,
147
+ rank, latencyMs: 0, candidatesReturned: alts.length,
148
+ };
149
+ const reason = await classifyFailure(tc, caseResult, alts);
150
+ attributed.push({
151
+ caseId: `${file}:${tc.goal}`,
152
+ goal: tc.goal,
153
+ protocol: tc.protocol,
154
+ violationType: tc.violationType,
155
+ expectedRepair: tc.expected,
156
+ plannerTop1: alts.length > 0 ? alts[0].fixPath : undefined,
157
+ plannerTop3: alts.slice(0, 3).map(a => a.fixPath),
158
+ candidatesReturned: alts.length,
159
+ rank,
160
+ failureReason: reason,
161
+ });
162
+ }
163
+ }
164
+ return attributed;
165
+ }
166
+ function computeErrorBudget(attributed) {
167
+ const total = attributed.length;
168
+ const successes = attributed.filter(a => a.failureReason === "success").length;
169
+ const breakdown = {};
170
+ for (const a of attributed) {
171
+ breakdown[a.failureReason] = (breakdown[a.failureReason] || 0) + 1;
172
+ }
173
+ const percentages = {};
174
+ for (const [k, v] of Object.entries(breakdown)) {
175
+ percentages[k] = total > 0 ? v / total : 0;
176
+ }
177
+ // Generate recommendation
178
+ const missingPct = percentages["missing_candidate"] || 0;
179
+ const rankingPct = percentages["bad_ranking"] || 0;
180
+ const protocolPct = percentages["bad_protocol_model"] || 0;
181
+ let recommendation;
182
+ if (missingPct > 0.3) {
183
+ recommendation = "P0: Fix candidate discovery (ProtocolStrategy BFS). Don't touch Reward Model until candidates are found.";
184
+ }
185
+ else if (rankingPct > 0.3) {
186
+ recommendation = "P0: Improve ranking (LearningRanker, more feedback data). Ranking is the bottleneck.";
187
+ }
188
+ else if (protocolPct > 0.3) {
189
+ recommendation = "P0: Expand protocol rules. Protocol model doesn't cover enough transitions.";
190
+ }
191
+ else {
192
+ recommendation = "Balanced error profile. Proceed with small improvements across all dimensions.";
193
+ }
194
+ return {
195
+ totalCases: total,
196
+ successes,
197
+ successRate: total > 0 ? successes / total : 0,
198
+ breakdown: breakdown,
199
+ percentages: percentages,
200
+ recommendation,
201
+ };
202
+ }
203
+ function printErrorBudget(budget) {
204
+ console.log("\n╔════════════════════════════════════════════════════╗");
205
+ console.log("║ Error Budget Dashboard ║");
206
+ console.log("╚════════════════════════════════════════════════════╝\n");
207
+ console.log(`Total Cases: ${budget.totalCases}`);
208
+ console.log(`Successes: ${budget.successes} (${(budget.successRate * 100).toFixed(0)}%)`);
209
+ console.log();
210
+ console.log("─── Failure Breakdown ───");
211
+ console.log("Reason Count Pct Bar");
212
+ console.log("──────────────────────────────────────────────");
213
+ const order = ["missing_candidate", "bad_ranking", "bad_protocol_model", "goal_mismatch", "insufficient_history", "success"];
214
+ for (const reason of order) {
215
+ const count = budget.breakdown[reason] || 0;
216
+ const pct = budget.percentages[reason] || 0;
217
+ const bar = "█".repeat(Math.round(pct * 40));
218
+ const label = reason.padEnd(22);
219
+ const pctStr = (pct * 100).toFixed(0).padStart(3) + "%";
220
+ console.log(` ${label} ${String(count).padStart(4)} ${pctStr} ${bar}`);
221
+ }
222
+ console.log();
223
+ console.log(`─── Recommendation ───`);
224
+ console.log(` ${budget.recommendation}`);
225
+ console.log();
226
+ }
227
+ /**
228
+ * Replay historical decisions with a new candidate ranking.
229
+ *
230
+ * Given PlannerTrace data (what was shown to the user and what they chose)
231
+ * and a LearningRanker (which re-scores candidates using feedback data),
232
+ * compute how often the new ranker's top-1 matches the user's choice.
233
+ */
234
+ function replayDecisions(traceStore, telemetry) {
235
+ const traces = traceStore.all();
236
+ const withChoice = traces.filter((t) => t.selectedFingerprint && t.candidates.length > 0);
237
+ const results = [];
238
+ for (const trace of withChoice) {
239
+ const userChose = trace.selectedFingerprint;
240
+ // Build RepairCandidate-like objects from snapshots
241
+ const candidates = trace.candidates.map((c) => ({
242
+ fingerprint: c.fingerprint,
243
+ oldRank: c.rank,
244
+ oldScore: c.score,
245
+ source: c.source,
246
+ actions: c.actions,
247
+ evidenceSources: c.evidenceSources,
248
+ }));
249
+ // Re-rank using telemetry acceptance data
250
+ // Higher acceptance = better rank
251
+ const reranked = candidates.map((c) => ({
252
+ ...c,
253
+ acceptance: telemetry.getCandidateAcceptance(c.fingerprint, 1),
254
+ })).sort((a, b) => {
255
+ // Sort by acceptance descending, then by old score
256
+ if (a.acceptance !== b.acceptance)
257
+ return b.acceptance - a.acceptance;
258
+ return b.oldScore - a.oldScore;
259
+ });
260
+ const newRankerChose = reranked[0]?.fingerprint ?? null;
261
+ const matched = newRankerChose === userChose;
262
+ // Find user's choice in new ranking
263
+ const userIdx = reranked.findIndex((c) => c.fingerprint === userChose);
264
+ const userChoiceRank = userIdx >= 0 ? userIdx + 1 : null;
265
+ results.push({
266
+ traceId: trace.traceId,
267
+ goal: trace.goal,
268
+ protocol: trace.protocol,
269
+ userChose,
270
+ userChoseRank: userChoiceRank,
271
+ newRankerChose,
272
+ matched,
273
+ candidates: reranked.map((c, i) => ({
274
+ fingerprint: c.fingerprint,
275
+ oldRank: c.oldRank,
276
+ newRank: i + 1,
277
+ score: c.acceptance,
278
+ })),
279
+ });
280
+ }
281
+ const matches = results.filter(r => r.matched).length;
282
+ const avgRank = results.reduce((s, r) => s + (r.userChoseRank ?? results.length), 0) / Math.max(1, results.length);
283
+ return {
284
+ ranker: "LearningRanker (acceptance-based)",
285
+ totalDecisions: results.length,
286
+ matches,
287
+ matchRate: results.length > 0 ? matches / results.length : 0,
288
+ avgUserChoiceRank: avgRank,
289
+ results,
290
+ };
291
+ }
292
+ /**
293
+ * Compare two ranking strategies by replay accuracy.
294
+ */
295
+ function compareRankers(traceStore, telemetry) {
296
+ const traces = traceStore.all();
297
+ const withChoice = traces.filter((t) => t.selectedFingerprint && t.candidates.length > 0);
298
+ if (withChoice.length === 0) {
299
+ const empty = { ranker: "", totalDecisions: 0, matches: 0, matchRate: 0, avgUserChoiceRank: 0, results: [] };
300
+ return { baseline: empty, learning: empty, delta: 0 };
301
+ }
302
+ // Baseline: original ranker (rank-1 = what planner showed first)
303
+ let baselineMatches = 0;
304
+ for (const trace of withChoice) {
305
+ const top1 = trace.candidates[0]?.fingerprint;
306
+ if (top1 === trace.selectedFingerprint)
307
+ baselineMatches++;
308
+ }
309
+ const baseline = {
310
+ ranker: "LinearRanker (original)",
311
+ totalDecisions: withChoice.length,
312
+ matches: baselineMatches,
313
+ matchRate: baselineMatches / withChoice.length,
314
+ avgUserChoiceRank: 0, // N/A for baseline
315
+ results: [],
316
+ };
317
+ // Learning: acceptance-based reranking
318
+ const learning = replayDecisions(traceStore, telemetry);
319
+ return {
320
+ baseline,
321
+ learning,
322
+ delta: learning.matchRate - baseline.matchRate,
323
+ };
324
+ }
325
+ function printReplayReport(report) {
326
+ console.log("\n╔════════════════════════════════════════════════════╗");
327
+ console.log("║ Offline Replay Report ║");
328
+ console.log("╚════════════════════════════════════════════════════╝\n");
329
+ console.log(`Ranker: ${report.ranker}`);
330
+ console.log(`Decisions Replayed: ${report.totalDecisions}`);
331
+ console.log(`User Choice Matched: ${report.matches}/${report.totalDecisions}`);
332
+ console.log(`Replay Accuracy: ${(report.matchRate * 100).toFixed(1)}%`);
333
+ console.log(`Avg User Choice Rank: ${report.avgUserChoiceRank.toFixed(1)}`);
334
+ console.log();
335
+ }
336
+ function printRankerComparison(baseline, learning, delta) {
337
+ console.log("\n╔════════════════════════════════════════════════════╗");
338
+ console.log("║ Ranker A/B Comparison ║");
339
+ console.log("╚════════════════════════════════════════════════════╝\n");
340
+ const basePct = (baseline.matchRate * 100).toFixed(1);
341
+ const learnPct = (learning.matchRate * 100).toFixed(1);
342
+ const deltaPct = (delta * 100).toFixed(1);
343
+ const sign = delta > 0 ? "+" : "";
344
+ console.log(` Baseline (LinearRanker): ${basePct}% (${baseline.matches}/${baseline.totalDecisions})`);
345
+ console.log(` Learning (Acceptance): ${learnPct}% (${learning.matches}/${learning.totalDecisions})`);
346
+ console.log(` Δ: ${sign}${deltaPct}%`);
347
+ console.log();
348
+ if (delta > 0.05) {
349
+ console.log(` ✅ LearningRanker outperforms baseline by ${sign}${deltaPct}%`);
350
+ console.log(" The Telemetry→Feedback→Learning loop is effective.");
351
+ }
352
+ else if (delta > 0) {
353
+ console.log(" ⚠️ Marginal improvement. More feedback data needed.");
354
+ }
355
+ else {
356
+ console.log(" ❌ No improvement. Check data quality or increase sample size.");
357
+ }
358
+ console.log();
359
+ }
@@ -0,0 +1,181 @@
1
+ "use strict";
2
+ /**
3
+ * P3.9: Evaluation Campaign Tests
4
+ *
5
+ * Verifying:
6
+ * 1. Failure attribution classifies benchmark misses correctly
7
+ * 2. Error budget dashboard produces actionable breakdown
8
+ * 3. Offline replay computes match rate against user choices
9
+ * 4. Ranker A/B comparison produces delta
10
+ */
11
+ var __createBinding = (this && this.__createBinding) || (Object.create ? (function(o, m, k, k2) {
12
+ if (k2 === undefined) k2 = k;
13
+ var desc = Object.getOwnPropertyDescriptor(m, k);
14
+ if (!desc || ("get" in desc ? !m.__esModule : desc.writable || desc.configurable)) {
15
+ desc = { enumerable: true, get: function() { return m[k]; } };
16
+ }
17
+ Object.defineProperty(o, k2, desc);
18
+ }) : (function(o, m, k, k2) {
19
+ if (k2 === undefined) k2 = k;
20
+ o[k2] = m[k];
21
+ }));
22
+ var __setModuleDefault = (this && this.__setModuleDefault) || (Object.create ? (function(o, v) {
23
+ Object.defineProperty(o, "default", { enumerable: true, value: v });
24
+ }) : function(o, v) {
25
+ o["default"] = v;
26
+ });
27
+ var __importStar = (this && this.__importStar) || (function () {
28
+ var ownKeys = function(o) {
29
+ ownKeys = Object.getOwnPropertyNames || function (o) {
30
+ var ar = [];
31
+ for (var k in o) if (Object.prototype.hasOwnProperty.call(o, k)) ar[ar.length] = k;
32
+ return ar;
33
+ };
34
+ return ownKeys(o);
35
+ };
36
+ return function (mod) {
37
+ if (mod && mod.__esModule) return mod;
38
+ var result = {};
39
+ if (mod != null) for (var k = ownKeys(mod), i = 0; i < k.length; i++) if (k[i] !== "default") __createBinding(result, mod, k[i]);
40
+ __setModuleDefault(result, mod);
41
+ return result;
42
+ };
43
+ })();
44
+ Object.defineProperty(exports, "__esModule", { value: true });
45
+ const vitest_1 = require("vitest");
46
+ const fs = __importStar(require("fs"));
47
+ const path = __importStar(require("path"));
48
+ const evaluation_campaign_1 = require("./evaluation-campaign");
49
+ const planner_telemetry_1 = require("./planner-telemetry");
50
+ const planner_trace_1 = require("./planner-trace");
51
+ // ═══════════════════════════════════════════════════════════════
52
+ // Failure Attribution
53
+ // ═══════════════════════════════════════════════════════════════
54
+ (0, vitest_1.describe)("Failure Attribution", () => {
55
+ (0, vitest_1.it)("classifies all 49 benchmark cases", async () => {
56
+ const attributed = await (0, evaluation_campaign_1.runFailureAttribution)();
57
+ (0, vitest_1.expect)(attributed.length).toBeGreaterThanOrEqual(49);
58
+ // Count by failure reason
59
+ const counts = {};
60
+ for (const a of attributed) {
61
+ counts[a.failureReason] = (counts[a.failureReason] || 0) + 1;
62
+ }
63
+ // Should have at least some successes and some failures
64
+ (0, vitest_1.expect)(counts["success"]).toBeGreaterThanOrEqual(1);
65
+ (0, vitest_1.expect)(Object.keys(counts).length).toBeGreaterThanOrEqual(2);
66
+ // Every attributed case has required fields
67
+ for (const a of attributed) {
68
+ (0, vitest_1.expect)(a.failureReason).toBeDefined();
69
+ (0, vitest_1.expect)(a.expectedRepair.length).toBeGreaterThan(0);
70
+ (0, vitest_1.expect)(["success", "missing_candidate", "bad_ranking", "bad_protocol_model", "goal_mismatch", "insufficient_history"]).toContain(a.failureReason);
71
+ }
72
+ }, 60000);
73
+ (0, vitest_1.it)("produces actionable error budget", async () => {
74
+ const attributed = await (0, evaluation_campaign_1.runFailureAttribution)();
75
+ const budget = (0, evaluation_campaign_1.computeErrorBudget)(attributed);
76
+ (0, vitest_1.expect)(budget.totalCases).toBeGreaterThanOrEqual(49);
77
+ (0, vitest_1.expect)(budget.successRate).toBeGreaterThanOrEqual(0);
78
+ (0, vitest_1.expect)(budget.successRate).toBeLessThanOrEqual(1);
79
+ (0, vitest_1.expect)(budget.recommendation.length).toBeGreaterThan(0);
80
+ // All failure reasons should sum to total
81
+ const sum = Object.values(budget.breakdown).reduce((s, v) => s + v, 0);
82
+ (0, vitest_1.expect)(sum).toBe(budget.totalCases);
83
+ (0, evaluation_campaign_1.printErrorBudget)(budget);
84
+ }, 60000);
85
+ });
86
+ // ═══════════════════════════════════════════════════════════════
87
+ // Offline Replay
88
+ // ═══════════════════════════════════════════════════════════════
89
+ const REPLAY_DIR = path.resolve(__dirname, "..", "test-evaluation-replay");
90
+ process.env.PROGMUNE_PROJECT_DIR = REPLAY_DIR;
91
+ fs.mkdirSync(REPLAY_DIR, { recursive: true });
92
+ fs.mkdirSync(path.join(REPLAY_DIR, ".progmune_corpus", "telemetry"), { recursive: true });
93
+ fs.mkdirSync(path.join(REPLAY_DIR, ".progmune_corpus", "traces"), { recursive: true });
94
+ function seedReplayData() {
95
+ const telemetry = new planner_telemetry_1.PlannerTelemetry(path.join(REPLAY_DIR, ".progmune_corpus", "telemetry", `replay-${Date.now()}.jsonl`));
96
+ const traceStore = new planner_trace_1.PlannerTraceStore(path.join(REPLAY_DIR, ".progmune_corpus", "traces", `replay-${Date.now()}.jsonl`));
97
+ // Seed: candidate A is safe (high acceptance), candidate B is fast (low acceptance)
98
+ const fpA = (0, planner_telemetry_1.candidateFingerprint)("FileProtocol", ["open_file", "write_file", "close_file"], "resource_leak");
99
+ const fpB = (0, planner_telemetry_1.candidateFingerprint)("FileProtocol", ["atomic_write"], "resource_leak");
100
+ // A accepted 80 times, B rejected 50 times
101
+ for (let i = 0; i < 80; i++) {
102
+ const id = telemetry.recordDecision({
103
+ goal: "safely write file",
104
+ protocol: "FileProtocol",
105
+ violationType: "resource_leak",
106
+ candidates: [
107
+ { candidateId: fpA, source: "protocol", evidenceSources: ["protocol"], actions: ["open_file", "write_file", "close_file"], explanation: "safe" },
108
+ { candidateId: fpB, source: "corpus", evidenceSources: ["corpus"], actions: ["atomic_write"], explanation: "fast" },
109
+ ],
110
+ selectedCandidateId: fpA,
111
+ });
112
+ telemetry.recordFeedback(id, { decision: "accepted", executionResult: { success: true, violations: [] }, timestamp: Date.now() });
113
+ }
114
+ for (let i = 0; i < 50; i++) {
115
+ const id = telemetry.recordDecision({
116
+ goal: "quick write",
117
+ protocol: "FileProtocol",
118
+ violationType: "resource_leak",
119
+ candidates: [
120
+ { candidateId: fpA, source: "protocol", evidenceSources: ["protocol"], actions: ["open_file", "write_file", "close_file"], explanation: "safe" },
121
+ { candidateId: fpB, source: "corpus", evidenceSources: ["corpus"], actions: ["atomic_write"], explanation: "fast" },
122
+ ],
123
+ selectedCandidateId: fpB,
124
+ });
125
+ telemetry.recordFeedback(id, { decision: "rejected", timestamp: Date.now() });
126
+ }
127
+ // Create traces where user chose A over B (original ranker put B first, user chose A)
128
+ for (let i = 0; i < 20; i++) {
129
+ traceStore.recordTrace({
130
+ decisionId: `pd-replay-${i}`,
131
+ goal: "safely write config file",
132
+ protocol: "FileProtocol",
133
+ violationType: "resource_leak",
134
+ candidates: [
135
+ { fingerprint: fpB, source: "corpus", evidenceSources: ["corpus"], actions: ["atomic_write"], score: 0.73, rank: 1 },
136
+ { fingerprint: fpA, source: "protocol", evidenceSources: ["protocol"], actions: ["open_file", "write_file", "close_file"], score: 0.68, rank: 2 },
137
+ ],
138
+ selectedFingerprint: fpA, // user chose A even though B was rank-1
139
+ accepted: true,
140
+ });
141
+ }
142
+ // Traces where user chose rank-1
143
+ for (let i = 20; i < 30; i++) {
144
+ traceStore.recordTrace({
145
+ decisionId: `pd-replay-${i}`,
146
+ goal: "safely write config file",
147
+ protocol: "FileProtocol",
148
+ violationType: "resource_leak",
149
+ candidates: [
150
+ { fingerprint: fpA, source: "protocol", evidenceSources: ["protocol"], actions: ["open_file", "write_file", "close_file"], score: 0.81, rank: 1 },
151
+ { fingerprint: fpB, source: "corpus", evidenceSources: ["corpus"], actions: ["atomic_write"], score: 0.73, rank: 2 },
152
+ ],
153
+ selectedFingerprint: fpA,
154
+ accepted: true,
155
+ });
156
+ }
157
+ return { telemetry, traceStore };
158
+ }
159
+ (0, vitest_1.describe)("Offline Replay", () => {
160
+ (0, vitest_1.it)("replays decisions and computes match rate", () => {
161
+ const { telemetry, traceStore } = seedReplayData();
162
+ const report = (0, evaluation_campaign_1.replayDecisions)(traceStore, telemetry);
163
+ (0, vitest_1.expect)(report.totalDecisions).toBeGreaterThanOrEqual(30);
164
+ (0, vitest_1.expect)(report.matchRate).toBeGreaterThanOrEqual(0);
165
+ (0, vitest_1.expect)(report.matchRate).toBeLessThanOrEqual(1);
166
+ // LearningRanker should match > 80% (A has 80 accepts vs B has 50 rejects)
167
+ (0, vitest_1.expect)(report.matchRate).toBeGreaterThan(0.8);
168
+ (0, evaluation_campaign_1.printReplayReport)(report);
169
+ });
170
+ (0, vitest_1.it)("compares rankers and shows delta", () => {
171
+ const { telemetry, traceStore } = seedReplayData();
172
+ const { baseline, learning, delta } = (0, evaluation_campaign_1.compareRankers)(traceStore, telemetry);
173
+ (0, vitest_1.expect)(baseline.totalDecisions).toBeGreaterThanOrEqual(30);
174
+ (0, vitest_1.expect)(learning.totalDecisions).toBeGreaterThanOrEqual(30);
175
+ (0, vitest_1.expect)(delta).toBeGreaterThan(0); // LearningRanker outperforms baseline
176
+ // Baseline (rank-1 = what planner showed first): B was rank-1 in 20/30 traces
177
+ // but user chose A. So baseline matches only when A was rank-1 (10/30 ≈ 33%)
178
+ (0, vitest_1.expect)(baseline.matchRate).toBeLessThan(learning.matchRate);
179
+ (0, evaluation_campaign_1.printRankerComparison)(baseline, learning, delta);
180
+ });
181
+ });