progmune-runtime 2.1.5 → 3.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (356) hide show
  1. package/README.md +108 -468
  2. package/dist/ablation-study.js +144 -0
  3. package/dist/ablation-study.test.js +18 -0
  4. package/dist/action-runtime.js +3 -1
  5. package/dist/active-learning.js +211 -0
  6. package/dist/analytics.js +139 -0
  7. package/dist/asset-factory.js +309 -0
  8. package/dist/asset-growth.js +244 -0
  9. package/dist/asset-promotion.js +382 -0
  10. package/dist/asset-quality.js +550 -0
  11. package/dist/audit/business-translator.js +285 -0
  12. package/dist/audit/cli.js +66 -0
  13. package/dist/audit/formatters/html.js +379 -0
  14. package/dist/audit/formatters/json.js +11 -0
  15. package/dist/audit/formatters/markdown.js +192 -0
  16. package/dist/audit/formatters/terminal.js +189 -0
  17. package/dist/audit/index.js +25 -0
  18. package/dist/audit/report-builder.js +318 -0
  19. package/dist/audit/types.js +8 -0
  20. package/dist/audit.js +3 -3
  21. package/dist/auto-benchmark-generator.js +137 -0
  22. package/dist/auto-benchmark-generator.test.js +45 -0
  23. package/dist/auto-protocol-synthesizer.js +362 -0
  24. package/dist/auto-protocol-synthesizer.test.js +82 -0
  25. package/dist/autonomous-patch.js +175 -0
  26. package/dist/autonomous-patch.test.js +128 -0
  27. package/dist/badge/badge-server.js +98 -0
  28. package/dist/behavior-miner.js +442 -0
  29. package/dist/belief-layer.js +475 -0
  30. package/dist/benchmark-count.js +5 -0
  31. package/dist/benchmark-generator.js +211 -0
  32. package/dist/benchmark-harness.js +201 -0
  33. package/dist/benchmark-pass-rate.js +7 -0
  34. package/dist/benchmark-report.js +8 -3
  35. package/dist/benchmark-save.js +14 -1
  36. package/dist/bootstrap-validation.js +197 -0
  37. package/dist/bootstrap-validation.test.js +51 -0
  38. package/dist/branch-ledger.js +1 -1
  39. package/dist/capability-gap.js +130 -0
  40. package/dist/certify-html.js +351 -0
  41. package/dist/certify.js +326 -0
  42. package/dist/check.js +4 -4
  43. package/dist/compliance-miner.js +447 -0
  44. package/dist/continuous-benchmark.js +194 -0
  45. package/dist/continuous-benchmark.test.js +116 -0
  46. package/dist/corpus-stats.js +173 -0
  47. package/dist/counterfactual-engine.js +288 -0
  48. package/dist/coverage-dashboard.js +109 -0
  49. package/dist/coverage-system.test.js +205 -0
  50. package/dist/cross-repo-precision.js +352 -0
  51. package/dist/cve-benchmark.js +180 -0
  52. package/dist/cve-benchmark.test.js +28 -0
  53. package/dist/cve-collector.js +73 -0
  54. package/dist/data-quality.js +141 -0
  55. package/dist/decision-engine.js +388 -0
  56. package/dist/derive-metadata.js +250 -0
  57. package/dist/difficulty-active.test.js +198 -0
  58. package/dist/difficulty-map.js +244 -0
  59. package/dist/discovery-analytics.js +125 -0
  60. package/dist/discovery-model.js +149 -0
  61. package/dist/discovery-optimize.test.js +199 -0
  62. package/dist/discovery-trace.js +276 -0
  63. package/dist/discovery-trace.test.js +97 -0
  64. package/dist/emitter.js +83 -1
  65. package/dist/enterprise-dashboard.js +405 -0
  66. package/dist/eval-hardening.js +297 -0
  67. package/dist/eval-hardening.test.js +85 -0
  68. package/dist/evaluation-campaign.js +359 -0
  69. package/dist/evaluation-campaign.test.js +181 -0
  70. package/dist/evidence-growth.js +143 -0
  71. package/dist/evidence-repository.js +209 -0
  72. package/dist/evidence-system.js +441 -0
  73. package/dist/execute.js +15 -7
  74. package/dist/experimental/software-physics.js +291 -0
  75. package/dist/experimental/state-inference.js +516 -0
  76. package/dist/experimental/unsupervised-physics.js +230 -0
  77. package/dist/extract-ir-python.js +54 -7
  78. package/dist/extract-ir.js +376 -12
  79. package/dist/failure-collector.js +2 -2
  80. package/dist/failure-corpus.js +322 -9
  81. package/dist/feedback.js +16 -5
  82. package/dist/feedback.test.js +49 -0
  83. package/dist/file-lock.js +1 -1
  84. package/dist/flywheel-batch.js +292 -0
  85. package/dist/frameworks/express-cli.js +237 -0
  86. package/dist/frameworks/express-detector.js +445 -0
  87. package/dist/frameworks/express-detector.test.js +206 -0
  88. package/dist/frameworks/index.js +30 -0
  89. package/dist/frameworks/nestjs-detector.js +302 -0
  90. package/dist/frameworks/trpc-detector.js +161 -0
  91. package/dist/frameworks/version-awareness.js +179 -0
  92. package/dist/function-synonyms.js +164 -0
  93. package/dist/function-synonyms.test.js +68 -0
  94. package/dist/generalization.test.js +352 -0
  95. package/dist/goal-annotator.js +113 -0
  96. package/dist/goal-planner.js +563 -0
  97. package/dist/gold-cve.js +164 -0
  98. package/dist/gold-cve.test.js +104 -0
  99. package/dist/gold-quality.js +206 -0
  100. package/dist/gold-tiers.js +241 -0
  101. package/dist/governance-dashboard.js +327 -0
  102. package/dist/graph-viz.js +240 -0
  103. package/dist/guided-frontier.js +195 -0
  104. package/dist/hierarchical-planner.js +148 -0
  105. package/dist/identifier-parser.js +260 -0
  106. package/dist/immune-metrics.js +93 -0
  107. package/dist/immune-receiver.js +158 -0
  108. package/dist/immune-reporter.js +1 -1
  109. package/dist/improvement-orchestrator.js +206 -0
  110. package/dist/inject-p0-vocabulary.js +300 -0
  111. package/dist/intent-parser.js +218 -0
  112. package/dist/invariant-algebra.js +476 -0
  113. package/dist/invariant-calculus.js +533 -0
  114. package/dist/ir-utils.js +70 -0
  115. package/dist/ir-utils.test.js +50 -0
  116. package/dist/knowledge-api.js +312 -0
  117. package/dist/knowledge-evolution.js +452 -0
  118. package/dist/knowledge-explorer.js +506 -0
  119. package/dist/knowledge-flywheel.js +274 -0
  120. package/dist/knowledge-governance.js +338 -0
  121. package/dist/knowledge-governance.test.js +150 -0
  122. package/dist/knowledge-graph.js +181 -0
  123. package/dist/knowledge-guided-synth.js +246 -0
  124. package/dist/knowledge-loop.test.js +77 -0
  125. package/dist/knowledge-object.js +316 -0
  126. package/dist/knowledge-package.js +98 -0
  127. package/dist/kpi-dashboard.js +561 -0
  128. package/dist/l3-cross-function.js +280 -0
  129. package/dist/learning-ranker.js +148 -0
  130. package/dist/learning-ranker.test.js +291 -0
  131. package/dist/ledger/accountability.js +322 -0
  132. package/dist/ledger/chain-builder.js +185 -0
  133. package/dist/ledger/cli.js +222 -0
  134. package/dist/ledger/index.js +13 -0
  135. package/dist/ledger/signatures.js +193 -0
  136. package/dist/ledger/types.js +9 -0
  137. package/dist/llm.js +74 -3
  138. package/dist/load-benchmarks.js +8 -3
  139. package/dist/logger.js +66 -0
  140. package/dist/logger.test.js +37 -0
  141. package/dist/logistic-reward.js +339 -0
  142. package/dist/logistic-reward.test.js +180 -0
  143. package/dist/macro-graph.js +193 -0
  144. package/dist/macro-repair.js +183 -0
  145. package/dist/mcp-server.mjs +1202 -483
  146. package/dist/memory-layer.js +42 -5
  147. package/dist/multi-repo-precision.js +422 -0
  148. package/dist/name-free-protocol.js +425 -0
  149. package/dist/name-free-protocol.test.js +170 -0
  150. package/dist/name-scrambling.js +138 -0
  151. package/dist/name-scrambling.test.js +16 -0
  152. package/dist/p3-observability.test.js +281 -0
  153. package/dist/p5-orchestrator.test.js +225 -0
  154. package/dist/pairwise-preference.js +294 -0
  155. package/dist/pairwise-preference.test.js +140 -0
  156. package/dist/planner-constraints.js +104 -0
  157. package/dist/planner-prompts.js +155 -0
  158. package/dist/planner-telemetry.js +415 -0
  159. package/dist/planner-trace.js +214 -0
  160. package/dist/planner.js +162 -167
  161. package/dist/plsb/artifact.js +116 -0
  162. package/dist/plsb/cli.js +71 -0
  163. package/dist/plsb/index.js +19 -0
  164. package/dist/plsb/leaderboard.js +249 -0
  165. package/dist/plsb/report-md.js +156 -0
  166. package/dist/plsb/schema.js +179 -0
  167. package/dist/plsb-benchmark.js +284 -0
  168. package/dist/plsb-benchmark.test.js +119 -0
  169. package/dist/policy/cli.js +134 -0
  170. package/dist/policy/engine.js +333 -0
  171. package/dist/policy/index.js +12 -0
  172. package/dist/policy/types.js +59 -0
  173. package/dist/policy-miner.js +505 -0
  174. package/dist/precision-analyze.js +229 -0
  175. package/dist/precision-benchmark.js +147 -0
  176. package/dist/precision-label-c.js +134 -0
  177. package/dist/precision-label.js +193 -0
  178. package/dist/precision-report-c.js +149 -0
  179. package/dist/precision-report.js +246 -0
  180. package/dist/progmune-status.js +108 -0
  181. package/dist/proof-engine.js +479 -0
  182. package/dist/proof-provenance.js +315 -0
  183. package/dist/protocol-coverage.js +294 -0
  184. package/dist/protocol-detector.js +1189 -0
  185. package/dist/protocol-embedding-expanded.js +297 -0
  186. package/dist/protocol-embedding-expanded.test.js +97 -0
  187. package/dist/protocol-embedding.js +195 -0
  188. package/dist/protocol-embedding.test.js +82 -0
  189. package/dist/protocol-extractor-v2.js +354 -0
  190. package/dist/protocol-extractor-v2.test.js +140 -0
  191. package/dist/protocol-extractor.js +310 -0
  192. package/dist/protocol-extractor.test.js +113 -0
  193. package/dist/protocol-foundation.js +322 -0
  194. package/dist/protocol-foundation.test.js +163 -0
  195. package/dist/protocol-frontier.js +243 -0
  196. package/dist/protocol-frontier.test.js +92 -0
  197. package/dist/protocol-gap-analyzer.js +228 -0
  198. package/dist/protocol-gap-analyzer.test.js +49 -0
  199. package/dist/protocol-invariants.js +276 -0
  200. package/dist/protocol-invariants.test.js +111 -0
  201. package/dist/protocol-knowledge.js +464 -0
  202. package/dist/protocol-miner.js +343 -0
  203. package/dist/protocol-mining.js +207 -0
  204. package/dist/protocol-mining.test.js +37 -0
  205. package/dist/protocol-registry.js +1 -1
  206. package/dist/protocol-security-benchmark.js +222 -0
  207. package/dist/protocol-vulnerability.js +257 -0
  208. package/dist/protocol-vulnerability.test.js +60 -0
  209. package/dist/python-benchmark.js +120 -0
  210. package/dist/python-emitter.js +163 -45
  211. package/dist/python-protocol-extractor.js +187 -0
  212. package/dist/python-protocol-extractor.test.js +116 -0
  213. package/dist/realworld-benchmark.js +646 -0
  214. package/dist/realworld-benchmark.test.js +36 -0
  215. package/dist/repair-arch.test.js +411 -0
  216. package/dist/repair-evolution.test.js +454 -0
  217. package/dist/repair-executor.js +719 -0
  218. package/dist/repair-proposal.js +4 -4
  219. package/dist/repair-ranker.js +141 -0
  220. package/dist/repair-strategies.js +419 -0
  221. package/dist/repair-taxonomy.js +234 -0
  222. package/dist/repair-types.js +12 -0
  223. package/dist/repo-evaluator.js +250 -0
  224. package/dist/repo-evaluator.test.js +128 -0
  225. package/dist/resource-abstraction.js +242 -0
  226. package/dist/resource-detector.js +211 -0
  227. package/dist/result.test.js +43 -0
  228. package/dist/reward-system.js +411 -0
  229. package/dist/reward-system.test.js +175 -0
  230. package/dist/risk-model.js +215 -0
  231. package/dist/rule-miner.js +234 -7
  232. package/dist/rule-specificity.js +254 -0
  233. package/dist/runtime-types.js +27 -0
  234. package/dist/scaffold.js +208 -0
  235. package/dist/scale-collector.test.js +101 -0
  236. package/dist/scale-trajectory-collector.js +128 -0
  237. package/dist/sdk.js +250 -0
  238. package/dist/search-planner.js +4 -41
  239. package/dist/semantic-snapshot.js +1 -1
  240. package/dist/semantic-topology.js +121 -0
  241. package/dist/semantic-trace.js +310 -317
  242. package/dist/sequence-extractor.js +343 -0
  243. package/dist/skill-library.js +245 -0
  244. package/dist/skill-planner.test.js +189 -0
  245. package/dist/software-physics.js +291 -0
  246. package/dist/software-physics.test.js +81 -0
  247. package/dist/ssg-precision.js +478 -0
  248. package/dist/ssg-validator.js +71 -21
  249. package/dist/state-inference-doubleblind.test.js +160 -0
  250. package/dist/state-inference.js +516 -0
  251. package/dist/state-inference.test.js +115 -0
  252. package/dist/state-machine-fingerprint.js +345 -0
  253. package/dist/state-machine-fingerprint.test.js +120 -0
  254. package/dist/state-miner.js +386 -0
  255. package/dist/state-name-inference.js +213 -0
  256. package/dist/state-name-inference.test.js +69 -0
  257. package/dist/strategy-planner.js +326 -59
  258. package/dist/strategy-planner.test.js +135 -0
  259. package/dist/telemetry-analytics.test.js +402 -0
  260. package/dist/terminal-format.js +68 -0
  261. package/dist/terminal-format.test.js +83 -0
  262. package/dist/topology-factory.js +196 -0
  263. package/dist/topology-representation.js +242 -0
  264. package/dist/topology-representation.test.js +27 -0
  265. package/dist/trajectory-augmentation.js +254 -0
  266. package/dist/trajectory-augmentation.test.js +63 -0
  267. package/dist/trajectory-corpus.js +440 -0
  268. package/dist/trajectory-corpus.test.js +32 -0
  269. package/dist/trajectory-feedback.test.js +116 -0
  270. package/dist/transition-synthesizer.js +286 -0
  271. package/dist/transition-synthesizer.test.js +123 -0
  272. package/dist/trust/api-semantic-mapper.js +809 -0
  273. package/dist/trust/call-graph-propagator.js +225 -0
  274. package/dist/trust/cli.js +122 -0
  275. package/dist/trust/compliance-scorer.js +283 -0
  276. package/dist/trust/confidence-calculator.js +261 -0
  277. package/dist/trust/engine.js +1145 -0
  278. package/dist/trust/explainability.js +85 -0
  279. package/dist/trust/formatters/ci.js +42 -0
  280. package/dist/trust/formatters/json.js +11 -0
  281. package/dist/trust/formatters/terminal.js +152 -0
  282. package/dist/trust/index.js +39 -0
  283. package/dist/trust/phase1-verify.js +171 -0
  284. package/dist/trust/protocol-domain-validator.js +697 -0
  285. package/dist/trust/score-calculator.js +282 -0
  286. package/dist/trust/ssg-bridge.js +641 -0
  287. package/dist/trust/ssg-bridge.test.js +269 -0
  288. package/dist/trust/types.js +67 -0
  289. package/dist/trust/violation-trace.js +335 -0
  290. package/dist/trust-api.js +179 -0
  291. package/dist/trust-calibration.js +279 -0
  292. package/dist/unknown-protocol-discovery.js +339 -0
  293. package/dist/unknown-protocol-discovery.test.js +102 -0
  294. package/dist/unsupervised-physics.js +230 -0
  295. package/dist/unsupervised-physics.test.js +95 -0
  296. package/dist/utils.test.js +37 -0
  297. package/dist/validator.js +187 -10
  298. package/dist/verification-intelligence.js +475 -0
  299. package/dist/verify-api.js +432 -0
  300. package/dist/vi-impact-report.js +293 -0
  301. package/dist/wl-fingerprint.js +162 -0
  302. package/dist/wl-fingerprint.test.js +130 -0
  303. package/dist/zeroshot-strategy.js +139 -0
  304. package/docs/Progmune_/346/212/225/350/265/204/344/272/272/347/231/275/347/232/256/344/271/246_v2.0.html +576 -0
  305. package/docs/Progmune_/351/241/271/347/233/256/345/205/250/350/247/243.html +710 -0
  306. package/package.json +74 -7
  307. package/protocols.json +1956 -50
  308. package/.dockerignore +0 -14
  309. package/.mcp.json +0 -11
  310. package/.progmune_allowlist +0 -50
  311. package/.test_report/test_report.md +0 -87
  312. package/Dockerfile +0 -9
  313. package/FAQ.md +0 -167
  314. package/WHITEPAPER.md +0 -540
  315. package/demo-project/auth.ts +0 -55
  316. package/demo-project/tsconfig.json +0 -8
  317. package/dist/acl-breakdown.js +0 -13
  318. package/dist/all-sessions.js +0 -11
  319. package/dist/antibody-stats.js +0 -11
  320. package/dist/branch-tree-count.js +0 -14
  321. package/dist/common-fixpath.js +0 -12
  322. package/dist/constraint-types.js +0 -12
  323. package/dist/exec-metrics.js +0 -11
  324. package/dist/failure-report.js +0 -11
  325. package/dist/fast-path-hits.js +0 -13
  326. package/dist/fingerprint-list.js +0 -15
  327. package/dist/gen-history-log.js +0 -13
  328. package/dist/heatmap-data.js +0 -11
  329. package/dist/recent-session.js +0 -12
  330. package/dist/svl-distribution.js +0 -11
  331. package/dist/terminal-status.js +0 -11
  332. package/dist/token-savings.js +0 -11
  333. package/dist/total-repairs.js +0 -12
  334. package/dist/unresolved-count.js +0 -12
  335. package/dist/valid-fingerprints.js +0 -13
  336. package/dist/verify-ledgers.js +0 -11
  337. package/docs/whitepaper-style.css +0 -77
  338. package/docs/whitepaper-v2.1.md +0 -609
  339. package/docs/whitepaper-v2.2.md +0 -1064
  340. package/docs/whitepaper-v2.2.pdf +0 -0
  341. package/fly.toml +0 -31
  342. package/public/dashboard.html +0 -119
  343. package/server/hub.js +0 -116
  344. package/test/replay-golden/sess_1780063202050_mgeld.json +0 -9
  345. package/test/replay-golden/sess_1780064032560_gocld.json +0 -354
  346. package/test/replay-golden/sess_1780064413331_s2709.json +0 -606
  347. package/test/replay-golden/sess_1780064792710_y3avo.json +0 -614
  348. package/test/replay-golden.ts +0 -84
  349. package/test_benchmark.js +0 -165
  350. package/test_comprehensive.mjs +0 -638
  351. package/test_concurrency.js +0 -129
  352. package/test_ir_robustness.js +0 -85
  353. package/test_semantic_contracts.js +0 -269
  354. package/test_ssg_stress.js +0 -156
  355. package/test_svl3.js +0 -58
  356. package/tsconfig.json +0 -17
@@ -0,0 +1,411 @@
1
+ "use strict";
2
+ /**
3
+ * P4.1-4.4: Reward System — Pairwise + Evaluator + Context + Dataset
4
+ *
5
+ * P4.1 Bradley-Terry Pairwise Model:
6
+ * P(A > B) = σ(score(A) - score(B))
7
+ * Trains on RepairPreference data from P3.11
8
+ * Signal quality: pairwise (A > B) > binary (accepted/rejected)
9
+ *
10
+ * P4.2 Off-Policy Evaluator++:
11
+ * Ranking-aware metrics: NDCG, Top1 Lift, Top3 Lift, Acceptance Lift
12
+ * Any new Ranker must pass Acceptance Lift > 0 before deployment
13
+ *
14
+ * P4.3 Contextual Reward:
15
+ * One-hot goal/protocol/violation features
16
+ * Same logistic regression, richer feature space
17
+ *
18
+ * P4.4 Reward Dataset:
19
+ * RewardExample JSONL export for future retraining at scale
20
+ */
21
+ var __createBinding = (this && this.__createBinding) || (Object.create ? (function(o, m, k, k2) {
22
+ if (k2 === undefined) k2 = k;
23
+ var desc = Object.getOwnPropertyDescriptor(m, k);
24
+ if (!desc || ("get" in desc ? !m.__esModule : desc.writable || desc.configurable)) {
25
+ desc = { enumerable: true, get: function() { return m[k]; } };
26
+ }
27
+ Object.defineProperty(o, k2, desc);
28
+ }) : (function(o, m, k, k2) {
29
+ if (k2 === undefined) k2 = k;
30
+ o[k2] = m[k];
31
+ }));
32
+ var __setModuleDefault = (this && this.__setModuleDefault) || (Object.create ? (function(o, v) {
33
+ Object.defineProperty(o, "default", { enumerable: true, value: v });
34
+ }) : function(o, v) {
35
+ o["default"] = v;
36
+ });
37
+ var __importStar = (this && this.__importStar) || (function () {
38
+ var ownKeys = function(o) {
39
+ ownKeys = Object.getOwnPropertyNames || function (o) {
40
+ var ar = [];
41
+ for (var k in o) if (Object.prototype.hasOwnProperty.call(o, k)) ar[ar.length] = k;
42
+ return ar;
43
+ };
44
+ return ownKeys(o);
45
+ };
46
+ return function (mod) {
47
+ if (mod && mod.__esModule) return mod;
48
+ var result = {};
49
+ if (mod != null) for (var k = ownKeys(mod), i = 0; i < k.length; i++) if (k[i] !== "default") __createBinding(result, mod, k[i]);
50
+ __setModuleDefault(result, mod);
51
+ return result;
52
+ };
53
+ })();
54
+ Object.defineProperty(exports, "__esModule", { value: true });
55
+ exports.ContextualRewardModel = exports.PairwiseRewardModel = void 0;
56
+ exports.buildRewardDataset = buildRewardDataset;
57
+ exports.saveRewardDataset = saveRewardDataset;
58
+ exports.loadRewardDataset = loadRewardDataset;
59
+ exports.computeNDCG = computeNDCG;
60
+ exports.compareRankersOffPolicy = compareRankersOffPolicy;
61
+ exports.deploymentGate = deploymentGate;
62
+ exports.printRankingMetrics = printRankingMetrics;
63
+ exports.buildContextualFeatures = buildContextualFeatures;
64
+ exports.printRewardSystemReport = printRewardSystemReport;
65
+ const fs = __importStar(require("fs"));
66
+ const path = __importStar(require("path"));
67
+ const REWARD_DS_DIR = path.resolve(process.env.PROGMUNE_PROJECT_DIR || process.cwd(), ".progmune_corpus", "reward_dataset");
68
+ function ensureDir(dir) {
69
+ if (!fs.existsSync(dir))
70
+ fs.mkdirSync(dir, { recursive: true });
71
+ }
72
+ /** Export all telemetry decisions as RewardExample JSONL. */
73
+ function buildRewardDataset(telemetry) {
74
+ const examples = [];
75
+ for (const d of telemetry.all()) {
76
+ if (!d.feedback || !d.selectedCandidateId)
77
+ continue;
78
+ const sel = d.candidates.find(c => c.candidateId === d.selectedCandidateId);
79
+ if (!sel || sel.actions.length === 0)
80
+ continue;
81
+ const accepted = d.feedback.decision === "accepted";
82
+ const executed = d.feedback.executionResult?.success === true;
83
+ const success = accepted && executed;
84
+ const actionCount = sel.actions.length;
85
+ const maxActions = Math.max(actionCount, 8);
86
+ const safety = Math.max(0, 1.0 - actionCount / 10);
87
+ examples.push({
88
+ fingerprint: sel.candidateId,
89
+ features: [
90
+ safety, // protocolSafety
91
+ 0.5, // historicalSuccessRate
92
+ actionCount / maxActions, // normActionCount
93
+ Math.min(1, actionCount / maxActions), // latencyCost
94
+ 1.0 - actionCount / maxActions, // auditability
95
+ accepted ? 1.0 : 0.0, // acceptanceRate
96
+ executed ? 1.0 : 0.0, // executionSuccessRate
97
+ ],
98
+ accepted, executed, success,
99
+ goal: d.goal,
100
+ protocol: d.protocol,
101
+ violationType: d.violationType,
102
+ timestamp: d.timestamp,
103
+ });
104
+ }
105
+ return examples;
106
+ }
107
+ /** Persist reward dataset to JSONL. */
108
+ function saveRewardDataset(examples, dir) {
109
+ const outDir = dir || REWARD_DS_DIR;
110
+ ensureDir(outDir);
111
+ const filepath = path.join(outDir, `reward-${new Date().toISOString().slice(0, 10)}.jsonl`);
112
+ const lines = examples.map(e => JSON.stringify(e)).join("\n") + "\n";
113
+ fs.writeFileSync(filepath, lines);
114
+ return filepath;
115
+ }
116
+ /** Load all reward datasets. */
117
+ function loadRewardDataset(dir) {
118
+ const outDir = dir || REWARD_DS_DIR;
119
+ if (!fs.existsSync(outDir))
120
+ return [];
121
+ const examples = [];
122
+ const files = fs.readdirSync(outDir).filter(f => f.endsWith(".jsonl"));
123
+ for (const file of files) {
124
+ const lines = fs.readFileSync(path.join(outDir, file), "utf-8").trim().split("\n");
125
+ for (const line of lines) {
126
+ if (!line.trim())
127
+ continue;
128
+ try {
129
+ examples.push(JSON.parse(line));
130
+ }
131
+ catch { /* skip */ }
132
+ }
133
+ }
134
+ return examples;
135
+ }
136
+ // ═══════════════════════════════════════════════════════════════
137
+ // P4.1: Bradley-Terry Pairwise Model
138
+ // ═══════════════════════════════════════════════════════════════
139
+ function sigmoid(z) {
140
+ if (z > 20)
141
+ return 1.0;
142
+ if (z < -20)
143
+ return 0.0;
144
+ return 1.0 / (1.0 + Math.exp(-z));
145
+ }
146
+ function dot(a, b) {
147
+ return a.reduce((s, v, i) => s + v * b[i], 0);
148
+ }
149
+ /**
150
+ * Bradley-Terry pairwise reward model.
151
+ *
152
+ * score(x) = w·x + b
153
+ * P(A > B) = σ(score(A) - score(B))
154
+ *
155
+ * Trains on RepairPreference data (winner > loser pairs).
156
+ */
157
+ class PairwiseRewardModel {
158
+ constructor(weights, bias) {
159
+ this.weights = weights || new Array(7).fill(0);
160
+ this.bias = bias || 0;
161
+ this.trained = weights !== undefined;
162
+ this.trainedSamples = 0;
163
+ }
164
+ get isTrained() { return this.trained; }
165
+ get sampleCount() { return this.trainedSamples; }
166
+ /** Score a feature vector. */
167
+ score(features) {
168
+ return dot(this.weights, features) + this.bias;
169
+ }
170
+ /** Probability that A beats B. */
171
+ predictPair(winnerFeatures, loserFeatures) {
172
+ return sigmoid(this.score(winnerFeatures) - this.score(loserFeatures));
173
+ }
174
+ /** Convert preferences to pairwise samples. */
175
+ static preferencesToSamples(preferences) {
176
+ // We need feature vectors for each fingerprint — extract from telemetry
177
+ // For now, use heuristic features based on action length
178
+ return preferences.map(p => {
179
+ // Heuristic: winner has shorter action chain? or parse from fingerprint
180
+ const wCount = p.winnerActionCount || 1;
181
+ const lCount = p.loserActionCount || 2;
182
+ const maxA = Math.max(wCount, lCount, 8);
183
+ return {
184
+ winnerFeatures: [
185
+ 1.0 - wCount / maxA, 0.5, wCount / maxA,
186
+ Math.min(1, wCount / maxA), 1.0 - wCount / maxA,
187
+ 0.8, 0.9,
188
+ ],
189
+ loserFeatures: [
190
+ 1.0 - lCount / maxA, 0.5, lCount / maxA,
191
+ Math.min(1, lCount / maxA), 1.0 - lCount / maxA,
192
+ 0.2, 0.1,
193
+ ],
194
+ goal: p.goal,
195
+ protocol: p.protocol,
196
+ };
197
+ });
198
+ }
199
+ /** Train on pairwise preference samples. */
200
+ static train(samples, learningRate = 0.01, epochs = 100, l2Lambda = 0.001) {
201
+ if (samples.length === 0)
202
+ return new PairwiseRewardModel();
203
+ let w = new Array(7).fill(0).map(() => (Math.random() - 0.5) * 0.1);
204
+ let b = 0.0;
205
+ for (let epoch = 0; epoch < epochs; epoch++) {
206
+ const shuffled = [...samples].sort(() => Math.random() - 0.5);
207
+ const wGrad = new Array(7).fill(0);
208
+ for (const s of shuffled) {
209
+ const diff = featureDiff(s.winnerFeatures, s.loserFeatures);
210
+ const z = dot(w, diff); // + b cancels out for pairwise
211
+ const p = sigmoid(z);
212
+ const error = p - 1.0; // winner should win
213
+ for (let i = 0; i < 7; i++) {
214
+ wGrad[i] += error * diff[i];
215
+ }
216
+ }
217
+ const n = shuffled.length;
218
+ for (let i = 0; i < 7; i++) {
219
+ w[i] -= learningRate * (wGrad[i] / n + l2Lambda * w[i]);
220
+ }
221
+ }
222
+ const model = new PairwiseRewardModel(w, b);
223
+ model.trained = true;
224
+ model.trainedSamples = samples.length;
225
+ return model;
226
+ }
227
+ }
228
+ exports.PairwiseRewardModel = PairwiseRewardModel;
229
+ function featureDiff(a, b) {
230
+ return a.map((v, i) => v - b[i]);
231
+ }
232
+ /**
233
+ * Compute NDCG for a ranked list against ground truth relevance.
234
+ * relevance: binary array (1 = correct, 0 = incorrect).
235
+ */
236
+ function computeNDCG(relevance, k) {
237
+ const K = k || relevance.length;
238
+ let dcg = 0;
239
+ for (let i = 0; i < Math.min(K, relevance.length); i++) {
240
+ dcg += relevance[i] / Math.log2(i + 2); // i+2 because log2(1) = 0, so start at log2(2)
241
+ }
242
+ // Ideal DCG: sorted descending
243
+ const ideal = [...relevance].sort((a, b) => b - a);
244
+ let idcg = 0;
245
+ for (let i = 0; i < Math.min(K, ideal.length); i++) {
246
+ idcg += ideal[i] / Math.log2(i + 2);
247
+ }
248
+ return idcg > 0 ? dcg / idcg : 0;
249
+ }
250
+ /**
251
+ * Compare two ranking systems using NDCG and lift metrics.
252
+ *
253
+ * @param decisions Historical decisions with ground truth (user choice)
254
+ * @param oldRanker Baseline ranker
255
+ * @param newRanker New ranker to evaluate
256
+ * @returns Lift metrics
257
+ */
258
+ function compareRankersOffPolicy(decisions, _oldRanker, newRanker) {
259
+ let oldNdcgTotal = 0;
260
+ let newNdcgTotal = 0;
261
+ let oldTop1 = 0;
262
+ let newTop1 = 0;
263
+ let oldTop3 = 0;
264
+ let newTop3 = 0;
265
+ let oldAccept = 0;
266
+ let newAccept = 0;
267
+ for (const d of decisions) {
268
+ // Old ranker: assume original order (index 0 = rank 1)
269
+ const oldRelevance = d.candidates.map(c => (c.accepted ? 1 : 0));
270
+ const oldNdcg = computeNDCG(oldRelevance, 3);
271
+ // New ranker: re-rank by newRanker scores
272
+ const scored = d.candidates.map((c, i) => ({ ...c, origIdx: i, score: newRanker(c.features) }));
273
+ scored.sort((a, b) => b.score - a.score);
274
+ const newRelevance = scored.map(c => (c.accepted ? 1 : 0));
275
+ const newNdcg = computeNDCG(newRelevance, 3);
276
+ oldNdcgTotal += oldNdcg;
277
+ newNdcgTotal += newNdcg;
278
+ // Top-1/Top-3 matches
279
+ const oldTop = d.candidates.slice(0, 1);
280
+ const newTop = scored.slice(0, 1);
281
+ if (oldTop.some(c => c.accepted))
282
+ oldTop1++;
283
+ if (newTop.some(c => c.accepted))
284
+ newTop1++;
285
+ const oldTop3Cands = d.candidates.slice(0, 3);
286
+ const newTop3Cands = scored.slice(0, 3);
287
+ if (oldTop3Cands.some(c => c.accepted))
288
+ oldTop3++;
289
+ if (newTop3Cands.some(c => c.accepted))
290
+ newTop3++;
291
+ // Acceptance: did the top candidate get accepted?
292
+ if (d.candidates[0]?.accepted)
293
+ oldAccept++;
294
+ if (scored[0]?.accepted)
295
+ newAccept++;
296
+ }
297
+ const n = decisions.length || 1;
298
+ return {
299
+ ndcg: newNdcgTotal / n,
300
+ top1Lift: (newTop1 - oldTop1) / n,
301
+ top3Lift: (newTop3 - oldTop3) / n,
302
+ acceptanceLift: (newAccept - oldAccept) / n,
303
+ };
304
+ }
305
+ /**
306
+ * Deployment gate: new ranker must have Acceptance Lift > 0.
307
+ */
308
+ function deploymentGate(metrics) {
309
+ if (metrics.acceptanceLift <= 0) {
310
+ return { passed: false, reason: `Acceptance Lift ${(metrics.acceptanceLift * 100).toFixed(1)}% ≤ 0 — rejected` };
311
+ }
312
+ if (metrics.ndcg < 0.3) {
313
+ return { passed: false, reason: `NDCG ${(metrics.ndcg * 100).toFixed(1)}% < 30% — ranking quality insufficient` };
314
+ }
315
+ return { passed: true, reason: `All gates passed. Lift: ${(metrics.acceptanceLift * 100).toFixed(1)}%, NDCG: ${(metrics.ndcg * 100).toFixed(1)}%` };
316
+ }
317
+ function printRankingMetrics(metrics) {
318
+ console.log("\n─── Off-Policy Ranking Evaluation ───");
319
+ console.log(` NDCG: ${(metrics.ndcg * 100).toFixed(1)}%`);
320
+ console.log(` Top-1 Lift: ${(metrics.top1Lift > 0 ? "+" : "")}${(metrics.top1Lift * 100).toFixed(1)}%`);
321
+ console.log(` Top-3 Lift: ${(metrics.top3Lift > 0 ? "+" : "")}${(metrics.top3Lift * 100).toFixed(1)}%`);
322
+ console.log(` Acceptance Lift: ${(metrics.acceptanceLift > 0 ? "+" : "")}${(metrics.acceptanceLift * 100).toFixed(1)}%`);
323
+ const gate = deploymentGate(metrics);
324
+ console.log(`\n Gate: ${gate.passed ? "✅ PASS" : "❌ FAIL"} — ${gate.reason}`);
325
+ console.log();
326
+ }
327
+ // ═══════════════════════════════════════════════════════════════
328
+ // P4.3: Contextual Reward Features
329
+ // ═══════════════════════════════════════════════════════════════
330
+ const KNOWN_GOALS = [
331
+ "safely write", "authenticate", "logout", "query database",
332
+ "extract IR", "validate action", "emit code", "record session",
333
+ ];
334
+ const KNOWN_PROTOCOLS = ["FileProtocol", "AuthProtocol", "DBProtocol", "IRProtocol"];
335
+ const KNOWN_VIOLATIONS = ["resource_leak", "missing_prerequisite", "illegal_state_transition"];
336
+ /**
337
+ * Extend base 7-d features with one-hot contextual features.
338
+ * Total: 7 + 8 + 4 + 3 = 22 features.
339
+ */
340
+ function buildContextualFeatures(baseFeatures, goal, protocol, violationType) {
341
+ const features = [...baseFeatures];
342
+ // One-hot goal encoding (8 dims)
343
+ for (const g of KNOWN_GOALS) {
344
+ features.push(goal.toLowerCase().includes(g.toLowerCase()) ? 1.0 : 0.0);
345
+ }
346
+ // One-hot protocol encoding (4 dims)
347
+ for (const p of KNOWN_PROTOCOLS) {
348
+ features.push(protocol === p ? 1.0 : 0.0);
349
+ }
350
+ // One-hot violation encoding (3 dims)
351
+ for (const v of KNOWN_VIOLATIONS) {
352
+ features.push(violationType === v ? 1.0 : 0.0);
353
+ }
354
+ return features;
355
+ }
356
+ /** ContextualRewardModel: logistic regression on 22-d features. */
357
+ class ContextualRewardModel {
358
+ constructor(weights, bias) {
359
+ this.weights = weights || new Array(ContextualRewardModel.FEATURE_DIM).fill(0);
360
+ this.bias = bias || 0;
361
+ this.trained = weights !== undefined;
362
+ }
363
+ get isTrained() { return this.trained; }
364
+ score(features) {
365
+ return sigmoid(dot(this.weights, features) + this.bias);
366
+ }
367
+ static train(examples, learningRate = 0.01, epochs = 100) {
368
+ if (examples.length < 20)
369
+ return new ContextualRewardModel();
370
+ const dim = ContextualRewardModel.FEATURE_DIM;
371
+ let w = new Array(dim).fill(0).map(() => (Math.random() - 0.5) * 0.1);
372
+ let b = 0.0;
373
+ for (let epoch = 0; epoch < epochs; epoch++) {
374
+ const shuffled = [...examples].sort(() => Math.random() - 0.5);
375
+ for (const ex of shuffled) {
376
+ // Build contextual features
377
+ const ctx = buildContextualFeatures(ex.features, ex.goal, ex.protocol, ex.violationType);
378
+ const z = dot(w, ctx) + b;
379
+ const p = sigmoid(z);
380
+ const error = p - (ex.success ? 1 : 0);
381
+ for (let i = 0; i < dim; i++) {
382
+ w[i] -= learningRate * error * ctx[i];
383
+ }
384
+ b -= learningRate * error;
385
+ }
386
+ }
387
+ return new ContextualRewardModel(w, b);
388
+ }
389
+ featureImportance() {
390
+ const baseNames = ["safety", "histSR", "actCount", "latCost", "audit", "accept", "execOk"];
391
+ const allNames = [...baseNames, ...KNOWN_GOALS.map(g => `goal:${g}`), ...KNOWN_PROTOCOLS.map(p => `proto:${p}`), ...KNOWN_VIOLATIONS.map(v => `viol:${v}`)];
392
+ const absW = this.weights.map(Math.abs);
393
+ const total = absW.reduce((s, v) => s + v, 1);
394
+ return allNames.map((name, i) => ({ name, weight: this.weights[i], importance: absW[i] / total }))
395
+ .sort((a, b) => b.importance - a.importance);
396
+ }
397
+ }
398
+ exports.ContextualRewardModel = ContextualRewardModel;
399
+ ContextualRewardModel.FEATURE_DIM = 22;
400
+ // ═══════════════════════════════════════════════════════════════
401
+ // Full Pipeline
402
+ // ═══════════════════════════════════════════════════════════════
403
+ function printRewardSystemReport(datasetSize, pairwiseSamples, rankingMetrics) {
404
+ console.log("\n╔════════════════════════════════════════════════════╗");
405
+ console.log("║ P4 Reward System Report ║");
406
+ console.log("╚════════════════════════════════════════════════════╝\n");
407
+ console.log(` Reward Dataset: ${datasetSize} examples`);
408
+ console.log(` Pairwise Samples: ${pairwiseSamples} pairs`);
409
+ console.log();
410
+ printRankingMetrics(rankingMetrics);
411
+ }
@@ -0,0 +1,175 @@
1
+ "use strict";
2
+ /**
3
+ * P4.1-4.4: Reward System Tests
4
+ */
5
+ var __createBinding = (this && this.__createBinding) || (Object.create ? (function(o, m, k, k2) {
6
+ if (k2 === undefined) k2 = k;
7
+ var desc = Object.getOwnPropertyDescriptor(m, k);
8
+ if (!desc || ("get" in desc ? !m.__esModule : desc.writable || desc.configurable)) {
9
+ desc = { enumerable: true, get: function() { return m[k]; } };
10
+ }
11
+ Object.defineProperty(o, k2, desc);
12
+ }) : (function(o, m, k, k2) {
13
+ if (k2 === undefined) k2 = k;
14
+ o[k2] = m[k];
15
+ }));
16
+ var __setModuleDefault = (this && this.__setModuleDefault) || (Object.create ? (function(o, v) {
17
+ Object.defineProperty(o, "default", { enumerable: true, value: v });
18
+ }) : function(o, v) {
19
+ o["default"] = v;
20
+ });
21
+ var __importStar = (this && this.__importStar) || (function () {
22
+ var ownKeys = function(o) {
23
+ ownKeys = Object.getOwnPropertyNames || function (o) {
24
+ var ar = [];
25
+ for (var k in o) if (Object.prototype.hasOwnProperty.call(o, k)) ar[ar.length] = k;
26
+ return ar;
27
+ };
28
+ return ownKeys(o);
29
+ };
30
+ return function (mod) {
31
+ if (mod && mod.__esModule) return mod;
32
+ var result = {};
33
+ if (mod != null) for (var k = ownKeys(mod), i = 0; i < k.length; i++) if (k[i] !== "default") __createBinding(result, mod, k[i]);
34
+ __setModuleDefault(result, mod);
35
+ return result;
36
+ };
37
+ })();
38
+ Object.defineProperty(exports, "__esModule", { value: true });
39
+ const vitest_1 = require("vitest");
40
+ const fs = __importStar(require("fs"));
41
+ const path = __importStar(require("path"));
42
+ const reward_system_1 = require("./reward-system");
43
+ const planner_telemetry_1 = require("./planner-telemetry");
44
+ const REW_DIR = path.resolve(__dirname, "..", "test-reward-system");
45
+ process.env.PROGMUNE_PROJECT_DIR = REW_DIR;
46
+ fs.mkdirSync(REW_DIR, { recursive: true });
47
+ fs.mkdirSync(path.join(REW_DIR, ".progmune_corpus", "telemetry"), { recursive: true });
48
+ function seedTelemetry(n) {
49
+ const t = new planner_telemetry_1.PlannerTelemetry(path.join(REW_DIR, ".progmune_corpus", "telemetry", `reward-${Date.now()}.jsonl`));
50
+ for (let i = 0; i < n; i++) {
51
+ const safe = i % 3 !== 0;
52
+ const actions = safe ? ["open_file", "write_file", "close_file"] : ["open_file", "write_file"];
53
+ const fp = (0, planner_telemetry_1.candidateFingerprint)("FileProtocol", actions, "resource_leak");
54
+ const id = t.recordDecision({
55
+ goal: safe ? "safely write config" : "quick write",
56
+ protocol: "FileProtocol", violationType: "resource_leak",
57
+ candidates: [{ candidateId: fp, source: "protocol", evidenceSources: ["protocol"], actions, explanation: safe ? "safe" : "quick" }],
58
+ selectedCandidateId: fp,
59
+ });
60
+ const accepted = safe ? Math.random() < 0.9 : Math.random() < 0.2;
61
+ t.recordFeedback(id, {
62
+ decision: accepted ? "accepted" : "rejected",
63
+ executionResult: accepted ? { success: true, violations: [] } : { success: false, violations: ["resource_leak"] },
64
+ timestamp: Date.now(),
65
+ });
66
+ }
67
+ return t;
68
+ }
69
+ (0, vitest_1.describe)("P4.4 Reward Dataset", () => {
70
+ (0, vitest_1.it)("builds and persists reward dataset", () => {
71
+ const telemetry = seedTelemetry(100);
72
+ const examples = (0, reward_system_1.buildRewardDataset)(telemetry);
73
+ (0, vitest_1.expect)(examples.length).toBeGreaterThanOrEqual(50);
74
+ for (const e of examples) {
75
+ (0, vitest_1.expect)(e.features.length).toBe(7);
76
+ (0, vitest_1.expect)(typeof e.accepted).toBe("boolean");
77
+ }
78
+ const fp = (0, reward_system_1.saveRewardDataset)(examples, path.join(REW_DIR, "reward_dataset"));
79
+ (0, vitest_1.expect)(fs.existsSync(fp)).toBe(true);
80
+ const loaded = (0, reward_system_1.loadRewardDataset)(path.join(REW_DIR, "reward_dataset"));
81
+ (0, vitest_1.expect)(loaded.length).toBe(examples.length);
82
+ });
83
+ });
84
+ (0, vitest_1.describe)("P4.1 PairwiseRewardModel", () => {
85
+ (0, vitest_1.it)("trains on pairwise samples", () => {
86
+ const samples = [];
87
+ for (let i = 0; i < 100; i++) {
88
+ samples.push({
89
+ winnerFeatures: [1.0, 0.8, 0.3, 0.4, 0.75, 0.85, 0.9],
90
+ loserFeatures: [0.3, 0.3, 0.6, 0.7, 0.3, 0.15, 0.1],
91
+ goal: "safely write", protocol: "FileProtocol",
92
+ });
93
+ }
94
+ const model = reward_system_1.PairwiseRewardModel.train(samples);
95
+ (0, vitest_1.expect)(model.isTrained).toBe(true);
96
+ (0, vitest_1.expect)(model.sampleCount).toBe(100);
97
+ // Safe should beat leaky
98
+ const prob = model.predictPair([1.0, 0.8, 0.3, 0.4, 0.75, 0.85, 0.9], [0.3, 0.3, 0.6, 0.7, 0.3, 0.15, 0.1]);
99
+ (0, vitest_1.expect)(prob).toBeGreaterThan(0.5);
100
+ });
101
+ });
102
+ (0, vitest_1.describe)("P4.2 Off-Policy Evaluator++", () => {
103
+ (0, vitest_1.it)("computes NDCG correctly", () => {
104
+ const perfect = [1, 1, 1, 0, 0];
105
+ (0, vitest_1.expect)((0, reward_system_1.computeNDCG)(perfect)).toBe(1.0);
106
+ const terrible = [0, 0, 0, 0, 1];
107
+ (0, vitest_1.expect)((0, reward_system_1.computeNDCG)(terrible)).toBeLessThan(0.5);
108
+ const mixed = [1, 0, 1, 0, 0];
109
+ const ndcg = (0, reward_system_1.computeNDCG)(mixed);
110
+ (0, vitest_1.expect)(ndcg).toBeGreaterThan(0.5);
111
+ (0, vitest_1.expect)(ndcg).toBeLessThan(1.0);
112
+ });
113
+ (0, vitest_1.it)("compares rankers with lift metrics", () => {
114
+ const decisions = Array.from({ length: 20 }, () => {
115
+ const safe = Math.random() < 0.7;
116
+ return {
117
+ candidates: [
118
+ { features: safe ? [1.0, 0.8, 0.3, 0.4, 0.75, 0.85, 0.9] : [0.3, 0.3, 0.6, 0.5, 0.3, 0.15, 0.1], accepted: safe },
119
+ { features: safe ? [0.3, 0.3, 0.6, 0.5, 0.3, 0.15, 0.1] : [1.0, 0.8, 0.3, 0.4, 0.75, 0.85, 0.9], accepted: !safe },
120
+ ],
121
+ userChoseIndex: 0,
122
+ };
123
+ });
124
+ // Old ranker: score = sum(features) — puts all weight on first feature magnitude
125
+ const oldRanker = (f) => f.reduce((a, b) => a + b, 0);
126
+ // New ranker: score = 2*acceptance + 2*execution — better aligned with truth
127
+ const newRanker = (f) => f[5] * 2 + f[6] * 2;
128
+ const metrics = (0, reward_system_1.compareRankersOffPolicy)(decisions, oldRanker, newRanker);
129
+ (0, vitest_1.expect)(metrics.ndcg).toBeGreaterThan(0);
130
+ (0, vitest_1.expect)(metrics.acceptanceLift).toBeGreaterThanOrEqual(-1);
131
+ (0, vitest_1.expect)(metrics.acceptanceLift).toBeLessThanOrEqual(1);
132
+ (0, reward_system_1.printRankingMetrics)(metrics);
133
+ const gate = (0, reward_system_1.deploymentGate)(metrics);
134
+ console.log(` Deployment gate: ${gate.passed ? "PASS" : "FAIL"} — ${gate.reason}`);
135
+ });
136
+ (0, vitest_1.it)("deployment gate rejects negative lift", () => {
137
+ const badMetrics = { ndcg: 0.45, top1Lift: 0.1, top3Lift: 0.05, acceptanceLift: -0.02 };
138
+ (0, vitest_1.expect)((0, reward_system_1.deploymentGate)(badMetrics).passed).toBe(false);
139
+ });
140
+ (0, vitest_1.it)("deployment gate approves positive lift", () => {
141
+ const goodMetrics = { ndcg: 0.55, top1Lift: 0.15, top3Lift: 0.1, acceptanceLift: 0.05 };
142
+ (0, vitest_1.expect)((0, reward_system_1.deploymentGate)(goodMetrics).passed).toBe(true);
143
+ });
144
+ });
145
+ (0, vitest_1.describe)("P4.3 ContextualRewardModel", () => {
146
+ (0, vitest_1.it)("builds 22-d contextual features", () => {
147
+ const base = [0.8, 0.5, 0.3, 0.4, 0.7, 0.85, 0.9];
148
+ const ctx = (0, reward_system_1.buildContextualFeatures)(base, "safely write config", "FileProtocol", "resource_leak");
149
+ (0, vitest_1.expect)(ctx.length).toBe(22); // 7 + 8 + 4 + 3
150
+ // "safely write" goal bit should be 1
151
+ (0, vitest_1.expect)(ctx[7]).toBe(1.0); // first goal feature
152
+ // "FileProtocol" protocol bit should be 1
153
+ (0, vitest_1.expect)(ctx[7 + 8]).toBe(1.0);
154
+ // "resource_leak" violation bit should be 1
155
+ (0, vitest_1.expect)(ctx[7 + 8 + 4]).toBe(1.0);
156
+ });
157
+ (0, vitest_1.it)("trains on reward examples", () => {
158
+ const telemetry = seedTelemetry(200);
159
+ const examples = (0, reward_system_1.buildRewardDataset)(telemetry);
160
+ const model = reward_system_1.ContextualRewardModel.train(examples);
161
+ (0, vitest_1.expect)(model.isTrained).toBe(true);
162
+ // Safe repair should score higher
163
+ const safeScore = model.score((0, reward_system_1.buildContextualFeatures)([0.8, 0.5, 0.3, 0.4, 0.7, 0.85, 0.9], "safely write config", "FileProtocol", "resource_leak"));
164
+ (0, vitest_1.expect)(safeScore).toBeGreaterThan(0.5);
165
+ const imp = model.featureImportance();
166
+ (0, vitest_1.expect)(imp.length).toBe(22);
167
+ (0, vitest_1.expect)(imp[0].importance).toBeGreaterThan(0);
168
+ });
169
+ });
170
+ (0, vitest_1.describe)("Reward System Report", () => {
171
+ (0, vitest_1.it)("prints full system report", () => {
172
+ const metrics = { ndcg: 0.62, top1Lift: 0.12, top3Lift: 0.08, acceptanceLift: 0.04 };
173
+ (0, reward_system_1.printRewardSystemReport)(250, 100, metrics);
174
+ });
175
+ });