chainlesschain 0.166.71 → 0.166.74

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (164) hide show
  1. package/README.md +26 -4
  2. package/package.json +2 -1
  3. package/scripts/pm-exploration-effect.mjs +188 -0
  4. package/src/assets/web-panel/assets/{AIOps-6n6RPfD2.js → AIOps-BlIvP3k1.js} +1 -1
  5. package/src/assets/web-panel/assets/{ActionButton-BRlz1g77.js → ActionButton-DYF7atSV.js} +1 -1
  6. package/src/assets/web-panel/assets/{Analytics-BFx6_nNR.js → Analytics-LZYaYek6.js} +3 -3
  7. package/src/assets/web-panel/assets/{AppLayout-CcbKe_m4.js → AppLayout-BJxrjNbj.js} +5 -5
  8. package/src/assets/web-panel/assets/{Artifacts-DlBMxBaf.js → Artifacts-gEDhxhmE.js} +1 -1
  9. package/src/assets/web-panel/assets/{Audit-aqQRi1IW.js → Audit-B0MR950Q.js} +1 -1
  10. package/src/assets/web-panel/assets/{BackgroundAgents-tzAQeiAS.js → BackgroundAgents-3e6Os_Kl.js} +1 -1
  11. package/src/assets/web-panel/assets/{Backup-C3sLPZvj.js → Backup-DB_X8KUg.js} +1 -1
  12. package/src/assets/web-panel/assets/{BaseInput-BdqG7xax.js → BaseInput-DTNcvNeG.js} +1 -1
  13. package/src/assets/web-panel/assets/{Chat-DXjdkR-r.js → Chat-SnqdA62j.js} +6 -6
  14. package/src/assets/web-panel/assets/{ChatBubbleRenderer-B-3SVP-4.js → ChatBubbleRenderer-BaiWNH_V.js} +1 -1
  15. package/src/assets/web-panel/assets/{Checkbox-CACOSDhS.js → Checkbox-Dj0O4dyP.js} +1 -1
  16. package/src/assets/web-panel/assets/{Codegen-ZY0zbHvS.js → Codegen-BeelW3dx.js} +1 -1
  17. package/src/assets/web-panel/assets/{Col-C511kzpa.js → Col-BmstiyBM.js} +1 -1
  18. package/src/assets/web-panel/assets/{Community-CIM5EtmR.js → Community-CrbmFk3f.js} +1 -1
  19. package/src/assets/web-panel/assets/{Compact-DpTxu8E7.js → Compact-DCg5ZPIw.js} +1 -1
  20. package/src/assets/web-panel/assets/{Compliance-Cu42ksf-.js → Compliance-B3sajhq7.js} +1 -1
  21. package/src/assets/web-panel/assets/{Cowork-BqUbCW9U.js → Cowork-DWcdvrDK.js} +4 -4
  22. package/src/assets/web-panel/assets/{Cron-OXYE4gRs.js → Cron-Bjwp_NDf.js} +2 -2
  23. package/src/assets/web-panel/assets/{Crosschain-DPqmhSi6.js → Crosschain-BXTLC9o8.js} +1 -1
  24. package/src/assets/web-panel/assets/{DID-CPp_yj85.js → DID-Di62M46b.js} +2 -2
  25. package/src/assets/web-panel/assets/{Dashboard-pki3T5PE.js → Dashboard-BQED8Rkt.js} +2 -2
  26. package/src/assets/web-panel/assets/{Dropdown-IyL1RsjP.js → Dropdown-DEf9VYaj.js} +1 -1
  27. package/src/assets/web-panel/assets/{EmailListRenderer-6Kbwb4Ux.js → EmailListRenderer-DhFpPMAM.js} +1 -1
  28. package/src/assets/web-panel/assets/{EvolutionSettings-DGBMs24I.js → EvolutionSettings-BOqmzkpS.js} +1 -1
  29. package/src/assets/web-panel/assets/{FamilyGuardDashboard-CLz3KJvw.js → FamilyGuardDashboard-BxBnJ4fe.js} +1 -1
  30. package/src/assets/web-panel/assets/{Federation-gMDCbNjS.js → Federation-BN9Q8nsZ.js} +1 -1
  31. package/src/assets/web-panel/assets/{FormItemContext-bSzQnHE6.js → FormItemContext-CGeqFEd-.js} +1 -1
  32. package/src/assets/web-panel/assets/{GenericCardRenderer-DCyBnm6B.js → GenericCardRenderer-DElmWQTc.js} +1 -1
  33. package/src/assets/web-panel/assets/{Git-BKqZtZMD.js → Git-BHBSgKck.js} +2 -2
  34. package/src/assets/web-panel/assets/{Governance-nZDU4Gz2.js → Governance-Breo3NTU.js} +1 -1
  35. package/src/assets/web-panel/assets/{Inference-VodzHgdz.js → Inference-CCABy4E0.js} +1 -1
  36. package/src/assets/web-panel/assets/{KnowledgeGraph-DHjlnyh4.js → KnowledgeGraph-B6_ryn-X.js} +1 -1
  37. package/src/assets/web-panel/assets/{Logs-CPsAssGM.js → Logs-CkmJjJok.js} +2 -2
  38. package/src/assets/web-panel/assets/{MarkdownRenderer-ChogLgLv.js → MarkdownRenderer-DQEQBaRu.js} +1 -1
  39. package/src/assets/web-panel/assets/{Marketplace-Df7Cr8cW.js → Marketplace-iip65jxv.js} +1 -1
  40. package/src/assets/web-panel/assets/{McpTools-Dy_V6M2q.js → McpTools-DDwp_QLb.js} +2 -2
  41. package/src/assets/web-panel/assets/{Memory-DjYWvH5O.js → Memory-2ZmNDRkC.js} +2 -2
  42. package/src/assets/web-panel/assets/{MobileBridge-qYw4QwQh.js → MobileBridge-C6HilwZl.js} +1 -1
  43. package/src/assets/web-panel/assets/MobileProjects-BeMtYRhx.js +1 -0
  44. package/src/assets/web-panel/assets/{Mtc-BQ2fg2aQ.js → Mtc-Bp-FEcoo.js} +5 -5
  45. package/src/assets/web-panel/assets/{MtcAudit-CaBVA8Cu.js → MtcAudit-DNdkTCe5.js} +2 -2
  46. package/src/assets/web-panel/assets/{Multisig-DsTt5nuM.js → Multisig-PnKg2_Eo.js} +3 -3
  47. package/src/assets/web-panel/assets/{NLProgramming-CqpsGgiz.js → NLProgramming-DDp4lZeG.js} +1 -1
  48. package/src/assets/web-panel/assets/{Notes-DpRTGRs9.js → Notes-DEjFktcO.js} +4 -4
  49. package/src/assets/web-panel/assets/{NotificationSettings-CmHeyjUS.js → NotificationSettings-FR8fJDAn.js} +1 -1
  50. package/src/assets/web-panel/assets/{OrderTableRenderer-D-JBdzGf.js → OrderTableRenderer-Cyc-XI0L.js} +1 -1
  51. package/src/assets/web-panel/assets/{Organization-QL01Gu7d.js → Organization-DNW2E_JU.js} +4 -4
  52. package/src/assets/web-panel/assets/{Overflow-DjUeTZ69.js → Overflow-BHfd6wk6.js} +1 -1
  53. package/src/assets/web-panel/assets/{P2P-DuM6UaEh.js → P2P-COIkaXH4.js} +2 -2
  54. package/src/assets/web-panel/assets/{PdhVaultBrowser-BsL5CbMs.js → PdhVaultBrowser-DuRP0UeX.js} +3 -3
  55. package/src/assets/web-panel/assets/{Permissions-Cfu4vMf9.js → Permissions-BfVRGMy8.js} +4 -4
  56. package/src/assets/web-panel/assets/{PersonalDataHub-HvEGAMl5.js → PersonalDataHub-BWYjO6Aa.js} +5 -5
  57. package/src/assets/web-panel/assets/{Pipeline-_CZIYXJs.js → Pipeline-C0e0fPBW.js} +1 -1
  58. package/src/assets/web-panel/assets/{Privacy-C6bTNjwi.js → Privacy-CnId4Jv-.js} +1 -1
  59. package/src/assets/web-panel/assets/{ProjectInit-D64UJRfZ.js → ProjectInit-CYf1grXg.js} +2 -2
  60. package/src/assets/web-panel/assets/{ProjectSettings-CSTAzKVe.js → ProjectSettings-HsXgJdiH.js} +2 -2
  61. package/src/assets/web-panel/assets/{Projects-BudNMgcm.js → Projects-Bi0r9QcC.js} +1 -1
  62. package/src/assets/web-panel/assets/{Providers-C7ddN75H.js → Providers-B-Tu_-_4.js} +1 -1
  63. package/src/assets/web-panel/assets/{QrScannerModal-CtsMC0eH.js → QrScannerModal-CPDqYT2F.js} +1 -1
  64. package/src/assets/web-panel/assets/{QuickAsk-iE7xKI4P.js → QuickAsk-Df6qcjXy.js} +1 -1
  65. package/src/assets/web-panel/assets/{Recommend-5FCqoy3g.js → Recommend-dP6sfGSY.js} +1 -1
  66. package/src/assets/web-panel/assets/{RemoteSession-DtX0VARB.js → RemoteSession-VAt9-tM3.js} +2 -2
  67. package/src/assets/web-panel/assets/{Reputation-Hd4dH5ci.js → Reputation-DtRVRiLI.js} +1 -1
  68. package/src/assets/web-panel/assets/{Row-CnLRjeGC.js → Row-B6Ukqvli.js} +1 -1
  69. package/src/assets/web-panel/assets/{RssFeed-Ds43ubF9.js → RssFeed-DfVRQMrZ.js} +3 -3
  70. package/src/assets/web-panel/assets/{Search-BmXOdvpt.js → Search-CyncAiqQ.js} +1 -1
  71. package/src/assets/web-panel/assets/{Security--yQqv9yt.js → Security-DEO7nrkb.js} +4 -4
  72. package/src/assets/web-panel/assets/{Services-CNNPxlyA.js → Services-CkusQEhN.js} +2 -2
  73. package/src/assets/web-panel/assets/{Skeleton-DPmZAdDN.js → Skeleton-CbGtO3kc.js} +1 -1
  74. package/src/assets/web-panel/assets/{Skills-Va_WxW7s.js → Skills-diOh0gAr.js} +1 -1
  75. package/src/assets/web-panel/assets/{Sla-BZoRsy3v.js → Sla-CIu9Ahak.js} +1 -1
  76. package/src/assets/web-panel/assets/{SpeechSettings-XgRV95VE.js → SpeechSettings-_Nzyi4K9.js} +1 -1
  77. package/src/assets/web-panel/assets/{SyncSettings-Dr7NXCEw.js → SyncSettings-BAOLFlRq.js} +2 -2
  78. package/src/assets/web-panel/assets/{Tasks-j-xRo5_-.js → Tasks-BpmvL1Eq.js} +1 -1
  79. package/src/assets/web-panel/assets/{Templates-BJSqGQuB.js → Templates-Dkr_vRdJ.js} +1 -1
  80. package/src/assets/web-panel/assets/{Tenant-zuVl_JY7.js → Tenant-kM7lHUFR.js} +1 -1
  81. package/src/assets/web-panel/assets/{Terminal-B-xNbgfG.js → Terminal-9zfc41aU.js} +2 -2
  82. package/src/assets/web-panel/assets/{TimelineRenderer-BtZQpa0b.js → TimelineRenderer-Cg_7j6cj.js} +1 -1
  83. package/src/assets/web-panel/assets/{Tokens-CXGqGXWd.js → Tokens-Cs1aikaM.js} +1 -1
  84. package/src/assets/web-panel/assets/{Trigger-CtDmejyH.js → Trigger-DKYPJ9gl.js} +1 -1
  85. package/src/assets/web-panel/assets/{Trust-BfDvBu2B.js → Trust-BFy1Sm8B.js} +1 -1
  86. package/src/assets/web-panel/assets/{UkeySign-CgHq72xW.js → UkeySign-B31L9NBH.js} +1 -1
  87. package/src/assets/web-panel/assets/{VideoEditing-pRBMqiGT.js → VideoEditing-OMmUYdIC.js} +1 -1
  88. package/src/assets/web-panel/assets/{Wallet-DHhZSHFo.js → Wallet-y3aE7oru.js} +4 -4
  89. package/src/assets/web-panel/assets/{WebAuthn-C-PDGO1-.js → WebAuthn-C0ICKvpu.js} +4 -4
  90. package/src/assets/web-panel/assets/{WorkflowEditor-Dqrh4ooE.js → WorkflowEditor-Cq5UYbr5.js} +1 -1
  91. package/src/assets/web-panel/assets/{chat-Du4rhkgo.js → chat-B1ijaKFv.js} +1 -1
  92. package/src/assets/web-panel/assets/{colors-CVVon0Yw.js → colors-CHkI8PDg.js} +1 -1
  93. package/src/assets/web-panel/assets/{compact-item-zcQv_R_9.js → compact-item-1BYvgWEs.js} +1 -1
  94. package/src/assets/web-panel/assets/{createContext-UfW2PpZM.js → createContext-BGti56u4.js} +1 -1
  95. package/src/assets/web-panel/assets/devWarning-Isft1nEX.js +1 -0
  96. package/src/assets/web-panel/assets/{hasIn-Dko6kq4D.js → hasIn-Cr87Fph1.js} +1 -1
  97. package/src/assets/web-panel/assets/{index-DMdBtncl.js → index-1WxCzqYq.js} +1 -1
  98. package/src/assets/web-panel/assets/{index-DwXm_6Df.js → index-6Ta8Pg47.js} +1 -1
  99. package/src/assets/web-panel/assets/{index-Dgu3vko5.js → index-B1yur8MU.js} +1 -1
  100. package/src/assets/web-panel/assets/{index-jjgfplWj.js → index-B4BSM3BK.js} +1 -1
  101. package/src/assets/web-panel/assets/{index-ija7mc1b.js → index-B8Jb_cGt.js} +1 -1
  102. package/src/assets/web-panel/assets/{index-Cs4qAz6z.js → index-BA0FMuDX.js} +1 -1
  103. package/src/assets/web-panel/assets/{index-BwcE8uKT.js → index-BKb16YKp.js} +1 -1
  104. package/src/assets/web-panel/assets/{index-5UfwyzQV.js → index-BOjTRSHE.js} +3 -3
  105. package/src/assets/web-panel/assets/{index-BPE8w-x1.js → index-BYpqBWlt.js} +1 -1
  106. package/src/assets/web-panel/assets/index-B_T9NUk7.js +1 -0
  107. package/src/assets/web-panel/assets/{index-DMxhURG4.js → index-Bdkrut2A.js} +1 -1
  108. package/src/assets/web-panel/assets/{index-CJ72HeXg.js → index-BjgtiMdY.js} +1 -1
  109. package/src/assets/web-panel/assets/{index-BoQJIWkq.js → index-BkFXPrdx.js} +1 -1
  110. package/src/assets/web-panel/assets/{index-BolCTH1Q.js → index-Bni5GodE.js} +1 -1
  111. package/src/assets/web-panel/assets/{index-DqQhty7h.js → index-CDTNnQDz.js} +1 -1
  112. package/src/assets/web-panel/assets/{index-BEOOJevs.js → index-CQWMHlY7.js} +1 -1
  113. package/src/assets/web-panel/assets/{index-CoXowCMA.js → index-CS0MIE2C.js} +1 -1
  114. package/src/assets/web-panel/assets/{index-7aNfa1q3.js → index-C_lEztok.js} +1 -1
  115. package/src/assets/web-panel/assets/{index-DhYE_jVw.js → index-CbkGda_k.js} +1 -1
  116. package/src/assets/web-panel/assets/{index-CV8kORgq.js → index-CevUJTXd.js} +1 -1
  117. package/src/assets/web-panel/assets/{index-DIFh_pNy.js → index-CjBkVrq9.js} +1 -1
  118. package/src/assets/web-panel/assets/{index-CrURt95p.js → index-CkcWhgLB.js} +1 -1
  119. package/src/assets/web-panel/assets/{index-BXqsNksk.js → index-CmmSJBNi.js} +1 -1
  120. package/src/assets/web-panel/assets/{index-Dkc1L9vQ.js → index-CvcpesEZ.js} +1 -1
  121. package/src/assets/web-panel/assets/{index-Dviw8PsD.js → index-Cvu95WPI.js} +1 -1
  122. package/src/assets/web-panel/assets/{index-Dh1USRx7.js → index-D6g4jxdE.js} +1 -1
  123. package/src/assets/web-panel/assets/index-DBqUxaat.js +1 -0
  124. package/src/assets/web-panel/assets/{index-BR5eNU_w.js → index-DCkvhzaY.js} +1 -1
  125. package/src/assets/web-panel/assets/{index-CPqtDZE1.js → index-DHJtAEx8.js} +1 -1
  126. package/src/assets/web-panel/assets/{index-0d3V2fKV.js → index-DJjnXDAD.js} +1 -1
  127. package/src/assets/web-panel/assets/{index-4YoVVwaY.js → index-Dam-LB_J.js} +1 -1
  128. package/src/assets/web-panel/assets/{index-BHq5RTAk.js → index-DcoI3ENn.js} +1 -1
  129. package/src/assets/web-panel/assets/{index-CrbxgYRx.js → index-Dnp-4yPc.js} +1 -1
  130. package/src/assets/web-panel/assets/{index-ubMuGX5r.js → index-Duv0TfJa.js} +1 -1
  131. package/src/assets/web-panel/assets/{index-B8nu6_LV.js → index-DzevrmLK.js} +1 -1
  132. package/src/assets/web-panel/assets/{index-UzXoeYbz.js → index-fhpMyq0u.js} +1 -1
  133. package/src/assets/web-panel/assets/{index-BPTQ-SUp.js → index-rO6VKJv5.js} +1 -1
  134. package/src/assets/web-panel/assets/{index-D4-E1s34.js → index-rVnXNFST.js} +1 -1
  135. package/src/assets/web-panel/assets/{index-DKAB6MRs.js → index-tY479EoQ.js} +1 -1
  136. package/src/assets/web-panel/assets/{initDefaultProps-DvSMOW2m.js → initDefaultProps-DAGtnwoX.js} +1 -1
  137. package/src/assets/web-panel/assets/{motion-C7PjuFda.js → motion-02FXBi5C.js} +1 -1
  138. package/src/assets/web-panel/assets/{move-qsgDmMqu.js → move-BLf7rhzJ.js} +1 -1
  139. package/src/assets/web-panel/assets/{mtc-parser-D7qHulCC.js → mtc-parser-Ct6i1vqQ.js} +1 -1
  140. package/src/assets/web-panel/assets/{omit-g_tGmcIM.js → omit-DKvTU8IT.js} +1 -1
  141. package/src/assets/web-panel/assets/{pickAttrs-DGs-br4p.js → pickAttrs-DqDsKoMe.js} +1 -1
  142. package/src/assets/web-panel/assets/{placementArrow-A1KZT5yH.js → placementArrow-kae6Cnwx.js} +1 -1
  143. package/src/assets/web-panel/assets/{responsiveObserve-B98U6Yi5.js → responsiveObserve-DzFM_TQf.js} +1 -1
  144. package/src/assets/web-panel/assets/{slide-Y5Q6ixV9.js → slide-qDjNFPuM.js} +1 -1
  145. package/src/assets/web-panel/assets/{statusUtils-CVC7D579.js → statusUtils-BihSr7G3.js} +1 -1
  146. package/src/assets/web-panel/assets/{styleChecker-CDWAgZKa.js → styleChecker-BzoT0TfJ.js} +1 -1
  147. package/src/assets/web-panel/assets/{useFlexGapSupport-C0UFofSm.js → useFlexGapSupport-CjXXcU1j.js} +1 -1
  148. package/src/assets/web-panel/assets/{useFs-BHEf6SSo.js → useFs-DrWcbDJ1.js} +1 -1
  149. package/src/assets/web-panel/assets/{usePersonalDataHub-Bb2M_BPj.js → usePersonalDataHub-DO1oBNTE.js} +1 -1
  150. package/src/assets/web-panel/assets/{vnode-DENyLaXR.js → vnode-BR6crWko.js} +1 -1
  151. package/src/assets/web-panel/assets/{zoom-BUOUP8-I.js → zoom-DmKLfw6W.js} +1 -1
  152. package/src/assets/web-panel/index.html +1 -1
  153. package/src/data/changelog.json +39 -0
  154. package/src/lib/decision-layer/benchmark.js +64 -3
  155. package/src/lib/decision-layer/runtime.js +18 -5
  156. package/src/lib/evolution/evolution-eval-gate.js +289 -32
  157. package/src/lib/evolution/pm-exploration-benchmark.js +2014 -49
  158. package/src/lib/evolution/skill-writer-inventory-manifest.js +4 -1
  159. package/src/lib/task-progress-tracker.js +47 -4
  160. package/src/runtime/agent-core.js +18 -7
  161. package/src/assets/web-panel/assets/MobileProjects-DDA7pzJd.js +0 -1
  162. package/src/assets/web-panel/assets/devWarning-CoA1apwL.js +0 -1
  163. package/src/assets/web-panel/assets/index-CsR4S29Z.js +0 -1
  164. package/src/assets/web-panel/assets/index-DvcAyRFh.js +0 -1
@@ -3,11 +3,15 @@ import { createHash } from "node:crypto";
3
3
  import { isProxy } from "node:util/types";
4
4
  import {
5
5
  buildEvolutionEvalSuite,
6
+ computeEvolutionEvalContextDigest,
6
7
  computeEvolutionEvalTrainingPartitionDigest,
7
8
  verifyEvolutionEvalPolicy,
9
+ verifyEvolutionEvalReceipt,
10
+ verifyEvolutionEvalResultEvidence,
8
11
  verifyEvolutionEvalSuite,
9
12
  } from "./evolution-eval-gate.js";
10
13
  import { createPmExplorationPlan } from "./pm-exploration-rounds.js";
14
+ import { capturePmExplorationProviderSettlementStore } from "./pm-exploration-provider-settlement-adapter.js";
11
15
  import grader from "./pm-result-grader.cjs";
12
16
 
13
17
  const GROUPS = ["template", "project", "principal", "timeWindow"];
@@ -43,11 +47,58 @@ const FAILURE_CLASSES = new Set([
43
47
  "grader",
44
48
  "unknown",
45
49
  ]);
50
+ const COMPLETION_FAILURE_CLASSES = new Set([
51
+ ...FAILURE_CLASSES,
52
+ "setup",
53
+ "timeout",
54
+ "cancelled",
55
+ ]);
46
56
 
47
57
  export const PM_EXPLORATION_EFFECT_PLAN_SCHEMA =
48
- "chainlesschain.pm-exploration-effect-plan/v1";
58
+ "chainlesschain.pm-exploration-effect-plan/v2";
49
59
  export const PM_EXPLORATION_EFFECT_REPORT_SCHEMA =
60
+ "chainlesschain.pm-exploration-effect-report/v2";
61
+ export const PM_EXPLORATION_EFFECT_EVIDENCE_REPORT_SCHEMA =
62
+ "chainlesschain.pm-exploration-effect-evidence-report/v1";
63
+ export const PM_EXPLORATION_PREPARATION_PROVIDER_EVIDENCE_SCHEMA =
64
+ "chainlesschain.pm-exploration-preparation-provider-evidence/v1";
65
+ export const PM_EXPLORATION_EFFECT_PROVIDER_EVIDENCE_BUNDLE_SCHEMA =
66
+ "chainlesschain.pm-exploration-effect-provider-evidence-bundle/v1";
67
+ export const PM_EXPLORATION_EFFECT_INTERRUPTED_EVIDENCE_SCHEMA =
68
+ "chainlesschain.pm-exploration-effect-interrupted-evidence/v1";
69
+ export const PM_EXPLORATION_EFFECT_RUNTIME_FAILURE_EVIDENCE_SCHEMA =
70
+ "chainlesschain.pm-exploration-effect-runtime-failure-evidence/v1";
71
+ export const PM_EXPLORATION_EFFECT_PREFLIGHT_REJECTION_EVIDENCE_SCHEMA =
72
+ "chainlesschain.pm-exploration-effect-preflight-rejection-evidence/v1";
73
+ export const PM_EXPLORATION_EFFECT_ATTEMPT_COHORT_SCHEMA =
74
+ "chainlesschain.pm-exploration-effect-attempt-cohort/v1";
75
+ export const PM_EXPLORATION_EFFECT_ATTEMPT_COHORT_V2_SCHEMA =
76
+ "chainlesschain.pm-exploration-effect-attempt-cohort/v2";
77
+ export const PM_EXPLORATION_EFFECT_ATTEMPT_COHORT_V3_SCHEMA =
78
+ "chainlesschain.pm-exploration-effect-attempt-cohort/v3";
79
+ export const PM_EXPLORATION_EFFECT_ATTEMPT_COHORT_V4_SCHEMA =
80
+ "chainlesschain.pm-exploration-effect-attempt-cohort/v4";
81
+ export const PM_EXPLORATION_EFFECT_SLOT_MANIFEST_SCHEMA =
82
+ "chainlesschain.pm-exploration-effect-slot-manifest/v1";
83
+ export const PM_EXPLORATION_EFFECT_MANIFEST_BOUND_COHORT_SCHEMA =
84
+ "chainlesschain.pm-exploration-effect-manifest-bound-cohort/v1";
85
+ export const PM_EXPLORATION_EFFECT_COHORT_USAGE_EVIDENCE_SCHEMA =
86
+ "chainlesschain.pm-exploration-effect-cohort-usage-evidence/v1";
87
+ const LEGACY_EFFECT_PLAN_SCHEMA =
88
+ "chainlesschain.pm-exploration-effect-plan/v1";
89
+ const LEGACY_EFFECT_REPORT_SCHEMA =
50
90
  "chainlesschain.pm-exploration-effect-report/v1";
91
+ const PRIMARY_METRIC = "strict-completion-rate";
92
+ const RESAMPLING_UNIT = "task-group-component";
93
+ const PREPARATION_PROVIDER_REQUESTS_SCHEMA =
94
+ "chainlesschain.pm-preparation-provider-requests/v1";
95
+ const SIGNED_EVAL_USAGE_FIELDS = Object.freeze([
96
+ "executionCount",
97
+ "totalTokens",
98
+ "totalLatencyMs",
99
+ "totalToolCalls",
100
+ "totalCostMicrounits",
101
+ ]);
51
102
 
52
103
  function exact(value, keys, label) {
53
104
  if (
@@ -214,6 +265,8 @@ function normalizeEffectBudget(value) {
214
265
  }
215
266
 
216
267
  export function verifyPmExplorationEffectPlan(value) {
268
+ assertPlainData(value, "PM effect plan");
269
+ const legacy = value?.schema === LEGACY_EFFECT_PLAN_SCHEMA;
217
270
  exact(
218
271
  value,
219
272
  [
@@ -233,7 +286,15 @@ export function verifyPmExplorationEffectPlan(value) {
233
286
  "seeds",
234
287
  "budgetPerArmPerSeed",
235
288
  "costPhases",
236
- "minimumScoreDelta",
289
+ ...(legacy
290
+ ? ["minimumScoreDelta"]
291
+ : [
292
+ "minimumPassRateDelta",
293
+ "minimumIndependentGroups",
294
+ "taskGroups",
295
+ "primaryMetric",
296
+ "resamplingUnit",
297
+ ]),
237
298
  "bootstrapSamples",
238
299
  "equalBudget",
239
300
  "promotionAuthority",
@@ -242,7 +303,7 @@ export function verifyPmExplorationEffectPlan(value) {
242
303
  "PM effect plan",
243
304
  );
244
305
  if (
245
- value.schema !== PM_EXPLORATION_EFFECT_PLAN_SCHEMA ||
306
+ (!legacy && value.schema !== PM_EXPLORATION_EFFECT_PLAN_SCHEMA) ||
246
307
  value.bootstrapSamples !== 1_000 ||
247
308
  value.equalBudget !== true ||
248
309
  value.promotionAuthority !== false
@@ -275,7 +336,14 @@ export function verifyPmExplorationEffectPlan(value) {
275
336
  seeds: normalizeSeeds(value.seeds),
276
337
  budgetPerArmPerSeed: normalizeEffectBudget(value.budgetPerArmPerSeed),
277
338
  costPhases: normalizeStringArray(value.costPhases, "costPhases"),
278
- minimumScoreDelta: finite(value.minimumScoreDelta, "minimumScoreDelta"),
339
+ ...(legacy
340
+ ? {
341
+ minimumScoreDelta: finite(
342
+ value.minimumScoreDelta,
343
+ "minimumScoreDelta",
344
+ ),
345
+ }
346
+ : normalizeCompletionPolicy(value)),
279
347
  bootstrapSamples: 1_000,
280
348
  equalBudget: true,
281
349
  promotionAuthority: false,
@@ -284,12 +352,58 @@ export function verifyPmExplorationEffectPlan(value) {
284
352
  throw new TypeError("PM effect plan cost phases are invalid");
285
353
  }
286
354
  const planDigest = digest(value.planDigest, "planDigest");
287
- if (planDigest !== hash(PM_EXPLORATION_EFFECT_PLAN_SCHEMA, core)) {
355
+ if (planDigest !== hash(value.schema, core)) {
288
356
  throw new Error("PM effect plan digest mismatch");
289
357
  }
290
358
  return deepFreeze({ ...core, planDigest });
291
359
  }
292
360
 
361
+ function normalizeCompletionPolicy(value) {
362
+ const minimumIndependentGroups = integer(
363
+ value.minimumIndependentGroups,
364
+ "minimumIndependentGroups",
365
+ 10_000,
366
+ );
367
+ if (
368
+ minimumIndependentGroups < 2 ||
369
+ value.primaryMetric !== PRIMARY_METRIC ||
370
+ value.resamplingUnit !== RESAMPLING_UNIT ||
371
+ !Array.isArray(value.taskGroups) ||
372
+ value.taskGroups.length !== value.testTaskIds.length
373
+ ) {
374
+ throw new TypeError("PM completion policy or task groups are invalid");
375
+ }
376
+ const taskGroups = value.taskGroups.map((task, index) => {
377
+ exact(task, ["taskId", "groupKeys"], "PM effect task group");
378
+ const groupKeys = normalizeStringArray(
379
+ task.groupKeys,
380
+ "PM effect groupKeys",
381
+ );
382
+ if (
383
+ task.taskId !== value.testTaskIds[index] ||
384
+ groupKeys.length !== PREFIXES.length ||
385
+ groupKeys.some(
386
+ (key, groupIndex) =>
387
+ !key.startsWith(`${PREFIXES[groupIndex]}-`) ||
388
+ !GROUP_DIGEST.test(key.slice(PREFIXES[groupIndex].length + 1)),
389
+ )
390
+ ) {
391
+ throw new TypeError("PM effect task group binding is invalid");
392
+ }
393
+ return Object.freeze({ taskId: task.taskId, groupKeys });
394
+ });
395
+ return {
396
+ minimumPassRateDelta: finite(
397
+ value.minimumPassRateDelta,
398
+ "minimumPassRateDelta",
399
+ ),
400
+ minimumIndependentGroups,
401
+ taskGroups: Object.freeze(taskGroups),
402
+ primaryMetric: PRIMARY_METRIC,
403
+ resamplingUnit: RESAMPLING_UNIT,
404
+ };
405
+ }
406
+
293
407
  function normalizeStringArray(value, label) {
294
408
  if (
295
409
  !Array.isArray(value) ||
@@ -335,6 +449,7 @@ function normalizeSeeds(value) {
335
449
  * per-seed budget and differ only by the two declared immutable artifacts.
336
450
  */
337
451
  export function buildPmExplorationEffectPlan(input) {
452
+ assertPlainData(input, "PM effect plan input");
338
453
  exact(
339
454
  input,
340
455
  [
@@ -351,7 +466,8 @@ export function buildPmExplorationEffectPlan(input) {
351
466
  "resetProtocolDigest",
352
467
  "seeds",
353
468
  "budgetPerArmPerSeed",
354
- "minimumScoreDelta",
469
+ "minimumPassRateDelta",
470
+ "minimumIndependentGroups",
355
471
  ],
356
472
  "PM effect plan input",
357
473
  );
@@ -390,12 +506,18 @@ export function buildPmExplorationEffectPlan(input) {
390
506
  seeds,
391
507
  budgetPerArmPerSeed: normalizeEffectBudget(input.budgetPerArmPerSeed),
392
508
  costPhases: EFFECT_PHASES,
393
- minimumScoreDelta: finite(input.minimumScoreDelta, "minimumScoreDelta"),
509
+ minimumPassRateDelta: input.minimumPassRateDelta,
510
+ minimumIndependentGroups: input.minimumIndependentGroups,
511
+ taskGroups: suite.tasks
512
+ .filter((task) => task.split === "test")
513
+ .map((task) => ({ taskId: task.id, groupKeys: task.groupKeys })),
514
+ primaryMetric: PRIMARY_METRIC,
515
+ resamplingUnit: RESAMPLING_UNIT,
394
516
  bootstrapSamples: 1_000,
395
517
  equalBudget: true,
396
518
  promotionAuthority: false,
397
519
  };
398
- return deepFreeze({
520
+ return verifyPmExplorationEffectPlan({
399
521
  ...core,
400
522
  planDigest: hash(PM_EXPLORATION_EFFECT_PLAN_SCHEMA, core),
401
523
  });
@@ -447,7 +569,7 @@ function normalizePhases(value, label) {
447
569
  );
448
570
  }
449
571
 
450
- function normalizeArm(value, label) {
572
+ function normalizeArm(value, label, legacy) {
451
573
  exact(
452
574
  value,
453
575
  [
@@ -464,19 +586,36 @@ function normalizeArm(value, label) {
464
586
  );
465
587
  if (
466
588
  typeof value.passed !== "boolean" ||
467
- !FAILURE_CLASSES.has(value.failureClass)
589
+ !(legacy ? FAILURE_CLASSES : COMPLETION_FAILURE_CLASSES).has(
590
+ value.failureClass,
591
+ )
468
592
  ) {
469
593
  throw new TypeError(`${label} result fields are invalid`);
470
594
  }
595
+ if (
596
+ !legacy &&
597
+ value.passed &&
598
+ (value.score !== 1 ||
599
+ value.failureClass !== "none" ||
600
+ value.securityViolations !== 0 ||
601
+ value.permissionViolations !== 0)
602
+ ) {
603
+ throw new TypeError(`${label} passed requires a complete, safe outcome`);
604
+ }
605
+ const ungradedFailure =
606
+ !legacy &&
607
+ value.graderReceiptDigest === null &&
608
+ value.passed === false &&
609
+ value.score === 0 &&
610
+ value.failureClass !== "none";
471
611
  return Object.freeze({
472
612
  outcomeReceiptDigest: digest(
473
613
  value.outcomeReceiptDigest,
474
614
  `${label} outcomeReceiptDigest`,
475
615
  ),
476
- graderReceiptDigest: digest(
477
- value.graderReceiptDigest,
478
- `${label} graderReceiptDigest`,
479
- ),
616
+ graderReceiptDigest: ungradedFailure
617
+ ? null
618
+ : digest(value.graderReceiptDigest, `${label} graderReceiptDigest`),
480
619
  score: finite(value.score, `${label} score`),
481
620
  passed: value.passed,
482
621
  usage: normalizeUsage(value.usage, `${label} usage`),
@@ -492,11 +631,11 @@ function normalizeArm(value, label) {
492
631
  });
493
632
  }
494
633
 
495
- function addUsage(target, usage) {
496
- target.tokens += usage.tokens;
497
- target.toolCalls += usage.toolCalls;
498
- target.wallClockMs += usage.wallClockMs;
499
- target.costMicrounits += usage.costMicrounits;
634
+ function addUsage(target, usage, strict = false) {
635
+ for (const key of Object.keys(target)) {
636
+ target[key] += usage[key];
637
+ if (strict) integer(target[key], `PM effect total ${key}`);
638
+ }
500
639
  }
501
640
 
502
641
  function emptyUsage() {
@@ -535,20 +674,35 @@ function prngFromDigest(value) {
535
674
  };
536
675
  }
537
676
 
538
- function armTotals(runs, arm) {
677
+ function armTotals(runs, arm, legacy) {
539
678
  const usage = emptyUsage();
540
679
  let securityViolations = 0;
541
680
  let permissionViolations = 0;
542
681
  let passCount = 0;
682
+ const failureCounts = Object.fromEntries(
683
+ [...COMPLETION_FAILURE_CLASSES]
684
+ .filter((key) => key !== "none")
685
+ .concat("incomplete")
686
+ .map((key) => [key, 0]),
687
+ );
543
688
  const scores = [];
544
689
  for (const run of runs) {
545
- for (const phase of run.phases[arm]) addUsage(usage, phase.usage);
690
+ for (const phase of run.phases[arm]) addUsage(usage, phase.usage, !legacy);
546
691
  for (const item of run.cases) {
547
692
  const result = item[arm];
548
- addUsage(usage, result.usage);
693
+ addUsage(usage, result.usage, !legacy);
549
694
  securityViolations += result.securityViolations;
550
695
  permissionViolations += result.permissionViolations;
696
+ if (!legacy) {
697
+ integer(securityViolations, "PM effect total securityViolations");
698
+ integer(permissionViolations, "PM effect total permissionViolations");
699
+ }
551
700
  passCount += result.passed ? 1 : 0;
701
+ if (!result.passed) {
702
+ failureCounts[
703
+ result.failureClass === "none" ? "incomplete" : result.failureClass
704
+ ] += 1;
705
+ }
552
706
  scores.push(result.score);
553
707
  }
554
708
  }
@@ -560,9 +714,119 @@ function armTotals(runs, arm) {
560
714
  usage: Object.freeze(usage),
561
715
  securityViolations,
562
716
  permissionViolations,
717
+ ...(!legacy
718
+ ? {
719
+ failureCount: scores.length - passCount,
720
+ failureCounts: Object.freeze(failureCounts),
721
+ }
722
+ : {}),
723
+ });
724
+ }
725
+
726
+ // Shared template/project/principal/time-window keys form transitive clusters.
727
+ // All seeds of a task stay together; repeated samples never add independent units.
728
+ function completionGroups(plan) {
729
+ const parents = plan.taskGroups.map((_, index) => index);
730
+ const find = (index) => {
731
+ while (parents[index] !== index) {
732
+ parents[index] = parents[parents[index]];
733
+ index = parents[index];
734
+ }
735
+ return index;
736
+ };
737
+ const owners = new Map();
738
+ plan.taskGroups.forEach((task, index) => {
739
+ for (const key of task.groupKeys) {
740
+ if (owners.has(key)) parents[find(index)] = find(owners.get(key));
741
+ else owners.set(key, index);
742
+ }
743
+ });
744
+ const groups = new Map();
745
+ plan.taskGroups.forEach((task, index) => {
746
+ const root = find(index);
747
+ if (!groups.has(root)) groups.set(root, []);
748
+ groups.get(root).push(index);
749
+ });
750
+ return [...groups.values()];
751
+ }
752
+
753
+ /** Planning diagnostics only; this neither launches nor authenticates a run. */
754
+ export function inspectPmExplorationEffectPlan(value) {
755
+ const plan = verifyPmExplorationEffectPlan(value);
756
+ if (plan.schema !== PM_EXPLORATION_EFFECT_PLAN_SCHEMA) {
757
+ throw new TypeError("PM completion diagnostics require an effect plan v2");
758
+ }
759
+ const groups = completionGroups(plan).map((indices) =>
760
+ indices.map((index) => plan.taskGroups[index].taskId),
761
+ );
762
+ const groupCountSatisfied = groups.length >= plan.minimumIndependentGroups;
763
+ return deepFreeze({
764
+ schema: "chainlesschain.pm-exploration-effect-plan-inspection/v1",
765
+ planDigest: plan.planDigest,
766
+ status: groupCountSatisfied
767
+ ? "group-count-satisfied"
768
+ : "insufficient-independent-groups",
769
+ testTaskCount: plan.testTaskIds.length,
770
+ plannedRunCount: plan.seeds.length,
771
+ plannedPairedObservationCount: plan.testTaskIds.length * plan.seeds.length,
772
+ plannedArmObservationCount: 2 * plan.testTaskIds.length * plan.seeds.length,
773
+ independentGroupCount: groups.length,
774
+ minimumIndependentGroups: plan.minimumIndependentGroups,
775
+ independentGroups: groups,
776
+ runtimeVerified: false,
777
+ evidenceAuthenticated: false,
778
+ qualifiesForPromotion: false,
563
779
  });
564
780
  }
565
781
 
782
+ function clusteredDeltas(plan, perTask) {
783
+ const groups = completionGroups(plan).map((indices) =>
784
+ indices.map((index) => perTask[index]),
785
+ );
786
+ const totals = groups.map((group) => ({
787
+ taskCount: group.length,
788
+ score: group.reduce((sum, task) => sum + task.scoreDelta, 0),
789
+ passRate: group.reduce((sum, task) => sum + task.passRateDelta, 0),
790
+ }));
791
+ // Use the frozen grouping as the random seed: adding identical seed repeats
792
+ // must not narrow the interval through either pseudoreplication or RNG drift.
793
+ const random = prngFromDigest(hash(RESAMPLING_UNIT, plan.taskGroups));
794
+ const scoreSamples = [];
795
+ const passSamples = [];
796
+ for (let sample = 0; sample < plan.bootstrapSamples; sample += 1) {
797
+ let count = 0;
798
+ let score = 0;
799
+ let passRate = 0;
800
+ for (let index = 0; index < totals.length; index += 1) {
801
+ const selected = totals[Math.floor(random() * totals.length)];
802
+ count += selected.taskCount;
803
+ score += selected.score;
804
+ passRate += selected.passRate;
805
+ }
806
+ scoreSamples.push(score / count);
807
+ passSamples.push(passRate / count);
808
+ }
809
+ const summarize = (samples, key) => {
810
+ samples.sort((left, right) => left - right);
811
+ return Object.freeze({
812
+ mean: mean(perTask.map((task) => task[key])),
813
+ // A single cluster cannot estimate between-cluster uncertainty.
814
+ bootstrap95Ci:
815
+ groups.length < 2
816
+ ? null
817
+ : Object.freeze([
818
+ percentile(samples, 0.025),
819
+ percentile(samples, 0.975),
820
+ ]),
821
+ });
822
+ };
823
+ return {
824
+ independentGroupCount: groups.length,
825
+ pairedScoreDelta: summarize(scoreSamples, "scoreDelta"),
826
+ pairedPassRateDelta: summarize(passSamples, "passRateDelta"),
827
+ };
828
+ }
829
+
566
830
  /**
567
831
  * Recomputes a complete paired PM effect report. It is deliberately not a
568
832
  * promotion receipt; an independent signed Eval Gate/Pilot decision remains
@@ -570,6 +834,8 @@ function armTotals(runs, arm) {
570
834
  */
571
835
  export function buildPmExplorationEffectReport({ plan, runs } = {}) {
572
836
  const verifiedPlan = verifyPmExplorationEffectPlan(plan);
837
+ const legacy = verifiedPlan.schema === LEGACY_EFFECT_PLAN_SCHEMA;
838
+ assertPlainData(runs, "PM effect runs");
573
839
  if (
574
840
  !Array.isArray(runs) ||
575
841
  isProxy(runs) ||
@@ -581,6 +847,7 @@ export function buildPmExplorationEffectReport({ plan, runs } = {}) {
581
847
  }
582
848
  const expectedTasks = new Set(verifiedPlan.testTaskIds);
583
849
  const seenSeeds = new Set();
850
+ const seenRunIds = new Set();
584
851
  const normalizedRuns = runs.map((run, runIndex) => {
585
852
  exact(
586
853
  run,
@@ -598,6 +865,11 @@ export function buildPmExplorationEffectReport({ plan, runs } = {}) {
598
865
  );
599
866
  }
600
867
  seenSeeds.add(seed);
868
+ const runId = boundedString(run.runId, `PM effect run ${runIndex} runId`);
869
+ if (!legacy && seenRunIds.has(runId)) {
870
+ throw new TypeError("PM effect runId is duplicated");
871
+ }
872
+ seenRunIds.add(runId);
601
873
  exact(
602
874
  run.phases,
603
875
  ["baseline", "candidate"],
@@ -640,10 +912,15 @@ export function buildPmExplorationEffectReport({ plan, runs } = {}) {
640
912
  seenTasks.add(taskId);
641
913
  return Object.freeze({
642
914
  taskId,
643
- baseline: normalizeArm(item.baseline, `PM effect ${taskId} baseline`),
915
+ baseline: normalizeArm(
916
+ item.baseline,
917
+ `PM effect ${taskId} baseline`,
918
+ legacy,
919
+ ),
644
920
  candidate: normalizeArm(
645
921
  item.candidate,
646
922
  `PM effect ${taskId} candidate`,
923
+ legacy,
647
924
  ),
648
925
  });
649
926
  });
@@ -651,14 +928,14 @@ export function buildPmExplorationEffectReport({ plan, runs } = {}) {
651
928
  const budgetViolations = [];
652
929
  for (const arm of ["baseline", "candidate"]) {
653
930
  const usage = emptyUsage();
654
- for (const phase of phases[arm]) addUsage(usage, phase.usage);
655
- for (const item of cases) addUsage(usage, item[arm].usage);
931
+ for (const phase of phases[arm]) addUsage(usage, phase.usage, !legacy);
932
+ for (const item of cases) addUsage(usage, item[arm].usage, !legacy);
656
933
  if (exceedsBudget(usage, verifiedPlan.budgetPerArmPerSeed)) {
657
934
  budgetViolations.push(arm);
658
935
  }
659
936
  }
660
937
  return Object.freeze({
661
- runId: boundedString(run.runId, `PM effect run ${runIndex} runId`),
938
+ runId,
662
939
  seed,
663
940
  phases,
664
941
  cases: Object.freeze(cases),
@@ -674,16 +951,18 @@ export function buildPmExplorationEffectReport({ plan, runs } = {}) {
674
951
  run.cases.map((item) => item.candidate.score - item.baseline.score),
675
952
  );
676
953
  const random = prngFromDigest(verifiedPlan.planDigest);
677
- const bootstrap = Array.from({ length: verifiedPlan.bootstrapSamples }, () =>
678
- mean(
679
- Array.from(
680
- { length: deltas.length },
681
- () => deltas[Math.floor(random() * deltas.length)],
682
- ),
683
- ),
684
- ).sort((left, right) => left - right);
685
- const baseline = armTotals(normalizedRuns, "baseline");
686
- const candidate = armTotals(normalizedRuns, "candidate");
954
+ const bootstrap = legacy
955
+ ? Array.from({ length: verifiedPlan.bootstrapSamples }, () =>
956
+ mean(
957
+ Array.from(
958
+ { length: deltas.length },
959
+ () => deltas[Math.floor(random() * deltas.length)],
960
+ ),
961
+ ),
962
+ ).sort((left, right) => left - right)
963
+ : null;
964
+ const baseline = armTotals(normalizedRuns, "baseline", legacy);
965
+ const candidate = armTotals(normalizedRuns, "candidate", legacy);
687
966
  const perTask = Object.freeze(
688
967
  verifiedPlan.testTaskIds.map((taskId) => {
689
968
  const observations = normalizedRuns.map((run) =>
@@ -705,27 +984,50 @@ export function buildPmExplorationEffectReport({ plan, runs } = {}) {
705
984
  scoreDelta: mean(candidateScores) - mean(baselineScores),
706
985
  baselinePassRate,
707
986
  candidatePassRate,
987
+ ...(!legacy
988
+ ? { passRateDelta: candidatePassRate - baselinePassRate }
989
+ : {}),
708
990
  });
709
991
  }),
710
992
  );
711
- const pairedScoreDelta = Object.freeze({
712
- mean: mean(deltas),
713
- bootstrap95Ci: Object.freeze([
714
- percentile(bootstrap, 0.025),
715
- percentile(bootstrap, 0.975),
716
- ]),
717
- });
993
+ const completion = legacy ? null : clusteredDeltas(verifiedPlan, perTask);
994
+ const pairedScoreDelta = legacy
995
+ ? Object.freeze({
996
+ mean: mean(deltas),
997
+ bootstrap95Ci: Object.freeze([
998
+ percentile(bootstrap, 0.025),
999
+ percentile(bootstrap, 0.975),
1000
+ ]),
1001
+ })
1002
+ : completion.pairedScoreDelta;
718
1003
  const budgetViolationCount = normalizedRuns.reduce(
719
1004
  (sum, run) => sum + run.budgetViolations.length,
720
1005
  0,
721
1006
  );
722
- const thresholdMet =
1007
+ const safetyAndBudgetSatisfied =
723
1008
  budgetViolationCount === 0 &&
724
1009
  candidate.securityViolations === 0 &&
725
1010
  candidate.permissionViolations === 0 &&
726
- pairedScoreDelta.bootstrap95Ci[0] >= verifiedPlan.minimumScoreDelta;
1011
+ (legacy ||
1012
+ (baseline.securityViolations === 0 &&
1013
+ baseline.permissionViolations === 0));
1014
+ const thresholdMet =
1015
+ safetyAndBudgetSatisfied &&
1016
+ (legacy
1017
+ ? pairedScoreDelta.bootstrap95Ci[0] >= verifiedPlan.minimumScoreDelta
1018
+ : completion.independentGroupCount >=
1019
+ verifiedPlan.minimumIndependentGroups &&
1020
+ completion.pairedPassRateDelta.mean > 0 &&
1021
+ completion.pairedPassRateDelta.bootstrap95Ci[0] >=
1022
+ verifiedPlan.minimumPassRateDelta);
1023
+ const insufficient =
1024
+ !legacy &&
1025
+ safetyAndBudgetSatisfied &&
1026
+ completion.independentGroupCount < verifiedPlan.minimumIndependentGroups;
727
1027
  const core = {
728
- schema: PM_EXPLORATION_EFFECT_REPORT_SCHEMA,
1028
+ schema: legacy
1029
+ ? LEGACY_EFFECT_REPORT_SCHEMA
1030
+ : PM_EXPLORATION_EFFECT_REPORT_SCHEMA,
729
1031
  planDigest: verifiedPlan.planDigest,
730
1032
  runCount: normalizedRuns.length,
731
1033
  pairedObservationCount: deltas.length,
@@ -733,25 +1035,41 @@ export function buildPmExplorationEffectReport({ plan, runs } = {}) {
733
1035
  candidate,
734
1036
  perTask,
735
1037
  pairedScoreDelta,
1038
+ ...(!legacy
1039
+ ? {
1040
+ primaryMetric: PRIMARY_METRIC,
1041
+ resamplingUnit: RESAMPLING_UNIT,
1042
+ independentGroupCount: completion.independentGroupCount,
1043
+ pairedPassRateDelta: completion.pairedPassRateDelta,
1044
+ }
1045
+ : {}),
736
1046
  budgetViolationCount,
737
- evidenceDecision: thresholdMet ? "threshold-met" : "threshold-not-met",
1047
+ evidenceDecision: insufficient
1048
+ ? "insufficient-evidence"
1049
+ : thresholdMet
1050
+ ? "threshold-met"
1051
+ : "threshold-not-met",
738
1052
  requiresIndependentPilotApproval: true,
739
1053
  qualifiesForPromotion: false,
740
1054
  runs: Object.freeze(normalizedRuns),
741
1055
  };
742
1056
  return deepFreeze({
743
1057
  ...core,
744
- reportDigest: hash(PM_EXPLORATION_EFFECT_REPORT_SCHEMA, core),
1058
+ reportDigest: hash(core.schema, core),
745
1059
  });
746
1060
  }
747
1061
 
748
1062
  export function verifyPmExplorationEffectReport({ plan, report } = {}) {
749
1063
  const verifiedPlan = verifyPmExplorationEffectPlan(plan);
1064
+ const legacy = verifiedPlan.schema === LEGACY_EFFECT_PLAN_SCHEMA;
750
1065
  if (!report || typeof report !== "object" || isProxy(report)) {
751
1066
  throw new TypeError("a canonical PM effect report is required");
752
1067
  }
753
1068
  assertPlainData(report, "PM effect report");
754
- if (report.schema !== PM_EXPLORATION_EFFECT_REPORT_SCHEMA) {
1069
+ if (
1070
+ report.schema !==
1071
+ (legacy ? LEGACY_EFFECT_REPORT_SCHEMA : PM_EXPLORATION_EFFECT_REPORT_SCHEMA)
1072
+ ) {
755
1073
  throw new TypeError("a canonical PM effect report is required");
756
1074
  }
757
1075
  exact(
@@ -765,6 +1083,14 @@ export function verifyPmExplorationEffectReport({ plan, report } = {}) {
765
1083
  "candidate",
766
1084
  "perTask",
767
1085
  "pairedScoreDelta",
1086
+ ...(!legacy
1087
+ ? [
1088
+ "primaryMetric",
1089
+ "resamplingUnit",
1090
+ "independentGroupCount",
1091
+ "pairedPassRateDelta",
1092
+ ]
1093
+ : []),
768
1094
  "budgetViolationCount",
769
1095
  "evidenceDecision",
770
1096
  "requiresIndependentPilotApproval",
@@ -805,6 +1131,1645 @@ export function verifyPmExplorationEffectReport({ plan, report } = {}) {
805
1131
  return recreated;
806
1132
  }
807
1133
 
1134
+ function snapshotEffectEvidence(value, label) {
1135
+ assertPlainData(value, label);
1136
+ const serialized = canonical(value);
1137
+ if (Buffer.byteLength(serialized, "utf8") > 16 * 1024 * 1024) {
1138
+ throw new TypeError(`${label} exceeds the 16 MiB limit`);
1139
+ }
1140
+ return deepFreeze(JSON.parse(serialized));
1141
+ }
1142
+
1143
+ function effectUsageFromEval(row) {
1144
+ const usage = {
1145
+ tokens: row.metrics.tokens,
1146
+ toolCalls: row.metrics.toolCalls,
1147
+ wallClockMs: row.metrics.latencyMs,
1148
+ costMicrounits: row.metrics.costMicrounits,
1149
+ };
1150
+ return {
1151
+ // The execution receipt is the signed source of these metrics. This is not
1152
+ // a claim to have re-read a provider settlement or a separate usage receipt.
1153
+ receiptDigest: Object.values(usage).some((amount) => amount > 0)
1154
+ ? row.executionDigest
1155
+ : null,
1156
+ ...usage,
1157
+ };
1158
+ }
1159
+
1160
+ function effectArmFromEval(row) {
1161
+ const failureClass =
1162
+ row.permissionViolations > 0
1163
+ ? "permission"
1164
+ : row.securityViolations > 0 || (!row.pass && row.metrics.errors > 0)
1165
+ ? "unknown"
1166
+ : "none";
1167
+ return {
1168
+ outcomeReceiptDigest: row.executionDigest,
1169
+ graderReceiptDigest: row.gradeDigest,
1170
+ score: row.qualityScore,
1171
+ passed: row.pass && row.qualityScore === 1 && failureClass === "none",
1172
+ usage: effectUsageFromEval(row),
1173
+ securityViolations: row.securityViolations,
1174
+ permissionViolations: row.permissionViolations,
1175
+ failureClass,
1176
+ };
1177
+ }
1178
+
1179
+ /**
1180
+ * Authenticate only the provider token count and estimated USD cost for
1181
+ * independently registered preparation requests. This does not establish that
1182
+ * the registry contains every request, or authenticate tools and elapsed time.
1183
+ */
1184
+ export function buildPmExplorationPreparationProviderEvidence(
1185
+ adapter,
1186
+ input,
1187
+ expected,
1188
+ ) {
1189
+ const store = capturePmExplorationProviderSettlementStore(adapter);
1190
+ const source = snapshotEffectEvidence(
1191
+ input,
1192
+ "PM preparation provider source",
1193
+ );
1194
+ const registered = snapshotEffectEvidence(
1195
+ expected,
1196
+ "PM preparation provider registered requests",
1197
+ );
1198
+ exact(
1199
+ source,
1200
+ ["plan", "phaseUsage", "settlements"],
1201
+ "PM preparation provider source",
1202
+ );
1203
+ exact(
1204
+ registered,
1205
+ ["planDigest", "descriptorDigest", "requests"],
1206
+ "PM preparation provider registered requests",
1207
+ );
1208
+ const plan = verifyPmExplorationEffectPlan(source.plan);
1209
+ if (
1210
+ plan.schema !== PM_EXPLORATION_EFFECT_PLAN_SCHEMA ||
1211
+ plan.planDigest !== registered.planDigest ||
1212
+ store.inspect().descriptorDigest !== registered.descriptorDigest
1213
+ ) {
1214
+ throw new Error(
1215
+ "PM preparation provider registration differs from plan or store",
1216
+ );
1217
+ }
1218
+ if (
1219
+ !Array.isArray(source.phaseUsage) ||
1220
+ source.phaseUsage.length !== plan.seeds.length ||
1221
+ !Array.isArray(source.settlements) ||
1222
+ !Array.isArray(registered.requests) ||
1223
+ source.settlements.length === 0 ||
1224
+ source.settlements.length !== registered.requests.length
1225
+ ) {
1226
+ throw new TypeError("PM preparation provider evidence coverage is invalid");
1227
+ }
1228
+ const preparation = new Map();
1229
+ for (const entry of source.phaseUsage) {
1230
+ exact(entry, ["seed", "baseline", "candidate"], "PM preparation usage");
1231
+ if (!plan.seeds.includes(entry.seed) || preparation.has(entry.seed))
1232
+ throw new TypeError("PM preparation usage seed is absent or duplicated");
1233
+ preparation.set(entry.seed, {
1234
+ baseline: normalizePhases(entry.baseline, "PM baseline preparation"),
1235
+ candidate: normalizePhases(entry.candidate, "PM candidate preparation"),
1236
+ });
1237
+ }
1238
+ const registrations = new Map();
1239
+ const requestDigests = new Set();
1240
+ for (const entry of registered.requests) {
1241
+ exact(
1242
+ entry,
1243
+ [
1244
+ "seed",
1245
+ "arm",
1246
+ "phase",
1247
+ "operationId",
1248
+ "executionRequestDigest",
1249
+ "requestDigest",
1250
+ ],
1251
+ "PM preparation provider request",
1252
+ );
1253
+ if (
1254
+ !plan.seeds.includes(entry.seed) ||
1255
+ !["baseline", "candidate"].includes(entry.arm) ||
1256
+ !EFFECT_PHASES.includes(entry.phase)
1257
+ ) {
1258
+ throw new TypeError(
1259
+ "PM preparation provider request has invalid coordinates",
1260
+ );
1261
+ }
1262
+ const executionDigest = digest(
1263
+ entry.executionRequestDigest,
1264
+ "PM preparation execution request digest",
1265
+ );
1266
+ digest(entry.requestDigest, "PM preparation provider request digest");
1267
+ if (
1268
+ typeof entry.operationId !== "string" ||
1269
+ !entry.operationId ||
1270
+ entry.operationId.length > 256 ||
1271
+ requestDigests.has(executionDigest)
1272
+ ) {
1273
+ throw new TypeError(
1274
+ "PM preparation provider request is duplicated or invalid",
1275
+ );
1276
+ }
1277
+ requestDigests.add(executionDigest);
1278
+ registrations.set(executionDigest, entry);
1279
+ }
1280
+ const seen = new Set();
1281
+ const totals = new Map();
1282
+ const entries = [];
1283
+ for (const entry of source.settlements) {
1284
+ exact(
1285
+ entry,
1286
+ ["seed", "arm", "phase", "settlement", "persistence"],
1287
+ "PM preparation provider settlement",
1288
+ );
1289
+ const settlement = entry.settlement;
1290
+ const registeredRequest = registrations.get(
1291
+ settlement?.executionRequestDigest,
1292
+ );
1293
+ if (
1294
+ !registeredRequest ||
1295
+ registeredRequest.seed !== entry.seed ||
1296
+ registeredRequest.arm !== entry.arm ||
1297
+ registeredRequest.phase !== entry.phase ||
1298
+ registeredRequest.operationId !== settlement.operationId ||
1299
+ registeredRequest.requestDigest !== settlement.requestDigest ||
1300
+ seen.has(settlement.executionRequestDigest)
1301
+ ) {
1302
+ throw new Error(
1303
+ "PM preparation settlement lacks unique registered request binding",
1304
+ );
1305
+ }
1306
+ store.verifySettlementPersistence(settlement, entry.persistence);
1307
+ seen.add(settlement.executionRequestDigest);
1308
+ const key = `${entry.seed}\0${entry.arm}\0${entry.phase}`;
1309
+ const previous = totals.get(key) ?? { tokens: 0, estimatedUsd: 0 };
1310
+ const tokens = previous.tokens + settlement.usage.totalTokens;
1311
+ const estimatedUsd = previous.estimatedUsd + settlement.estimatedCost.total;
1312
+ if (!Number.isSafeInteger(tokens) || !Number.isFinite(estimatedUsd))
1313
+ throw new TypeError(
1314
+ "PM preparation provider totals exceed numeric bounds",
1315
+ );
1316
+ totals.set(key, { tokens, estimatedUsd });
1317
+ entries.push({
1318
+ seed: entry.seed,
1319
+ arm: entry.arm,
1320
+ phase: entry.phase,
1321
+ executionRequestDigest: settlement.executionRequestDigest,
1322
+ settlementDigest: settlement.settlementDigest,
1323
+ durableRecordDigest: entry.persistence.recordDigest,
1324
+ tokens: settlement.usage.totalTokens,
1325
+ estimatedUsd: settlement.estimatedCost.total,
1326
+ });
1327
+ }
1328
+ if (seen.size !== registrations.size)
1329
+ throw new Error(
1330
+ "PM preparation provider registration is not fully covered",
1331
+ );
1332
+ const phaseTotals = [];
1333
+ for (const seed of plan.seeds) {
1334
+ for (const arm of ["baseline", "candidate"]) {
1335
+ for (const { phase, usage } of preparation.get(seed)[arm]) {
1336
+ const total = totals.get(`${seed}\0${arm}\0${phase}`);
1337
+ if (!total) continue;
1338
+ // Round upward: the integer micro-USD comparison must never understate
1339
+ // the provider's independently re-read estimated dollar amount.
1340
+ const costMicrounitsUpperBound = Math.ceil(total.estimatedUsd * 1e6);
1341
+ if (
1342
+ !Number.isSafeInteger(costMicrounitsUpperBound) ||
1343
+ total.tokens > usage.tokens ||
1344
+ costMicrounitsUpperBound > usage.costMicrounits
1345
+ ) {
1346
+ throw new Error(
1347
+ "PM preparation provider cost exceeds declared phase usage",
1348
+ );
1349
+ }
1350
+ phaseTotals.push({
1351
+ seed,
1352
+ arm,
1353
+ phase,
1354
+ tokens: total.tokens,
1355
+ estimatedUsd: total.estimatedUsd,
1356
+ costMicrounitsUpperBound,
1357
+ });
1358
+ }
1359
+ }
1360
+ }
1361
+ const core = {
1362
+ schema: PM_EXPLORATION_PREPARATION_PROVIDER_EVIDENCE_SCHEMA,
1363
+ planDigest: plan.planDigest,
1364
+ descriptorDigest: registered.descriptorDigest,
1365
+ registeredRequestDigest: hash(
1366
+ PREPARATION_PROVIDER_REQUESTS_SCHEMA,
1367
+ registered.requests,
1368
+ ),
1369
+ phaseUsageDigest: hash(
1370
+ "chainlesschain.pm-effect-preparation-usage/v1",
1371
+ plan.seeds.map((seed) => ({ seed, ...preparation.get(seed) })),
1372
+ ),
1373
+ entries,
1374
+ phaseTotals,
1375
+ authenticationScope: "registered-provider-tokens-and-estimated-cost-only",
1376
+ preparationEvidenceAuthenticated: false,
1377
+ reportAuthenticated: false,
1378
+ };
1379
+ return deepFreeze({
1380
+ ...core,
1381
+ evidenceDigest: hash(core.schema, core),
1382
+ });
1383
+ }
1384
+
1385
+ export function verifyPmExplorationPreparationProviderEvidence(
1386
+ adapter,
1387
+ input,
1388
+ expected,
1389
+ ) {
1390
+ exact(input, ["source", "evidence"], "PM preparation provider verification");
1391
+ const evidence = snapshotEffectEvidence(
1392
+ input.evidence,
1393
+ "PM preparation provider evidence",
1394
+ );
1395
+ const recreated = buildPmExplorationPreparationProviderEvidence(
1396
+ adapter,
1397
+ input.source,
1398
+ expected,
1399
+ );
1400
+ if (canonical(recreated) !== canonical(evidence))
1401
+ throw new Error(
1402
+ "PM preparation provider evidence differs from durable source",
1403
+ );
1404
+ return recreated;
1405
+ }
1406
+
1407
+ /**
1408
+ * Connect authenticated Eval Gate rows to the v2 statistics. Preparation usage
1409
+ * remains caller-supplied, so the envelope can never certify a complete effect
1410
+ * claim. `expected` is an independent, pre-registered host binding, not data to
1411
+ * derive from the report being verified.
1412
+ */
1413
+ export async function buildPmExplorationEffectEvidenceReport(
1414
+ verifier,
1415
+ input,
1416
+ expected,
1417
+ ) {
1418
+ const source = snapshotEffectEvidence(input, "PM effect evidence source");
1419
+ exact(
1420
+ source,
1421
+ ["plan", "suite", "policy", "receipt", "resultEvidence", "phaseUsage"],
1422
+ "PM effect evidence source",
1423
+ );
1424
+ const registered = snapshotEffectEvidence(
1425
+ expected,
1426
+ "PM effect registered context",
1427
+ );
1428
+ exact(
1429
+ registered,
1430
+ ["planDigest", "targetMatrixRoot", "cellId", "runtimeId", "receiptContext"],
1431
+ "PM effect registered context",
1432
+ );
1433
+ const { plan, context, evaluationContextDigest } =
1434
+ verifyPmEffectSourceRegistration(source, registered);
1435
+ return projectPmEffectEvidenceReport(
1436
+ verifier,
1437
+ source,
1438
+ plan,
1439
+ context,
1440
+ evaluationContextDigest,
1441
+ );
1442
+ }
1443
+
1444
+ function verifyPmEffectSourceRegistration(source, registered) {
1445
+ const plan = verifyPmExplorationEffectPlan(source.plan);
1446
+ if (
1447
+ plan.schema !== PM_EXPLORATION_EFFECT_PLAN_SCHEMA ||
1448
+ plan.planDigest !== registered.planDigest
1449
+ ) {
1450
+ throw new Error("PM effect requires the registered v2 plan");
1451
+ }
1452
+ // Rebuild from the actual suite and policy; a self-consistent rehash of
1453
+ // caller-invented task groups or a different seed list must not be enough.
1454
+ const planInput = Object.fromEntries(
1455
+ [
1456
+ "experimentId",
1457
+ "baselineVersion",
1458
+ "candidateVersion",
1459
+ "actorConfigDigest",
1460
+ "modelConfigDigest",
1461
+ "toolPolicyDigest",
1462
+ "permissionPolicyDigest",
1463
+ "environmentDigest",
1464
+ "resetProtocolDigest",
1465
+ "seeds",
1466
+ "budgetPerArmPerSeed",
1467
+ "minimumPassRateDelta",
1468
+ "minimumIndependentGroups",
1469
+ ].map((key) => [key, plan[key]]),
1470
+ );
1471
+ const reconstructed = buildPmExplorationEffectPlan({
1472
+ ...planInput,
1473
+ suite: source.suite,
1474
+ policy: source.policy,
1475
+ });
1476
+ if (reconstructed.planDigest !== plan.planDigest) {
1477
+ throw new Error("PM effect plan differs from its source suite or policy");
1478
+ }
1479
+ const context = registered.receiptContext;
1480
+ const bindings = {
1481
+ suiteDigest: plan.suiteDigest,
1482
+ policyDigest: plan.policyDigest,
1483
+ environmentDigest: plan.environmentDigest,
1484
+ candidateId: plan.candidateVersion.artifactDigest,
1485
+ baselineId: plan.baselineVersion.artifactDigest,
1486
+ };
1487
+ if (
1488
+ !context ||
1489
+ Object.entries(bindings).some(([key, value]) => context[key] !== value)
1490
+ ) {
1491
+ throw new Error(
1492
+ "PM effect registered receipt context differs from its plan",
1493
+ );
1494
+ }
1495
+ const evaluationContextDigest = computeEvolutionEvalContextDigest({
1496
+ ...bindings,
1497
+ planDigest: plan.planDigest,
1498
+ tenantId: context.tenantId,
1499
+ targetEnvironmentRef: context.targetEnvironmentRef,
1500
+ evaluationAuthorityRoot: context.evaluationAuthorityRoot,
1501
+ targetMatrixRoot: registered.targetMatrixRoot,
1502
+ cellId: registered.cellId,
1503
+ runtimeId: registered.runtimeId,
1504
+ });
1505
+ if (evaluationContextDigest !== context.evaluationContextDigest) {
1506
+ throw new Error(
1507
+ "PM effect evaluation context does not bind the registered plan",
1508
+ );
1509
+ }
1510
+ return { plan, context, evaluationContextDigest };
1511
+ }
1512
+
1513
+ async function projectPmEffectEvidenceReport(
1514
+ verifier,
1515
+ source,
1516
+ plan,
1517
+ context,
1518
+ evaluationContextDigest,
1519
+ ) {
1520
+ if (
1521
+ !Array.isArray(source.phaseUsage) ||
1522
+ source.phaseUsage.length !== plan.seeds.length
1523
+ ) {
1524
+ throw new TypeError("PM effect preparation usage must cover every seed");
1525
+ }
1526
+ const preparation = new Map();
1527
+ for (const entry of source.phaseUsage) {
1528
+ exact(
1529
+ entry,
1530
+ ["seed", "baseline", "candidate"],
1531
+ "PM effect preparation usage",
1532
+ );
1533
+ if (!plan.seeds.includes(entry.seed) || preparation.has(entry.seed)) {
1534
+ throw new TypeError("PM effect preparation seed is absent or duplicated");
1535
+ }
1536
+ preparation.set(entry.seed, {
1537
+ baseline: normalizePhases(entry.baseline, "PM baseline preparation"),
1538
+ candidate: normalizePhases(entry.candidate, "PM candidate preparation"),
1539
+ });
1540
+ }
1541
+ const evidence = await verifyEvolutionEvalResultEvidence(
1542
+ verifier,
1543
+ {
1544
+ receipt: source.receipt,
1545
+ resultEvidence: source.resultEvidence,
1546
+ suite: source.suite,
1547
+ policy: source.policy,
1548
+ },
1549
+ context,
1550
+ );
1551
+ const taskDigests = new Map(
1552
+ source.suite.tasks
1553
+ .filter((task) => task.split === "test")
1554
+ .map((task) => [task.id, task.taskDigest]),
1555
+ );
1556
+ const indexed = Object.fromEntries(
1557
+ ["baseline", "candidate"].map((arm) => [
1558
+ arm,
1559
+ new Map(
1560
+ evidence.test[arm].map((row) => [
1561
+ `${row.taskDigest}\0${row.seed}`,
1562
+ row,
1563
+ ]),
1564
+ ),
1565
+ ]),
1566
+ );
1567
+ const runs = plan.seeds.map((seed) => ({
1568
+ // A deterministic grouping ID, not a claim of an additional execution.
1569
+ runId: `pm-effect-${hash("chainlesschain.pm-effect-eval-seed/v1", { receiptDigest: source.receipt.receiptDigest, seed }).slice(7)}`,
1570
+ seed,
1571
+ phases: preparation.get(seed),
1572
+ cases: plan.testTaskIds.map((taskId) => ({
1573
+ taskId,
1574
+ ...Object.fromEntries(
1575
+ ["baseline", "candidate"].map((arm) => [
1576
+ arm,
1577
+ effectArmFromEval(
1578
+ indexed[arm].get(`${taskDigests.get(taskId)}\0${seed}`),
1579
+ ),
1580
+ ]),
1581
+ ),
1582
+ })),
1583
+ }));
1584
+ const report = buildPmExplorationEffectReport({ plan, runs });
1585
+ const validationUsage = plan.seeds.map((seed) => ({
1586
+ seed,
1587
+ ...Object.fromEntries(
1588
+ ["baseline", "candidate"].map((arm) => {
1589
+ const usage = emptyUsage();
1590
+ for (const row of evidence.validation[arm]) {
1591
+ if (row.seed === seed)
1592
+ addUsage(usage, effectUsageFromEval(row), true);
1593
+ }
1594
+ return [arm, usage];
1595
+ }),
1596
+ ),
1597
+ }));
1598
+ const knownUsage = {
1599
+ baseline: { ...report.baseline.usage },
1600
+ candidate: { ...report.candidate.usage },
1601
+ };
1602
+ const knownBudgetViolations = [];
1603
+ for (const [index, run] of report.runs.entries()) {
1604
+ for (const arm of ["baseline", "candidate"]) {
1605
+ const usage = { ...validationUsage[index][arm] };
1606
+ addUsage(knownUsage[arm], usage, true);
1607
+ for (const phase of run.phases[arm]) addUsage(usage, phase.usage, true);
1608
+ for (const row of run.cases) addUsage(usage, row[arm].usage, true);
1609
+ if (exceedsBudget(usage, plan.budgetPerArmPerSeed))
1610
+ knownBudgetViolations.push({ seed: run.seed, arm });
1611
+ }
1612
+ }
1613
+ const validationUnsafe = ["baseline", "candidate"].some((arm) =>
1614
+ evidence.validation[arm].some(
1615
+ (row) => row.securityViolations > 0 || row.permissionViolations > 0,
1616
+ ),
1617
+ );
1618
+ const blockingReasons = ["preparation-costs-unverified"];
1619
+ if (source.receipt.decision !== "accepted")
1620
+ blockingReasons.push("source-eval-not-accepted");
1621
+ if (knownBudgetViolations.length)
1622
+ blockingReasons.push("known-budget-exceeded");
1623
+ if (validationUnsafe) blockingReasons.push("validation-safety-violation");
1624
+ if (report.evidenceDecision !== "threshold-met")
1625
+ blockingReasons.push(
1626
+ report.evidenceDecision === "threshold-not-met"
1627
+ ? "statistical-threshold-not-met"
1628
+ : "independent-groups-insufficient",
1629
+ );
1630
+ const core = {
1631
+ schema: PM_EXPLORATION_EFFECT_EVIDENCE_REPORT_SCHEMA,
1632
+ planDigest: plan.planDigest,
1633
+ sourceEvalRunId: source.receipt.runId,
1634
+ sourceEvalDecision: source.receipt.decision,
1635
+ sourceEvalReceiptDigest: source.receipt.receiptDigest,
1636
+ sourceResultEvidenceDigest: evidence.evidenceDigest,
1637
+ evaluationContextDigest,
1638
+ phaseUsageDigest: hash(
1639
+ "chainlesschain.pm-effect-preparation-usage/v1",
1640
+ plan.seeds.map((seed) => ({ seed, ...preparation.get(seed) })),
1641
+ ),
1642
+ validationUsage,
1643
+ knownUsage,
1644
+ knownBudgetViolations,
1645
+ testExecutionErrors: Object.fromEntries(
1646
+ ["baseline", "candidate"].map((arm) => [
1647
+ arm,
1648
+ integer(
1649
+ evidence.test[arm].reduce((sum, row) => sum + row.metrics.errors, 0),
1650
+ "test execution errors",
1651
+ ),
1652
+ ]),
1653
+ ),
1654
+ report,
1655
+ executionEvidenceAuthenticated: true,
1656
+ preparationEvidenceAuthenticated: false,
1657
+ reportAuthenticated: false,
1658
+ evidenceDecision:
1659
+ source.receipt.decision !== "accepted" ||
1660
+ knownBudgetViolations.length > 0 ||
1661
+ validationUnsafe ||
1662
+ report.evidenceDecision === "threshold-not-met"
1663
+ ? "threshold-not-met"
1664
+ : "insufficient-evidence",
1665
+ blockingReasons,
1666
+ requiresIndependentPilotApproval: true,
1667
+ qualifiesForPromotion: false,
1668
+ };
1669
+ const result = deepFreeze({
1670
+ ...core,
1671
+ evidenceReportDigest: hash(core.schema, core),
1672
+ });
1673
+ // Projection/bootstrap work must not turn a receipt that expired since the
1674
+ // row verification into a freshly authenticated returned report.
1675
+ await verifyEvolutionEvalReceipt(verifier, source.receipt, context);
1676
+ return result;
1677
+ }
1678
+
1679
+ export async function verifyPmExplorationEffectEvidenceReport(
1680
+ verifier,
1681
+ input,
1682
+ expected,
1683
+ ) {
1684
+ exact(input, ["source", "report"], "PM effect evidence report verification");
1685
+ const report = snapshotEffectEvidence(
1686
+ input.report,
1687
+ "PM effect evidence report",
1688
+ );
1689
+ const recreated = await buildPmExplorationEffectEvidenceReport(
1690
+ verifier,
1691
+ input.source,
1692
+ expected,
1693
+ );
1694
+ if (canonical(recreated) !== canonical(report)) {
1695
+ throw new Error(
1696
+ "PM effect evidence report does not match its authenticated source",
1697
+ );
1698
+ }
1699
+ return recreated;
1700
+ }
1701
+
1702
+ /**
1703
+ * Preserve a signed, budget-interrupted attempt in the preregistered test
1704
+ * denominator without inventing task outcomes or per-arm costs. Other thrown
1705
+ * failures have no final signed receipt and cannot use this contract.
1706
+ */
1707
+ export async function buildPmExplorationEffectInterruptedEvidence(
1708
+ verifier,
1709
+ input,
1710
+ expected,
1711
+ ) {
1712
+ const source = snapshotEffectEvidence(input, "PM interrupted effect source");
1713
+ exact(
1714
+ source,
1715
+ ["plan", "suite", "policy", "receipt"],
1716
+ "PM interrupted effect source",
1717
+ );
1718
+ const registered = snapshotEffectEvidence(
1719
+ expected,
1720
+ "PM interrupted effect registration",
1721
+ );
1722
+ exact(
1723
+ registered,
1724
+ ["planDigest", "targetMatrixRoot", "cellId", "runtimeId", "receiptContext"],
1725
+ "PM interrupted effect registration",
1726
+ );
1727
+ const { plan, context, evaluationContextDigest } =
1728
+ verifyPmEffectSourceRegistration(source, registered);
1729
+ const receipt = source.receipt;
1730
+ await verifyEvolutionEvalReceipt(verifier, receipt, context);
1731
+ const splitCounts = Object.fromEntries(
1732
+ ["training", "validation", "test"].map((split) => [
1733
+ split,
1734
+ source.suite.tasks.filter((task) => task.split === split).length,
1735
+ ]),
1736
+ );
1737
+ const plannedTestObservationsPerArm =
1738
+ plan.seeds.length * plan.testTaskIds.length;
1739
+ const plannedExecutionCount =
1740
+ (splitCounts.validation + splitCounts.test) * plan.seeds.length * 2;
1741
+ if (
1742
+ canonical(receipt.splitCounts) !== canonical(splitCounts) ||
1743
+ receipt.decision !== "rejected" ||
1744
+ canonical(receipt.reasonCodes) !== canonical(["total-budget-exceeded"]) ||
1745
+ receipt.validation !== null ||
1746
+ receipt.test !== null ||
1747
+ !Number.isSafeInteger(plannedExecutionCount) ||
1748
+ !Number.isSafeInteger(plannedTestObservationsPerArm) ||
1749
+ receipt.usage.executionCount < 1 ||
1750
+ receipt.usage.executionCount > plannedExecutionCount
1751
+ ) {
1752
+ throw new Error(
1753
+ "PM interrupted effect requires a signed partial budget rejection",
1754
+ );
1755
+ }
1756
+ const core = {
1757
+ schema: PM_EXPLORATION_EFFECT_INTERRUPTED_EVIDENCE_SCHEMA,
1758
+ planDigest: plan.planDigest,
1759
+ sourceEvalRunId: receipt.runId,
1760
+ sourceEvalReceiptDigest: receipt.receiptDigest,
1761
+ evaluationContextDigest,
1762
+ interruptionReason: "total-budget-exceeded",
1763
+ plannedExecutionCount,
1764
+ signedExecutionCount: receipt.usage.executionCount,
1765
+ plannedTestObservationsPerArm,
1766
+ authenticatedTestOutcomesPerArm: 0,
1767
+ unresolvedTestObservationsPerArm: plannedTestObservationsPerArm,
1768
+ aggregateSignedUsage: receipt.usage,
1769
+ outcomeAvailability: "unavailable",
1770
+ denominatorTreatment: "unresolved-blocks-promotion",
1771
+ statisticalEstimateAvailable: false,
1772
+ evidenceDecision: "threshold-not-met",
1773
+ blockingReasons: [
1774
+ "source-eval-budget-exceeded",
1775
+ "test-outcomes-unavailable",
1776
+ ],
1777
+ requiresIndependentPilotApproval: true,
1778
+ qualifiesForPromotion: false,
1779
+ };
1780
+ const evidence = deepFreeze({
1781
+ ...core,
1782
+ interruptedEvidenceDigest: hash(core.schema, core),
1783
+ });
1784
+ await verifyEvolutionEvalReceipt(verifier, receipt, context);
1785
+ return evidence;
1786
+ }
1787
+
1788
+ export async function verifyPmExplorationEffectInterruptedEvidence(
1789
+ verifier,
1790
+ input,
1791
+ expected,
1792
+ ) {
1793
+ exact(input, ["source", "evidence"], "PM interrupted effect verification");
1794
+ const evidence = snapshotEffectEvidence(
1795
+ input.evidence,
1796
+ "PM interrupted effect evidence",
1797
+ );
1798
+ const recreated = await buildPmExplorationEffectInterruptedEvidence(
1799
+ verifier,
1800
+ input.source,
1801
+ expected,
1802
+ );
1803
+ if (canonical(recreated) !== canonical(evidence))
1804
+ throw new Error(
1805
+ "PM interrupted effect evidence differs from signed source",
1806
+ );
1807
+ return recreated;
1808
+ }
1809
+
1810
+ /** A signed post-execution runtime rejection leaves every test outcome unresolved. */
1811
+ export async function buildPmExplorationEffectRuntimeFailureEvidence(
1812
+ verifier,
1813
+ input,
1814
+ expected,
1815
+ ) {
1816
+ const source = snapshotEffectEvidence(input, "PM runtime failure source");
1817
+ exact(
1818
+ source,
1819
+ ["plan", "suite", "policy", "receipt"],
1820
+ "PM runtime failure source",
1821
+ );
1822
+ const registered = snapshotEffectEvidence(
1823
+ expected,
1824
+ "PM runtime failure registration",
1825
+ );
1826
+ exact(
1827
+ registered,
1828
+ ["planDigest", "targetMatrixRoot", "cellId", "runtimeId", "receiptContext"],
1829
+ "PM runtime failure registration",
1830
+ );
1831
+ const { plan, context, evaluationContextDigest } =
1832
+ verifyPmEffectSourceRegistration(source, registered);
1833
+ const receipt = source.receipt;
1834
+ await verifyEvolutionEvalReceipt(verifier, receipt, context);
1835
+ const splitCounts = Object.fromEntries(
1836
+ ["training", "validation", "test"].map((split) => [
1837
+ split,
1838
+ source.suite.tasks.filter((task) => task.split === split).length,
1839
+ ]),
1840
+ );
1841
+ const plannedTestObservationsPerArm =
1842
+ plan.seeds.length * plan.testTaskIds.length;
1843
+ const plannedExecutionCount =
1844
+ (splitCounts.validation + splitCounts.test) * plan.seeds.length * 2;
1845
+ const failureReason = receipt.reasonCodes?.[0];
1846
+ if (
1847
+ canonical(receipt.splitCounts) !== canonical(splitCounts) ||
1848
+ receipt.decision !== "rejected" ||
1849
+ receipt.reasonCodes?.length !== 1 ||
1850
+ ![
1851
+ "runtime-execution-failed",
1852
+ "runtime-grader-failed",
1853
+ "runtime-safety-failed",
1854
+ ].includes(failureReason) ||
1855
+ receipt.validation !== null ||
1856
+ receipt.test !== null ||
1857
+ !Number.isSafeInteger(plannedExecutionCount) ||
1858
+ !Number.isSafeInteger(plannedTestObservationsPerArm) ||
1859
+ receipt.usage.executionCount < 1 ||
1860
+ receipt.usage.executionCount > plannedExecutionCount
1861
+ ) {
1862
+ throw new Error(
1863
+ "PM runtime failure requires a signed post-execution rejection",
1864
+ );
1865
+ }
1866
+ const core = {
1867
+ schema: PM_EXPLORATION_EFFECT_RUNTIME_FAILURE_EVIDENCE_SCHEMA,
1868
+ planDigest: plan.planDigest,
1869
+ sourceEvalRunId: receipt.runId,
1870
+ sourceEvalReceiptDigest: receipt.receiptDigest,
1871
+ evaluationContextDigest,
1872
+ failureReason,
1873
+ plannedExecutionCount,
1874
+ signedExecutionCount: receipt.usage.executionCount,
1875
+ plannedTestObservationsPerArm,
1876
+ authenticatedTestOutcomesPerArm: 0,
1877
+ unresolvedTestObservationsPerArm: plannedTestObservationsPerArm,
1878
+ aggregateSignedUsage: receipt.usage,
1879
+ outcomeAvailability: "unavailable",
1880
+ denominatorTreatment: "unresolved-blocks-promotion",
1881
+ statisticalEstimateAvailable: false,
1882
+ evidenceDecision: "threshold-not-met",
1883
+ blockingReasons: [
1884
+ `source-eval-${failureReason}`,
1885
+ "test-outcomes-unavailable",
1886
+ ],
1887
+ requiresIndependentPilotApproval: true,
1888
+ qualifiesForPromotion: false,
1889
+ };
1890
+ const evidence = deepFreeze({
1891
+ ...core,
1892
+ runtimeFailureEvidenceDigest: hash(core.schema, core),
1893
+ });
1894
+ await verifyEvolutionEvalReceipt(verifier, receipt, context);
1895
+ return evidence;
1896
+ }
1897
+
1898
+ export async function verifyPmExplorationEffectRuntimeFailureEvidence(
1899
+ verifier,
1900
+ input,
1901
+ expected,
1902
+ ) {
1903
+ exact(input, ["source", "evidence"], "PM runtime failure verification");
1904
+ const evidence = snapshotEffectEvidence(
1905
+ input.evidence,
1906
+ "PM runtime failure evidence",
1907
+ );
1908
+ const recreated = await buildPmExplorationEffectRuntimeFailureEvidence(
1909
+ verifier,
1910
+ input.source,
1911
+ expected,
1912
+ );
1913
+ if (canonical(recreated) !== canonical(evidence))
1914
+ throw new Error("PM runtime failure evidence differs from signed source");
1915
+ return recreated;
1916
+ }
1917
+
1918
+ /** Preserve a signed zero-execution preflight rejection without inventing outcomes. */
1919
+ export async function buildPmExplorationEffectPreflightRejectionEvidence(
1920
+ verifier,
1921
+ input,
1922
+ expected,
1923
+ ) {
1924
+ const source = snapshotEffectEvidence(input, "PM preflight rejection source");
1925
+ exact(
1926
+ source,
1927
+ ["plan", "suite", "policy", "receipt"],
1928
+ "PM preflight rejection source",
1929
+ );
1930
+ const registered = snapshotEffectEvidence(
1931
+ expected,
1932
+ "PM preflight rejection registration",
1933
+ );
1934
+ exact(
1935
+ registered,
1936
+ ["planDigest", "targetMatrixRoot", "cellId", "runtimeId", "receiptContext"],
1937
+ "PM preflight rejection registration",
1938
+ );
1939
+ const { plan, context, evaluationContextDigest } =
1940
+ verifyPmEffectSourceRegistration(source, registered);
1941
+ const receipt = source.receipt;
1942
+ await verifyEvolutionEvalReceipt(verifier, receipt, context);
1943
+ const splitCounts = Object.fromEntries(
1944
+ ["training", "validation", "test"].map((split) => [
1945
+ split,
1946
+ source.suite.tasks.filter((task) => task.split === split).length,
1947
+ ]),
1948
+ );
1949
+ const plannedTestObservationsPerArm =
1950
+ plan.seeds.length * plan.testTaskIds.length;
1951
+ const plannedExecutionCount =
1952
+ (splitCounts.validation + splitCounts.test) * plan.seeds.length * 2;
1953
+ const insufficient = ["training", "validation", "test"]
1954
+ .filter(
1955
+ (split) =>
1956
+ splitCounts[split] <
1957
+ source.policy[`min${split[0].toUpperCase()}${split.slice(1)}Tasks`],
1958
+ )
1959
+ .map((split) => `insufficient-${split}`);
1960
+ const expectedDecision = insufficient.length
1961
+ ? "needs-more-evidence"
1962
+ : "rejected";
1963
+ const expectedReasons = insufficient.length
1964
+ ? insufficient
1965
+ : ["execution-budget-insufficient"];
1966
+ const genuinePreflight =
1967
+ insufficient.length > 0 ||
1968
+ plannedExecutionCount > source.policy.maxExecutions;
1969
+ if (
1970
+ canonical(receipt.splitCounts) !== canonical(splitCounts) ||
1971
+ !Number.isSafeInteger(plannedExecutionCount) ||
1972
+ !Number.isSafeInteger(plannedTestObservationsPerArm) ||
1973
+ !genuinePreflight ||
1974
+ receipt.decision !== expectedDecision ||
1975
+ canonical(receipt.reasonCodes) !== canonical(expectedReasons) ||
1976
+ receipt.validation !== null ||
1977
+ receipt.test !== null
1978
+ ) {
1979
+ throw new Error("PM preflight requires a signed zero-execution rejection");
1980
+ }
1981
+ exact(receipt.usage, SIGNED_EVAL_USAGE_FIELDS, "PM preflight signed usage");
1982
+ if (SIGNED_EVAL_USAGE_FIELDS.some((field) => receipt.usage[field] !== 0)) {
1983
+ throw new Error("PM preflight rejection cannot claim execution usage");
1984
+ }
1985
+ const core = {
1986
+ schema: PM_EXPLORATION_EFFECT_PREFLIGHT_REJECTION_EVIDENCE_SCHEMA,
1987
+ planDigest: plan.planDigest,
1988
+ sourceEvalRunId: receipt.runId,
1989
+ sourceEvalReceiptDigest: receipt.receiptDigest,
1990
+ evaluationContextDigest,
1991
+ preflightDecision: receipt.decision,
1992
+ preflightReasonCodes: receipt.reasonCodes,
1993
+ plannedExecutionCount,
1994
+ signedExecutionCount: 0,
1995
+ plannedTestObservationsPerArm,
1996
+ authenticatedTestOutcomesPerArm: 0,
1997
+ unresolvedTestObservationsPerArm: plannedTestObservationsPerArm,
1998
+ aggregateSignedUsage: receipt.usage,
1999
+ outcomeAvailability: "unavailable",
2000
+ denominatorTreatment: "unresolved-blocks-promotion",
2001
+ statisticalEstimateAvailable: false,
2002
+ evidenceDecision: "threshold-not-met",
2003
+ blockingReasons: [
2004
+ "source-eval-preflight-rejected",
2005
+ "test-outcomes-unavailable",
2006
+ ],
2007
+ requiresIndependentPilotApproval: true,
2008
+ qualifiesForPromotion: false,
2009
+ };
2010
+ const evidence = deepFreeze({
2011
+ ...core,
2012
+ preflightRejectionEvidenceDigest: hash(core.schema, core),
2013
+ });
2014
+ await verifyEvolutionEvalReceipt(verifier, receipt, context);
2015
+ return evidence;
2016
+ }
2017
+
2018
+ export async function verifyPmExplorationEffectPreflightRejectionEvidence(
2019
+ verifier,
2020
+ input,
2021
+ expected,
2022
+ ) {
2023
+ exact(input, ["source", "evidence"], "PM preflight rejection verification");
2024
+ const evidence = snapshotEffectEvidence(
2025
+ input.evidence,
2026
+ "PM preflight rejection evidence",
2027
+ );
2028
+ const recreated = await buildPmExplorationEffectPreflightRejectionEvidence(
2029
+ verifier,
2030
+ input.source,
2031
+ expected,
2032
+ );
2033
+ if (canonical(recreated) !== canonical(evidence))
2034
+ throw new Error("PM preflight rejection differs from signed source");
2035
+ return recreated;
2036
+ }
2037
+
2038
+ /** Freeze the intended cohort slots before collection; persist the digest in an independent host record. */
2039
+ export function buildPmExplorationEffectSlotManifest(input) {
2040
+ const source = snapshotEffectEvidence(input, "PM effect slot manifest input");
2041
+ exact(
2042
+ source,
2043
+ ["plan", "cohortId", "slotIds"],
2044
+ "PM effect slot manifest input",
2045
+ );
2046
+ const plan = verifyPmExplorationEffectPlan(source.plan);
2047
+ if (plan.schema !== PM_EXPLORATION_EFFECT_PLAN_SCHEMA)
2048
+ throw new TypeError("PM effect slot manifest requires a v2 plan");
2049
+ if (
2050
+ !Array.isArray(source.slotIds) ||
2051
+ source.slotIds.length === 0 ||
2052
+ source.slotIds.length > 32
2053
+ ) {
2054
+ throw new TypeError("PM effect slot manifest requires 1-32 slots");
2055
+ }
2056
+ const slotIds = source.slotIds.map((value) =>
2057
+ boundedString(value, "PM effect slot ID"),
2058
+ );
2059
+ if (new Set(slotIds).size !== slotIds.length)
2060
+ throw new TypeError("PM effect slot manifest has duplicate slots");
2061
+ const core = {
2062
+ schema: PM_EXPLORATION_EFFECT_SLOT_MANIFEST_SCHEMA,
2063
+ planDigest: plan.planDigest,
2064
+ cohortId: boundedString(source.cohortId, "PM effect cohort ID"),
2065
+ slotIds,
2066
+ plannedTestObservationsPerArm: integer(
2067
+ slotIds.length * plan.seeds.length * plan.testTaskIds.length,
2068
+ "PM effect slot manifest planned observations",
2069
+ ),
2070
+ promotionAuthority: false,
2071
+ };
2072
+ return deepFreeze({
2073
+ ...core,
2074
+ manifestDigest: hash(core.schema, core),
2075
+ });
2076
+ }
2077
+
2078
+ export function verifyPmExplorationEffectSlotManifest(input) {
2079
+ exact(input, ["plan", "manifest"], "PM effect slot manifest verification");
2080
+ const manifest = snapshotEffectEvidence(
2081
+ input.manifest,
2082
+ "PM effect slot manifest",
2083
+ );
2084
+ exact(
2085
+ manifest,
2086
+ [
2087
+ "schema",
2088
+ "planDigest",
2089
+ "cohortId",
2090
+ "slotIds",
2091
+ "plannedTestObservationsPerArm",
2092
+ "promotionAuthority",
2093
+ "manifestDigest",
2094
+ ],
2095
+ "PM effect slot manifest",
2096
+ );
2097
+ const recreated = buildPmExplorationEffectSlotManifest({
2098
+ plan: input.plan,
2099
+ cohortId: manifest.cohortId,
2100
+ slotIds: manifest.slotIds,
2101
+ });
2102
+ if (canonical(recreated) !== canonical(manifest))
2103
+ throw new Error("PM effect slot manifest differs from its frozen plan");
2104
+ return recreated;
2105
+ }
2106
+
2107
+ /**
2108
+ * Reconcile only the attempts named in an independent host slot manifest.
2109
+ * A scheduled slot without a signed receipt is counted as unresolved, never
2110
+ * as an authenticated launch. Cohort completeness and promotion remain false.
2111
+ */
2112
+ export async function buildPmExplorationEffectAttemptCohort(
2113
+ verifier,
2114
+ input,
2115
+ expected,
2116
+ ) {
2117
+ const source = snapshotEffectEvidence(input, "PM attempt cohort source");
2118
+ const registered = snapshotEffectEvidence(
2119
+ expected,
2120
+ "PM attempt cohort registration",
2121
+ );
2122
+ exact(source, ["plan", "attempts"], "PM attempt cohort source");
2123
+ exact(
2124
+ registered,
2125
+ ["cohortId", "planDigest", "slots"],
2126
+ "PM attempt cohort registration",
2127
+ );
2128
+ const plan = verifyPmExplorationEffectPlan(source.plan);
2129
+ if (
2130
+ plan.schema !== PM_EXPLORATION_EFFECT_PLAN_SCHEMA ||
2131
+ plan.planDigest !== registered.planDigest
2132
+ ) {
2133
+ throw new Error("PM attempt cohort requires the registered v2 plan");
2134
+ }
2135
+ if (
2136
+ !Array.isArray(source.attempts) ||
2137
+ !Array.isArray(registered.slots) ||
2138
+ source.attempts.length === 0 ||
2139
+ source.attempts.length > 32 ||
2140
+ source.attempts.length !== registered.slots.length
2141
+ ) {
2142
+ throw new TypeError("PM attempt cohort must cover every registered slot");
2143
+ }
2144
+ const cohortId = boundedString(registered.cohortId, "PM cohortId");
2145
+ const slots = new Map();
2146
+ const registeredReceipts = new Set();
2147
+ for (const slot of registered.slots) {
2148
+ exact(
2149
+ slot,
2150
+ ["slotId", "receiptDigest", "context"],
2151
+ "PM attempt cohort slot",
2152
+ );
2153
+ const slotId = boundedString(slot.slotId, "PM attempt slotId");
2154
+ const receiptDigest =
2155
+ slot.receiptDigest === null
2156
+ ? null
2157
+ : digest(slot.receiptDigest, "PM attempt receiptDigest");
2158
+ if (
2159
+ (receiptDigest === null && slot.context !== null) ||
2160
+ (receiptDigest !== null && slot.context === null)
2161
+ )
2162
+ throw new Error("PM attempt slot receipt and context must agree");
2163
+ if (
2164
+ slots.has(slotId) ||
2165
+ (receiptDigest !== null && registeredReceipts.has(receiptDigest))
2166
+ )
2167
+ throw new Error("PM attempt cohort registration is duplicated");
2168
+ slots.set(slotId, slot);
2169
+ if (receiptDigest !== null) registeredReceipts.add(receiptDigest);
2170
+ }
2171
+ const seen = new Set();
2172
+ const runIds = new Set();
2173
+ const attempts = [];
2174
+ const perArmDenominator = integer(
2175
+ plan.seeds.length * plan.testTaskIds.length * source.attempts.length,
2176
+ "PM cohort per-arm denominator",
2177
+ );
2178
+ const totals = {
2179
+ baseline: { authenticatedOutcomes: 0, verifiedPasses: 0, unresolved: 0 },
2180
+ candidate: { authenticatedOutcomes: 0, verifiedPasses: 0, unresolved: 0 },
2181
+ };
2182
+ let budgetInterruptedCount = 0;
2183
+ let runtimeFailedCount = 0;
2184
+ let missingReceiptCount = 0;
2185
+ let preflightRejectedCount = 0;
2186
+ let failedCompleteCount = 0;
2187
+ for (const attempt of source.attempts) {
2188
+ exact(
2189
+ attempt,
2190
+ ["slotId", "kind", "source", "evidence"],
2191
+ "PM attempt cohort entry",
2192
+ );
2193
+ const slotId = boundedString(attempt.slotId, "PM attempt slotId");
2194
+ const slot = slots.get(slotId);
2195
+ if (!slot || seen.has(slotId)) {
2196
+ throw new Error("PM attempt cohort contains a foreign or duplicate slot");
2197
+ }
2198
+ seen.add(slotId);
2199
+ if (attempt.kind === "receipt-unavailable") {
2200
+ if (
2201
+ slot.receiptDigest !== null ||
2202
+ slot.context !== null ||
2203
+ attempt.source !== null ||
2204
+ attempt.evidence !== null
2205
+ ) {
2206
+ throw new Error("PM missing receipt slot has a substituted source");
2207
+ }
2208
+ missingReceiptCount += 1;
2209
+ for (const arm of ["baseline", "candidate"]) {
2210
+ totals[arm].unresolved = integer(
2211
+ totals[arm].unresolved + plan.seeds.length * plan.testTaskIds.length,
2212
+ "PM cohort unresolved outcomes",
2213
+ );
2214
+ }
2215
+ attempts.push({
2216
+ slotId,
2217
+ kind: attempt.kind,
2218
+ runId: null,
2219
+ receiptDigest: null,
2220
+ evidenceDigest: null,
2221
+ });
2222
+ continue;
2223
+ }
2224
+ if (
2225
+ slot.receiptDigest === null ||
2226
+ canonical(attempt.source?.plan) !== canonical(plan)
2227
+ ) {
2228
+ throw new Error("PM attempt cohort contains a foreign signed source");
2229
+ }
2230
+ let verified;
2231
+ let evidenceDigest;
2232
+ if (attempt.kind === "complete") {
2233
+ verified = await verifyPmExplorationEffectEvidenceReport(
2234
+ verifier,
2235
+ { source: attempt.source, report: attempt.evidence },
2236
+ slot.context,
2237
+ );
2238
+ evidenceDigest = verified.evidenceReportDigest;
2239
+ if (verified.evidenceDecision === "threshold-not-met")
2240
+ failedCompleteCount += 1;
2241
+ for (const arm of ["baseline", "candidate"]) {
2242
+ totals[arm].authenticatedOutcomes = integer(
2243
+ totals[arm].authenticatedOutcomes + verified.report[arm].sampleCount,
2244
+ "PM cohort authenticated outcomes",
2245
+ );
2246
+ totals[arm].verifiedPasses = integer(
2247
+ totals[arm].verifiedPasses + verified.report[arm].passCount,
2248
+ "PM cohort verified passes",
2249
+ );
2250
+ }
2251
+ } else if (attempt.kind === "budget-interrupted") {
2252
+ verified = await verifyPmExplorationEffectInterruptedEvidence(
2253
+ verifier,
2254
+ { source: attempt.source, evidence: attempt.evidence },
2255
+ slot.context,
2256
+ );
2257
+ evidenceDigest = verified.interruptedEvidenceDigest;
2258
+ budgetInterruptedCount += 1;
2259
+ for (const arm of ["baseline", "candidate"]) {
2260
+ totals[arm].unresolved = integer(
2261
+ totals[arm].unresolved + verified.unresolvedTestObservationsPerArm,
2262
+ "PM cohort unresolved outcomes",
2263
+ );
2264
+ }
2265
+ } else if (attempt.kind === "runtime-failed") {
2266
+ verified = await verifyPmExplorationEffectRuntimeFailureEvidence(
2267
+ verifier,
2268
+ { source: attempt.source, evidence: attempt.evidence },
2269
+ slot.context,
2270
+ );
2271
+ evidenceDigest = verified.runtimeFailureEvidenceDigest;
2272
+ runtimeFailedCount += 1;
2273
+ for (const arm of ["baseline", "candidate"]) {
2274
+ totals[arm].unresolved = integer(
2275
+ totals[arm].unresolved + verified.unresolvedTestObservationsPerArm,
2276
+ "PM cohort unresolved outcomes",
2277
+ );
2278
+ }
2279
+ } else if (attempt.kind === "preflight-rejected") {
2280
+ verified = await verifyPmExplorationEffectPreflightRejectionEvidence(
2281
+ verifier,
2282
+ { source: attempt.source, evidence: attempt.evidence },
2283
+ slot.context,
2284
+ );
2285
+ evidenceDigest = verified.preflightRejectionEvidenceDigest;
2286
+ preflightRejectedCount += 1;
2287
+ for (const arm of ["baseline", "candidate"]) {
2288
+ totals[arm].unresolved = integer(
2289
+ totals[arm].unresolved + verified.unresolvedTestObservationsPerArm,
2290
+ "PM cohort unresolved outcomes",
2291
+ );
2292
+ }
2293
+ } else {
2294
+ throw new TypeError("PM attempt cohort kind is invalid");
2295
+ }
2296
+ if (
2297
+ verified.sourceEvalReceiptDigest !== slot.receiptDigest ||
2298
+ runIds.has(verified.sourceEvalRunId)
2299
+ ) {
2300
+ throw new Error("PM attempt cohort receipt is substituted or reused");
2301
+ }
2302
+ runIds.add(verified.sourceEvalRunId);
2303
+ attempts.push({
2304
+ slotId,
2305
+ kind: attempt.kind,
2306
+ runId: verified.sourceEvalRunId,
2307
+ receiptDigest: verified.sourceEvalReceiptDigest,
2308
+ evidenceDigest,
2309
+ });
2310
+ }
2311
+ for (const arm of ["baseline", "candidate"]) {
2312
+ if (
2313
+ totals[arm].authenticatedOutcomes + totals[arm].unresolved !==
2314
+ perArmDenominator
2315
+ ) {
2316
+ throw new Error("PM attempt cohort denominator is incomplete");
2317
+ }
2318
+ }
2319
+ attempts.sort(
2320
+ (left, right) =>
2321
+ registered.slots.findIndex((slot) => slot.slotId === left.slotId) -
2322
+ registered.slots.findIndex((slot) => slot.slotId === right.slotId),
2323
+ );
2324
+ // A receipt may expire while later slots are being checked.
2325
+ for (const attempt of source.attempts) {
2326
+ if (attempt.kind === "receipt-unavailable") continue;
2327
+ const slot = slots.get(attempt.slotId);
2328
+ await verifyEvolutionEvalReceipt(
2329
+ verifier,
2330
+ attempt.source.receipt,
2331
+ slot.context.receiptContext,
2332
+ );
2333
+ }
2334
+ const core = {
2335
+ schema:
2336
+ preflightRejectedCount > 0
2337
+ ? PM_EXPLORATION_EFFECT_ATTEMPT_COHORT_V4_SCHEMA
2338
+ : missingReceiptCount > 0
2339
+ ? PM_EXPLORATION_EFFECT_ATTEMPT_COHORT_V3_SCHEMA
2340
+ : runtimeFailedCount > 0
2341
+ ? PM_EXPLORATION_EFFECT_ATTEMPT_COHORT_V2_SCHEMA
2342
+ : PM_EXPLORATION_EFFECT_ATTEMPT_COHORT_SCHEMA,
2343
+ cohortId,
2344
+ planDigest: plan.planDigest,
2345
+ registeredSlotCount: slots.size,
2346
+ authenticatedAttemptCount: attempts.length - missingReceiptCount,
2347
+ budgetInterruptedCount,
2348
+ ...(runtimeFailedCount > 0 || missingReceiptCount > 0
2349
+ ? { runtimeFailedCount }
2350
+ : {}),
2351
+ ...(missingReceiptCount > 0 ? { missingReceiptCount } : {}),
2352
+ ...(preflightRejectedCount > 0 ? { preflightRejectedCount } : {}),
2353
+ failedCompleteCount,
2354
+ perArmDenominator,
2355
+ baseline: totals.baseline,
2356
+ candidate: totals.candidate,
2357
+ attempts,
2358
+ unresolvedBlocksPromotion:
2359
+ budgetInterruptedCount > 0 ||
2360
+ runtimeFailedCount > 0 ||
2361
+ missingReceiptCount > 0 ||
2362
+ preflightRejectedCount > 0,
2363
+ allRegisteredOutcomesAvailable:
2364
+ budgetInterruptedCount === 0 &&
2365
+ runtimeFailedCount === 0 &&
2366
+ missingReceiptCount === 0 &&
2367
+ preflightRejectedCount === 0,
2368
+ statisticalEstimateAvailable: false,
2369
+ cohortCompletenessAuthenticated: false,
2370
+ reportAuthenticated: false,
2371
+ evidenceDecision:
2372
+ budgetInterruptedCount > 0 ||
2373
+ runtimeFailedCount > 0 ||
2374
+ missingReceiptCount > 0 ||
2375
+ preflightRejectedCount > 0 ||
2376
+ failedCompleteCount > 0
2377
+ ? "threshold-not-met"
2378
+ : "insufficient-evidence",
2379
+ requiresIndependentPilotApproval: true,
2380
+ qualifiesForPromotion: false,
2381
+ };
2382
+ return deepFreeze({ ...core, cohortDigest: hash(core.schema, core) });
2383
+ }
2384
+
2385
+ export async function verifyPmExplorationEffectAttemptCohort(
2386
+ verifier,
2387
+ input,
2388
+ expected,
2389
+ ) {
2390
+ exact(input, ["source", "cohort"], "PM attempt cohort verification");
2391
+ const cohort = snapshotEffectEvidence(input.cohort, "PM attempt cohort");
2392
+ const recreated = await buildPmExplorationEffectAttemptCohort(
2393
+ verifier,
2394
+ input.source,
2395
+ expected,
2396
+ );
2397
+ if (canonical(recreated) !== canonical(cohort))
2398
+ throw new Error("PM attempt cohort differs from its signed sources");
2399
+ return recreated;
2400
+ }
2401
+
2402
+ /** Associate the independently saved slot schedule with a reverified cohort. */
2403
+ export async function buildPmExplorationEffectManifestBoundCohort(
2404
+ verifier,
2405
+ input,
2406
+ expected,
2407
+ ) {
2408
+ exact(
2409
+ input,
2410
+ ["cohortSource", "cohort", "slotManifest"],
2411
+ "PM manifest-bound cohort source",
2412
+ );
2413
+ exact(
2414
+ expected,
2415
+ ["cohort", "slotManifestDigest"],
2416
+ "PM manifest-bound cohort registration",
2417
+ );
2418
+ const cohortSource = snapshotEffectEvidence(
2419
+ input.cohortSource,
2420
+ "PM manifest-bound cohort source",
2421
+ );
2422
+ const cohort = snapshotEffectEvidence(input.cohort, "PM registered cohort");
2423
+ const slotManifest = snapshotEffectEvidence(
2424
+ input.slotManifest,
2425
+ "PM registered slot manifest",
2426
+ );
2427
+ const registeredCohort = snapshotEffectEvidence(
2428
+ expected.cohort,
2429
+ "PM registered cohort context",
2430
+ );
2431
+ const manifestDigest = digest(
2432
+ expected.slotManifestDigest,
2433
+ "PM registered slot manifest digest",
2434
+ );
2435
+ const manifest = verifyPmExplorationEffectSlotManifest({
2436
+ plan: cohortSource.plan,
2437
+ manifest: slotManifest,
2438
+ });
2439
+ if (
2440
+ manifest.manifestDigest !== manifestDigest ||
2441
+ manifest.cohortId !== registeredCohort.cohortId ||
2442
+ !Array.isArray(registeredCohort.slots) ||
2443
+ canonical(manifest.slotIds) !==
2444
+ canonical(registeredCohort.slots.map((slot) => slot.slotId))
2445
+ ) {
2446
+ throw new Error("PM cohort slots differ from the frozen manifest");
2447
+ }
2448
+ const verified = await verifyPmExplorationEffectAttemptCohort(
2449
+ verifier,
2450
+ { source: cohortSource, cohort },
2451
+ registeredCohort,
2452
+ );
2453
+ if (
2454
+ verified.cohortId !== manifest.cohortId ||
2455
+ verified.planDigest !== manifest.planDigest ||
2456
+ verified.perArmDenominator !== manifest.plannedTestObservationsPerArm
2457
+ ) {
2458
+ throw new Error("PM cohort denominator differs from the frozen manifest");
2459
+ }
2460
+ const core = {
2461
+ schema: PM_EXPLORATION_EFFECT_MANIFEST_BOUND_COHORT_SCHEMA,
2462
+ planDigest: manifest.planDigest,
2463
+ cohortId: manifest.cohortId,
2464
+ slotManifestDigest: manifest.manifestDigest,
2465
+ cohortDigest: verified.cohortDigest,
2466
+ registeredSlotCount: verified.registeredSlotCount,
2467
+ perArmDenominator: verified.perArmDenominator,
2468
+ slotScheduleBound: true,
2469
+ slotScheduleAuthenticated: false,
2470
+ cohortCompletenessAuthenticated: false,
2471
+ reportAuthenticated: false,
2472
+ evidenceDecision: verified.evidenceDecision,
2473
+ requiresIndependentPilotApproval: true,
2474
+ qualifiesForPromotion: false,
2475
+ };
2476
+ return deepFreeze({
2477
+ ...core,
2478
+ bindingDigest: hash(core.schema, core),
2479
+ });
2480
+ }
2481
+
2482
+ export async function verifyPmExplorationEffectManifestBoundCohort(
2483
+ verifier,
2484
+ input,
2485
+ expected,
2486
+ ) {
2487
+ exact(
2488
+ input,
2489
+ ["cohortSource", "cohort", "slotManifest", "binding"],
2490
+ "PM manifest-bound cohort verification",
2491
+ );
2492
+ const binding = snapshotEffectEvidence(
2493
+ input.binding,
2494
+ "PM manifest-bound cohort",
2495
+ );
2496
+ const recreated = await buildPmExplorationEffectManifestBoundCohort(
2497
+ verifier,
2498
+ {
2499
+ cohortSource: input.cohortSource,
2500
+ cohort: input.cohort,
2501
+ slotManifest: input.slotManifest,
2502
+ },
2503
+ expected,
2504
+ );
2505
+ if (canonical(recreated) !== canonical(binding))
2506
+ throw new Error("PM manifest-bound cohort differs from its sources");
2507
+ return recreated;
2508
+ }
2509
+
2510
+ /**
2511
+ * Sum only usage carried by independently verified final Eval receipts in a
2512
+ * frozen cohort. Missing receipts and preparation costs stay explicitly unknown.
2513
+ */
2514
+ export async function buildPmExplorationEffectCohortUsageEvidence(
2515
+ verifier,
2516
+ input,
2517
+ expected,
2518
+ ) {
2519
+ const source = snapshotEffectEvidence(input, "PM cohort usage source");
2520
+ const registered = snapshotEffectEvidence(
2521
+ expected,
2522
+ "PM cohort usage registration",
2523
+ );
2524
+ exact(
2525
+ source,
2526
+ ["cohortSource", "cohort", "slotManifest", "binding"],
2527
+ "PM cohort usage source",
2528
+ );
2529
+ exact(
2530
+ registered,
2531
+ ["cohort", "slotManifestDigest"],
2532
+ "PM cohort usage registration",
2533
+ );
2534
+ const binding = await verifyPmExplorationEffectManifestBoundCohort(
2535
+ verifier,
2536
+ source,
2537
+ registered,
2538
+ );
2539
+ const knownSignedUsage = Object.fromEntries(
2540
+ SIGNED_EVAL_USAGE_FIELDS.map((field) => [field, 0]),
2541
+ );
2542
+ const slots = new Map(
2543
+ registered.cohort.slots.map((slot) => [slot.slotId, slot]),
2544
+ );
2545
+ let signedReceiptCount = 0;
2546
+ let slotsWithoutSignedUsage = 0;
2547
+ for (const attempt of source.cohortSource.attempts) {
2548
+ if (attempt.kind === "receipt-unavailable") {
2549
+ slotsWithoutSignedUsage += 1;
2550
+ continue;
2551
+ }
2552
+ const usage = attempt.source.receipt.usage;
2553
+ exact(usage, SIGNED_EVAL_USAGE_FIELDS, "signed Eval usage");
2554
+ for (const field of SIGNED_EVAL_USAGE_FIELDS) {
2555
+ knownSignedUsage[field] = integer(
2556
+ knownSignedUsage[field] + integer(usage[field], `signed ${field}`),
2557
+ `cohort ${field}`,
2558
+ );
2559
+ }
2560
+ signedReceiptCount += 1;
2561
+ }
2562
+ if (
2563
+ signedReceiptCount !== source.cohort.authenticatedAttemptCount ||
2564
+ signedReceiptCount + slotsWithoutSignedUsage !== binding.registeredSlotCount
2565
+ ) {
2566
+ throw new Error("PM cohort signed usage coverage differs from its slots");
2567
+ }
2568
+ // A receipt can expire while other slots are being aggregated.
2569
+ for (const attempt of source.cohortSource.attempts) {
2570
+ if (attempt.kind === "receipt-unavailable") continue;
2571
+ await verifyEvolutionEvalReceipt(
2572
+ verifier,
2573
+ attempt.source.receipt,
2574
+ slots.get(attempt.slotId).context.receiptContext,
2575
+ );
2576
+ }
2577
+ const core = {
2578
+ schema: PM_EXPLORATION_EFFECT_COHORT_USAGE_EVIDENCE_SCHEMA,
2579
+ planDigest: binding.planDigest,
2580
+ cohortId: binding.cohortId,
2581
+ slotManifestDigest: binding.slotManifestDigest,
2582
+ cohortDigest: binding.cohortDigest,
2583
+ bindingDigest: binding.bindingDigest,
2584
+ registeredSlotCount: binding.registeredSlotCount,
2585
+ signedReceiptCount,
2586
+ slotsWithoutSignedUsage,
2587
+ knownSignedUsage,
2588
+ usageScope: "signed-eval-execution-aggregate-only",
2589
+ providerSettlementsAuthenticated: false,
2590
+ preparationCostsAuthenticated: false,
2591
+ totalCostAuthenticated: false,
2592
+ cohortCompletenessAuthenticated: false,
2593
+ reportAuthenticated: false,
2594
+ evidenceDecision: binding.evidenceDecision,
2595
+ requiresIndependentPilotApproval: true,
2596
+ qualifiesForPromotion: false,
2597
+ };
2598
+ return deepFreeze({
2599
+ ...core,
2600
+ usageEvidenceDigest: hash(core.schema, core),
2601
+ });
2602
+ }
2603
+
2604
+ export async function verifyPmExplorationEffectCohortUsageEvidence(
2605
+ verifier,
2606
+ input,
2607
+ expected,
2608
+ ) {
2609
+ exact(
2610
+ input,
2611
+ ["cohortSource", "cohort", "slotManifest", "binding", "usageEvidence"],
2612
+ "PM cohort usage verification",
2613
+ );
2614
+ const evidence = snapshotEffectEvidence(
2615
+ input.usageEvidence,
2616
+ "PM cohort usage evidence",
2617
+ );
2618
+ const recreated = await buildPmExplorationEffectCohortUsageEvidence(
2619
+ verifier,
2620
+ {
2621
+ cohortSource: input.cohortSource,
2622
+ cohort: input.cohort,
2623
+ slotManifest: input.slotManifest,
2624
+ binding: input.binding,
2625
+ },
2626
+ expected,
2627
+ );
2628
+ if (canonical(recreated) !== canonical(evidence))
2629
+ throw new Error("PM cohort usage differs from its signed sources");
2630
+ return recreated;
2631
+ }
2632
+
2633
+ /**
2634
+ * Re-read both independent authorities and bind their results to the same
2635
+ * frozen plan and preparation declaration. Provider evidence is partial, so
2636
+ * this bundle must never upgrade the effect report's authentication decision.
2637
+ */
2638
+ export async function buildPmExplorationEffectProviderEvidenceBundle(
2639
+ verifier,
2640
+ providerAdapter,
2641
+ input,
2642
+ expected,
2643
+ ) {
2644
+ exact(
2645
+ input,
2646
+ ["effectSource", "effectReport", "providerSource", "providerEvidence"],
2647
+ "PM effect provider bundle source",
2648
+ );
2649
+ exact(
2650
+ expected,
2651
+ ["effect", "provider"],
2652
+ "PM effect provider bundle registration",
2653
+ );
2654
+ // Capture all caller data before the first asynchronous receipt check.
2655
+ const effectSource = snapshotEffectEvidence(
2656
+ input.effectSource,
2657
+ "PM effect provider bundle effect source",
2658
+ );
2659
+ const effectReport = snapshotEffectEvidence(
2660
+ input.effectReport,
2661
+ "PM effect provider bundle effect report",
2662
+ );
2663
+ const providerSource = snapshotEffectEvidence(
2664
+ input.providerSource,
2665
+ "PM effect provider bundle provider source",
2666
+ );
2667
+ const providerEvidence = snapshotEffectEvidence(
2668
+ input.providerEvidence,
2669
+ "PM effect provider bundle provider evidence",
2670
+ );
2671
+ const effectExpected = snapshotEffectEvidence(
2672
+ expected.effect,
2673
+ "PM effect provider bundle effect registration",
2674
+ );
2675
+ const providerExpected = snapshotEffectEvidence(
2676
+ expected.provider,
2677
+ "PM effect provider bundle provider registration",
2678
+ );
2679
+ if (
2680
+ canonical(effectSource.plan) !== canonical(providerSource.plan) ||
2681
+ canonical(effectSource.phaseUsage) !== canonical(providerSource.phaseUsage)
2682
+ ) {
2683
+ throw new Error("PM provider evidence differs from effect plan or phases");
2684
+ }
2685
+ let verifiedProvider = verifyPmExplorationPreparationProviderEvidence(
2686
+ providerAdapter,
2687
+ { source: providerSource, evidence: providerEvidence },
2688
+ providerExpected,
2689
+ );
2690
+ const verifiedEffect = await verifyPmExplorationEffectEvidenceReport(
2691
+ verifier,
2692
+ { source: effectSource, report: effectReport },
2693
+ effectExpected,
2694
+ );
2695
+ // Recheck receipt freshness, then read the durable provider artifact last.
2696
+ await verifyEvolutionEvalReceipt(
2697
+ verifier,
2698
+ effectSource.receipt,
2699
+ effectExpected.receiptContext,
2700
+ );
2701
+ verifiedProvider = verifyPmExplorationPreparationProviderEvidence(
2702
+ providerAdapter,
2703
+ { source: providerSource, evidence: providerEvidence },
2704
+ providerExpected,
2705
+ );
2706
+ if (
2707
+ verifiedEffect.planDigest !== verifiedProvider.planDigest ||
2708
+ verifiedEffect.phaseUsageDigest !== verifiedProvider.phaseUsageDigest
2709
+ ) {
2710
+ throw new Error("PM provider evidence is not bound to the effect report");
2711
+ }
2712
+ const core = {
2713
+ schema: PM_EXPLORATION_EFFECT_PROVIDER_EVIDENCE_BUNDLE_SCHEMA,
2714
+ planDigest: verifiedEffect.planDigest,
2715
+ phaseUsageDigest: verifiedEffect.phaseUsageDigest,
2716
+ effectEvidenceReportDigest: verifiedEffect.evidenceReportDigest,
2717
+ providerEvidenceDigest: verifiedProvider.evidenceDigest,
2718
+ providerPhaseTotals: verifiedProvider.phaseTotals,
2719
+ evidenceDecision: verifiedEffect.evidenceDecision,
2720
+ blockingReasons: verifiedEffect.blockingReasons,
2721
+ executionEvidenceAuthenticated: true,
2722
+ providerCostsAuthenticatedForRegisteredRequests: true,
2723
+ preparationEvidenceAuthenticated: false,
2724
+ reportAuthenticated: false,
2725
+ requiresIndependentPilotApproval: true,
2726
+ qualifiesForPromotion: false,
2727
+ };
2728
+ return deepFreeze({
2729
+ ...core,
2730
+ bundleDigest: hash(core.schema, core),
2731
+ });
2732
+ }
2733
+
2734
+ export async function verifyPmExplorationEffectProviderEvidenceBundle(
2735
+ verifier,
2736
+ providerAdapter,
2737
+ input,
2738
+ expected,
2739
+ ) {
2740
+ exact(
2741
+ input,
2742
+ [
2743
+ "effectSource",
2744
+ "effectReport",
2745
+ "providerSource",
2746
+ "providerEvidence",
2747
+ "bundle",
2748
+ ],
2749
+ "PM effect provider bundle verification",
2750
+ );
2751
+ const bundle = snapshotEffectEvidence(
2752
+ input.bundle,
2753
+ "PM effect provider bundle",
2754
+ );
2755
+ const recreated = await buildPmExplorationEffectProviderEvidenceBundle(
2756
+ verifier,
2757
+ providerAdapter,
2758
+ {
2759
+ effectSource: input.effectSource,
2760
+ effectReport: input.effectReport,
2761
+ providerSource: input.providerSource,
2762
+ providerEvidence: input.providerEvidence,
2763
+ },
2764
+ expected,
2765
+ );
2766
+ if (canonical(recreated) !== canonical(bundle))
2767
+ throw new Error(
2768
+ "PM effect provider bundle differs from authenticated sources",
2769
+ );
2770
+ return recreated;
2771
+ }
2772
+
808
2773
  /** Trusted dataset-author input; store the result outside the Actor workspace. */
809
2774
  export function buildPmExplorationSuite(input) {
810
2775
  exact(input, ["suiteId", "datasetVersion", "tasks"], "PM suite input");