@tangle-network/agent-eval 0.171.0 → 0.172.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (114) hide show
  1. package/CHANGELOG.md +85 -0
  2. package/README.md +3 -0
  3. package/dist/adapters/http.d.ts +2 -2
  4. package/dist/analyst/index.d.ts +5 -5
  5. package/dist/analyst/index.js +2 -2
  6. package/dist/{experiment-tracker-Dm8yQMqb.d.ts → attestation-CJBGmMVh.d.ts} +78 -2
  7. package/dist/attestation-CJBGmMVh.d.ts.map +1 -0
  8. package/dist/{experiment-tracker-BKEumQug.js → attestation-XSUpbc4o.js} +96 -2
  9. package/dist/attestation-XSUpbc4o.js.map +1 -0
  10. package/dist/{benchmark-command-D8k3Gf0J.js → benchmark-command--qeZUHbu.js} +2 -2
  11. package/dist/{benchmark-command-D8k3Gf0J.js.map → benchmark-command--qeZUHbu.js.map} +1 -1
  12. package/dist/benchmarks/index.d.ts +4 -4
  13. package/dist/benchmarks/index.js +2 -2
  14. package/dist/bounded-process-VIi0KSL2.js +212 -0
  15. package/dist/bounded-process-VIi0KSL2.js.map +1 -0
  16. package/dist/builder-eval/index.js +39 -98
  17. package/dist/builder-eval/index.js.map +1 -1
  18. package/dist/campaign/index.d.ts +8 -7
  19. package/dist/campaign/index.js +4 -4
  20. package/dist/{campaign-B72njjHj.js → campaign-Dp35pBbS.js} +4 -4
  21. package/dist/{campaign-B72njjHj.js.map → campaign-Dp35pBbS.js.map} +1 -1
  22. package/dist/canonical-CFpojCN5.d.ts +31 -0
  23. package/dist/canonical-CFpojCN5.d.ts.map +1 -0
  24. package/dist/{chat-client-DEtybj5i.js → chat-client-DI79OPye.js} +2 -2
  25. package/dist/{chat-client-DEtybj5i.js.map → chat-client-DI79OPye.js.map} +1 -1
  26. package/dist/cli.js +1 -1
  27. package/dist/{client-CDtcZ3p9.d.ts → client-Df7wdslk.d.ts} +2 -2
  28. package/dist/{client-CDtcZ3p9.d.ts.map → client-Df7wdslk.d.ts.map} +1 -1
  29. package/dist/contract/index.d.ts +9 -9
  30. package/dist/contract/index.js +4 -4
  31. package/dist/{default-registry-B0s2zU-s.d.ts → default-registry-XxedTLwu.d.ts} +3 -3
  32. package/dist/{default-registry-B0s2zU-s.d.ts.map → default-registry-XxedTLwu.d.ts.map} +1 -1
  33. package/dist/{define-agent-eval-Cjy2yhqP.d.ts → define-agent-eval-0wW7gFhr.d.ts} +4 -4
  34. package/dist/{define-agent-eval-Cjy2yhqP.d.ts.map → define-agent-eval-0wW7gFhr.d.ts.map} +1 -1
  35. package/dist/{define-agent-eval-Dy8QgxAI.js → define-agent-eval-jS8xj_Q_.js} +2 -2
  36. package/dist/{define-agent-eval-Dy8QgxAI.js.map → define-agent-eval-jS8xj_Q_.js.map} +1 -1
  37. package/dist/descriptive-B2iPaT9J.d.ts +89 -0
  38. package/dist/descriptive-B2iPaT9J.d.ts.map +1 -0
  39. package/dist/{engine-CAmTUk52.d.ts → engine-BfRay1qD.d.ts} +2 -2
  40. package/dist/{engine-CAmTUk52.d.ts.map → engine-BfRay1qD.d.ts.map} +1 -1
  41. package/dist/experiment/index.d.ts +3 -54
  42. package/dist/experiment/index.d.ts.map +1 -1
  43. package/dist/experiment/index.js +1 -95
  44. package/dist/experiment/index.js.map +1 -1
  45. package/dist/{heldout-gate-Dh2b62w8.d.ts → heldout-gate-JgNRDZwZ.d.ts} +3 -3
  46. package/dist/{heldout-gate-Dh2b62w8.d.ts.map → heldout-gate-JgNRDZwZ.d.ts.map} +1 -1
  47. package/dist/hosted/index.d.ts +1 -1
  48. package/dist/{index-lfaSeKSD.d.ts → index-D-UdhAmg.d.ts} +3 -31
  49. package/dist/index-D-UdhAmg.d.ts.map +1 -0
  50. package/dist/{index-DT73JraI.d.ts → index-DDAPhUJJ.d.ts} +4 -4
  51. package/dist/{index-DT73JraI.d.ts.map → index-DDAPhUJJ.d.ts.map} +1 -1
  52. package/dist/{index-fNXZMCzX.d.ts → index-DnglhM0A.d.ts} +9 -9
  53. package/dist/{index-fNXZMCzX.d.ts.map → index-DnglhM0A.d.ts.map} +1 -1
  54. package/dist/{index-8VIogTyS.d.ts → index-_vPrVMRX.d.ts} +6 -6
  55. package/dist/{index-8VIogTyS.d.ts.map → index-_vPrVMRX.d.ts.map} +1 -1
  56. package/dist/index.d.ts +154 -15
  57. package/dist/index.d.ts.map +1 -1
  58. package/dist/index.js +7 -6
  59. package/dist/index.js.map +1 -1
  60. package/dist/{integrity-CyWSSoQS.js → integrity-BWywb34E.js} +34 -11
  61. package/dist/{integrity-CyWSSoQS.js.map → integrity-BWywb34E.js.map} +1 -1
  62. package/dist/ledger-core/index.d.ts +2 -1
  63. package/dist/{llm-judge-BtJ2Sfk_.js → llm-judge-aQHIk5_-.js} +16 -10
  64. package/dist/llm-judge-aQHIk5_-.js.map +1 -0
  65. package/dist/{matrix-DiHmUobV.d.ts → matrix-Ch8JO1pG.d.ts} +2 -2
  66. package/dist/{matrix-DiHmUobV.d.ts.map → matrix-Ch8JO1pG.d.ts.map} +1 -1
  67. package/dist/meta-eval/index.d.ts +162 -2
  68. package/dist/meta-eval/index.d.ts.map +1 -1
  69. package/dist/meta-eval/index.js +287 -2
  70. package/dist/meta-eval/index.js.map +1 -1
  71. package/dist/multishot/golden/index.d.ts +1 -1
  72. package/dist/multishot/index.d.ts +2 -2
  73. package/dist/openapi.json +1 -1
  74. package/dist/{produced-state-Cm6DU_Ao.js → produced-state-CxmbFxFd.js} +2 -2
  75. package/dist/{produced-state-Cm6DU_Ao.js.map → produced-state-CxmbFxFd.js.map} +1 -1
  76. package/dist/{promotion-policy-WSXtBgBb.d.ts → promotion-policy-BBBcz5_3.d.ts} +2 -2
  77. package/dist/{promotion-policy-WSXtBgBb.d.ts.map → promotion-policy-BBBcz5_3.d.ts.map} +1 -1
  78. package/dist/{provenance-CafMdZKM.d.ts → provenance-Dp-vvyrU.d.ts} +13 -5
  79. package/dist/provenance-Dp-vvyrU.d.ts.map +1 -0
  80. package/dist/rl.d.ts +1 -1
  81. package/dist/sandbox-harness-BlSOu4LX.d.ts.map +1 -1
  82. package/dist/{skillopt-optimization-method-DzlF2RM7.js → skillopt-optimization-method-LHi02MzH.js} +2 -2
  83. package/dist/{skillopt-optimization-method-DzlF2RM7.js.map → skillopt-optimization-method-LHi02MzH.js.map} +1 -1
  84. package/dist/{statistical-heldout-UhiexnjU.d.ts → statistical-heldout-Yldkntvy.d.ts} +2 -2
  85. package/dist/{statistical-heldout-UhiexnjU.d.ts.map → statistical-heldout-Yldkntvy.d.ts.map} +1 -1
  86. package/dist/{store-tool-spans-D_qMl2__.d.ts → store-tool-spans-BvdUbeOB.d.ts} +3 -3
  87. package/dist/{store-tool-spans-D_qMl2__.d.ts.map → store-tool-spans-BvdUbeOB.d.ts.map} +1 -1
  88. package/dist/supervisor-run/index.d.ts +35 -2
  89. package/dist/supervisor-run/index.d.ts.map +1 -1
  90. package/dist/supervisor-run/index.js +83 -38
  91. package/dist/supervisor-run/index.js.map +1 -1
  92. package/dist/{tool-groups-RGYfVWpc.d.ts → tool-groups-DjwlMBvW.d.ts} +2 -2
  93. package/dist/tool-groups-DjwlMBvW.d.ts.map +1 -0
  94. package/dist/trace-repair/index.d.ts +1 -1
  95. package/dist/traces.d.ts +2 -2
  96. package/dist/{types-CMyW4GnH.d.ts → types-CCZ34qmV.d.ts} +2 -2
  97. package/dist/{types-CMyW4GnH.d.ts.map → types-CCZ34qmV.d.ts.map} +1 -1
  98. package/dist/{types-DeIUdzNd.d.ts → types-CoPUTiXb.d.ts} +23 -92
  99. package/dist/types-CoPUTiXb.d.ts.map +1 -0
  100. package/dist/{types-JHMOqZI4.d.ts → types-nokrtr7M.d.ts} +11 -1
  101. package/dist/types-nokrtr7M.d.ts.map +1 -0
  102. package/docs/eval-surface-map.md +36 -0
  103. package/docs/insight-report.md +19 -0
  104. package/docs/plants.md +123 -0
  105. package/docs/public-api.md +45 -20
  106. package/package.json +1 -1
  107. package/dist/experiment-tracker-BKEumQug.js.map +0 -1
  108. package/dist/experiment-tracker-Dm8yQMqb.d.ts.map +0 -1
  109. package/dist/index-lfaSeKSD.d.ts.map +0 -1
  110. package/dist/llm-judge-BtJ2Sfk_.js.map +0 -1
  111. package/dist/provenance-CafMdZKM.d.ts.map +0 -1
  112. package/dist/tool-groups-RGYfVWpc.d.ts.map +0 -1
  113. package/dist/types-DeIUdzNd.d.ts.map +0 -1
  114. package/dist/types-JHMOqZI4.d.ts.map +0 -1
@@ -1,5 +1,5 @@
1
1
  import { p as CostProvenance } from "./cost-ledger-DbQdN3nO.js";
2
- import { w as JudgeScore } from "./types-JHMOqZI4.js";
2
+ import { w as JudgeScore } from "./types-nokrtr7M.js";
3
3
  import { o as MatrixResult } from "./index-DNgf5gyG.js";
4
4
  import { AgentProfile } from "@tangle-network/agent-interface";
5
5
  //#region src/multishot/types.d.ts
@@ -398,4 +398,4 @@ interface RunMultishotMatrixResult {
398
398
  declare function runMultishotMatrix<TPersona extends MultishotPersona>(opts: RunMultishotMatrixOptions<TPersona>): Promise<RunMultishotMatrixResult>;
399
399
  //#endregion
400
400
  export { MultishotToolExecutor as A, MultishotFatalToolError as C, MultishotShape as D, MultishotResult as E, assertMultishotShotResult as F, MultishotTransportRequest as M, MultishotTransportResponse as N, MultishotShotResultError as O, MultishotTransportToolCall as P, MultishotDriverEmptyError as S, MultishotPersona as T, JudgeRunResult as _, MultishotCellOutput as a, runJudge as b, RunMultishotMatrixResult as c, MultishotShot as d, RunMultishotOptions as f, JudgeDimension as g, JudgeConfig as h, ConversationJudgeInput as i, MultishotTransport as j, MultishotToolDefinition as k, computeCellComposite as l, DEFAULT_JUDGE_MODEL as m, CellCompositeInput as n, MultishotJudges as o, runMultishot as p, CellCompositeScore as r, RunMultishotMatrixOptions as s, ArtifactJudgeInput as t, runMultishotMatrix as u, renderDimensions as v, MultishotMessage as w, MultishotArtifact as x, renderJsonFooter as y };
401
- //# sourceMappingURL=matrix-DiHmUobV.d.ts.map
401
+ //# sourceMappingURL=matrix-Ch8JO1pG.d.ts.map
@@ -1 +1 @@
1
- {"version":3,"file":"matrix-DiHmUobV.d.ts","names":[],"sources":["../src/multishot/types.ts","../src/multishot/judges.ts","../src/multishot/multishot.ts","../src/multishot/matrix.ts"],"mappings":";;;;;UAIiB;EACf;EACA;EACA;EACA,YAAY;IAAQ;IAAY;IAAc,MAAM;;;UAGrC;EACf;EACA;EACA;IAAc;IAAc,MAAM;;EAClC;;UAGe;EACf,YAAY;EACZ,WAAW;EACX;EACA;;;EAGA;;;;;;;;EAQA,iBAAiB;;UAGF;EACf;EACA;IACE;IACA;IACA,YAAY;;;;;;UAOC;EACf;EACA,UAAU,MAAM;EAChB,QAAQ;EACR;EACA;EACA,SAAS;;UAGM;EACf;EACA;EACA;IAAY;IAAc;;;UAGX;EACf;IAAW;IAAyB,aAAa;;EACjD;IAAU;IAAwB;;;;EAGlC;;;;EAIA;;;;;;;;KASU,sBACV,KAAK,8BACF,QAAQ;KAED,yBACV,MAAM,yBACN;;EAEE,WAAW;EACX,SAAS;MAER;EAAU;EAAiB;;UAEf;;EAEf;;GAEC;;;;;;;;UASc,eAAe,iBAAiB;;EAE/C,eAAe,SAAS;;;EAGxB,2BAA2B,SAAS;;cAGzB,kCAAkC;WACjB;EAA5B,YAA4B;;cAMjB,gCAAgC;EAC3C,YAAY;;cAMD,iCAAiC;EAC5C,YAAY;;;;;;;;;;;;;;;;;;;iBAyBE,0BAA0B,yBAAyB,SAAS;;;cCvI/D;UAEI;;EAEf;;EAEA;;UAGe,YAAY;;EAE3B;;;EAGA,WAAW;;EAEX;;EAEA,YAAY;;EAEZ;;;EAGA,cAAc,OAAO;;EAErB;;UAGe;;EAEf,OAAO;;EAEP,MAAM;;iBAGc,SAAS,QAC7B,OAAO,YAAY,SACnB,OAAO,SACN,QAAQ;;;iBAsIK,iBAAiB,eAAe;;iBAKhC,iBAAiB,eAAe;;;UCxK/B,oBAAoB,iBAAiB;EACpD,SAAS;EACT,SAAS;;;EAGT,QAAQ,eAAe;;EAEvB,QAAQ;;EAER,gBAAgB,eAAe;;;;EAI/B,mBAAmB;EACnB;EACA;EACA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;;EAGA,gBAAgB;;;;EAIhB,iBAAiB;;;EAGjB,gBAAgB;EAChB,SAAS;;;;;;;;;;;KAYC,cAAc,iBAAiB,qBACzC,MAAM,oBAAoB,cACvB,QAAQ;;;;;;;;;;;;;;;;;iBA2BS,aAAa,iBAAiB,kBAClD,MAAM,oBAAoB,YACzB,QAAQ;;;UClFM,uBAAuB,iBAAiB;EACvD,YAAY;EACZ,SAAS;;UAGM,mBAAmB,iBAAiB;EACnD,UAAU;EACV,SAAS;;UAGM,gBAAgB,iBAAiB;;EAEhD,cAAc,YAAY,uBAAuB;;EAEjD,aAAa,YAAY,mBAAmB;;EAE5C,iBAAiB,YAAY,mBAAmB;;EAEhD;;EAEA;;UAGe;EACf;EACA,cAAc;EACd;IACE,aAAa,MAAM;MAAe;MAAc;;IAChD;;EAEF;IACE,aAAa,MAAM;MAAe;MAAc;;IAChD;;;UAIa,0BAA0B,iBAAiB;;EAE1D,UAAU;IAAQ;IAAY,OAAO;;;EAErC,UAAU;;;EAGV,QAAQ,eAAe;;EAEvB,QAAQ,gBAAgB;;EAExB,QAAQ;;EAER,gBAAgB,eAAe;;EAE/B,mBAAmB;;EAEnB;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;;;EAIA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;;;EAIA,gBAAgB;;EAEhB,iBAAiB;;;EAGjB,gBAAgB;;;;;;;;;;;;;;;EAehB,UAAU,cAAc;;;;;UAMT;EACf;EACA;EACA;;UAoBe;EACf,cAAc;;EAEd,cAAc,cAAc;;EAE5B,iBAAiB,cAAc;;;;;;;iBAQjB,qBAAqB,OAAO;EAC1C;EACA;EACA;EACA;;UAuBe;EACf,QAAQ,aAAa;;iBAGD,mBAAmB,iBAAiB,kBACxD,MAAM,0BAA0B,YAC/B,QAAQ"}
1
+ {"version":3,"file":"matrix-Ch8JO1pG.d.ts","names":[],"sources":["../src/multishot/types.ts","../src/multishot/judges.ts","../src/multishot/multishot.ts","../src/multishot/matrix.ts"],"mappings":";;;;;UAIiB;EACf;EACA;EACA;EACA,YAAY;IAAQ;IAAY;IAAc,MAAM;;;UAGrC;EACf;EACA;EACA;IAAc;IAAc,MAAM;;EAClC;;UAGe;EACf,YAAY;EACZ,WAAW;EACX;EACA;;;EAGA;;;;;;;;EAQA,iBAAiB;;UAGF;EACf;EACA;IACE;IACA;IACA,YAAY;;;;;;UAOC;EACf;EACA,UAAU,MAAM;EAChB,QAAQ;EACR;EACA;EACA,SAAS;;UAGM;EACf;EACA;EACA;IAAY;IAAc;;;UAGX;EACf;IAAW;IAAyB,aAAa;;EACjD;IAAU;IAAwB;;;;EAGlC;;;;EAIA;;;;;;;;KASU,sBACV,KAAK,8BACF,QAAQ;KAED,yBACV,MAAM,yBACN;;EAEE,WAAW;EACX,SAAS;MAER;EAAU;EAAiB;;UAEf;;EAEf;;GAEC;;;;;;;;UASc,eAAe,iBAAiB;;EAE/C,eAAe,SAAS;;;EAGxB,2BAA2B,SAAS;;cAGzB,kCAAkC;WACjB;EAA5B,YAA4B;;cAMjB,gCAAgC;EAC3C,YAAY;;cAMD,iCAAiC;EAC5C,YAAY;;;;;;;;;;;;;;;;;;;iBAyBE,0BAA0B,yBAAyB,SAAS;;;cCvI/D;UAEI;;EAEf;;EAEA;;UAGe,YAAY;;EAE3B;;;EAGA,WAAW;;EAEX;;EAEA,YAAY;;EAEZ;;;EAGA,cAAc,OAAO;;EAErB;;UAGe;;EAEf,OAAO;;EAEP,MAAM;;iBAGc,SAAS,QAC7B,OAAO,YAAY,SACnB,OAAO,SACN,QAAQ;;;iBAsIK,iBAAiB,eAAe;;iBAKhC,iBAAiB,eAAe;;;UCxK/B,oBAAoB,iBAAiB;EACpD,SAAS;EACT,SAAS;;;EAGT,QAAQ,eAAe;;EAEvB,QAAQ;;EAER,gBAAgB,eAAe;;;;EAI/B,mBAAmB;EACnB;EACA;EACA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;;EAGA,gBAAgB;;;;EAIhB,iBAAiB;;;EAGjB,gBAAgB;EAChB,SAAS;;;;;;;;;;;KAYC,cAAc,iBAAiB,qBACzC,MAAM,oBAAoB,cACvB,QAAQ;;;;;;;;;;;;;;;;;iBA2BS,aAAa,iBAAiB,kBAClD,MAAM,oBAAoB,YACzB,QAAQ;;;UClFM,uBAAuB,iBAAiB;EACvD,YAAY;EACZ,SAAS;;UAGM,mBAAmB,iBAAiB;EACnD,UAAU;EACV,SAAS;;UAGM,gBAAgB,iBAAiB;;EAEhD,cAAc,YAAY,uBAAuB;;EAEjD,aAAa,YAAY,mBAAmB;;EAE5C,iBAAiB,YAAY,mBAAmB;;EAEhD;;EAEA;;UAGe;EACf;EACA,cAAc;EACd;IACE,aAAa,MAAM;MAAe;MAAc;;IAChD;;EAEF;IACE,aAAa,MAAM;MAAe;MAAc;;IAChD;;;UAIa,0BAA0B,iBAAiB;;EAE1D,UAAU;IAAQ;IAAY,OAAO;;;EAErC,UAAU;;;EAGV,QAAQ,eAAe;;EAEvB,QAAQ,gBAAgB;;EAExB,QAAQ;;EAER,gBAAgB,eAAe;;EAE/B,mBAAmB;;EAEnB;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;;;EAIA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;;;EAIA,gBAAgB;;EAEhB,iBAAiB;;;EAGjB,gBAAgB;;;;;;;;;;;;;;;EAehB,UAAU,cAAc;;;;;UAMT;EACf;EACA;EACA;;UAoBe;EACf,cAAc;;EAEd,cAAc,cAAc;;EAE5B,iBAAiB,cAAc;;;;;;;iBAQjB,qBAAqB,OAAO;EAC1C;EACA;EACA;EACA;;UAuBe;EACf,QAAQ,aAAa;;iBAGD,mBAAmB,iBAAiB,kBACxD,MAAM,0BAA0B,YAC/B,QAAQ"}
@@ -1,7 +1,8 @@
1
1
  import { f as Run } from "../schema-DID1Cqct.js";
2
- import { s as TraceStore } from "../store-Cq9oOrI1.js";
3
2
  import { a as ContinuousCalibrationResult, n as CandidateScore, o as GoldenItem, r as ContinuousAgreement, t as CalibrationResult } from "../judge-calibration-C5CbMYce.js";
4
3
  import { n as SeriesConvergenceResult, o as CorpusAgreementReport, t as SeriesConvergenceOptions } from "../series-convergence-D9WgpXGi.js";
4
+ import { r as LedgerHash } from "../canonical-CFpojCN5.js";
5
+ import { s as TraceStore } from "../store-Cq9oOrI1.js";
5
6
  import { a as OutcomeFilter, i as InMemoryOutcomeStore, n as FileSystemOutcomeStore, o as OutcomeStore, r as FileSystemOutcomeStoreOptions, t as DeploymentOutcome } from "../outcome-store-BYHIuO0e.js";
6
7
  import { a as rubricPredictiveValidity, i as RubricRanking, n as RubricPredictiveValidityInput, r as RubricPredictiveValidityReport, t as RubricOutcomePair } from "../rubric-predictive-validity-DluJLCKQ.js";
7
8
  //#region src/meta-eval/correlation-study.d.ts
@@ -85,6 +86,165 @@ interface CalibrationPair {
85
86
  }
86
87
  declare function calibrationCurve(traceStore: TraceStore, outcomeStore: OutcomeStore, evalMetric: EvalMetricSpec, outcomeMetric: string, options?: CalibrationOptions): Promise<CalibrationReport | null>;
87
88
  //#endregion
89
+ //#region src/meta-eval/plants.d.ts
90
+ /**
91
+ * How a plant item was authored wrong. The class is reported separately in
92
+ * {@link CatchRateReport.byKind} because a grader is routinely sharp on one
93
+ * and blind to another.
94
+ */
95
+ type PlantKind =
96
+ /** A load-bearing value is altered: a number off by one, a comparison flipped. */
97
+ 'wrong-value' |
98
+ /** The item carries its own check, and that check passes without testing the claim. */
99
+ 'self-certifying' |
100
+ /** The check names an input that does not exist, so it cannot run at all. */
101
+ 'unreachable-input' |
102
+ /** A copy of an item already in the set, which is owed a duplicate flag rather than a second grade. */
103
+ 'duplicate';
104
+ /** What a working grader owes a seeded item. */
105
+ type PlantExpectation = 'reject' | 'accept';
106
+ interface Plant {
107
+ /** Name of the plant record. Reported in `missedIds` and `missingIds`. */
108
+ id: string;
109
+ kind: PlantKind;
110
+ /** The seeded item, indistinguishable from a real one once mixed. */
111
+ item: GoldenItem;
112
+ /** The verdict a working grader owes this item. */
113
+ expectedVerdict: PlantExpectation;
114
+ }
115
+ /**
116
+ * Build one plant record and refuse an incoherent one.
117
+ *
118
+ * The refusal that matters is the last: a record whose `expectedVerdict`
119
+ * disagrees with `item.humanScore` inverts the measurement silently, because
120
+ * the same item then reads as wrong here and as correct to every calibration
121
+ * instrument that joins on the id.
122
+ *
123
+ * `item` is copied field by field so a later mutation of the caller's object
124
+ * cannot change what the manifest sealed, and `group` is dropped when it is
125
+ * absent so the record always has a canonical JSON form.
126
+ */
127
+ declare function definePlant(input: {
128
+ id: string;
129
+ kind: PlantKind;
130
+ item: GoldenItem;
131
+ expectedVerdict: PlantExpectation;
132
+ }): Plant;
133
+ interface PlantManifest {
134
+ /**
135
+ * Digest over the seeded order, the plants, and the threshold. Publish it
136
+ * before the grading run: a manifest edited afterwards to match the results
137
+ * no longer matches its seal, and {@link catchRate} refuses it.
138
+ */
139
+ seal: LedgerHash;
140
+ /** The seed that fixed the mix order. */
141
+ seed: number;
142
+ /** The grade at or above which the graded policy's own gate accepts an item. */
143
+ acceptThreshold: number;
144
+ /** Every item id in the seeded set, in the order handed out. */
145
+ itemIds: string[];
146
+ /** The seeded plants. This is the answer key; keep it out of the graded workspace. */
147
+ plants: Plant[];
148
+ }
149
+ interface SeededGradingSet {
150
+ /**
151
+ * The mixed set in seeded order. Operator-side: it still carries every
152
+ * item's `humanScore`, so hand the graded policy the payload each `itemId`
153
+ * names, never this array.
154
+ */
155
+ items: GoldenItem[];
156
+ manifest: PlantManifest;
157
+ }
158
+ interface SeedPlantsOptions {
159
+ /**
160
+ * Fixes the mix order. The same dataset, plants, threshold, and seed always
161
+ * produce the same seeded order and the same seal.
162
+ */
163
+ seed?: number;
164
+ /**
165
+ * The grade at or above which the graded policy's own gate accepts an item.
166
+ * Supply the threshold your gate uses; the default suits a judge scoring in
167
+ * [0, 1] with a pass at the midpoint.
168
+ */
169
+ acceptThreshold?: number;
170
+ }
171
+ /**
172
+ * Mix plants into a grading set and seal which items they are.
173
+ *
174
+ * Refusals: a duplicate id anywhere in the mixed set (the join is by id, so a
175
+ * repeat makes one of the two unscoreable), a plant whose item id collides
176
+ * with a dataset item (it would shadow real work and grade it as a plant), a
177
+ * dataset item with no usable label, and a plant whose expectation the run's
178
+ * `acceptThreshold` contradicts.
179
+ */
180
+ declare function seedPlants(dataset: readonly GoldenItem[], plants: readonly Plant[], options?: SeedPlantsOptions): SeededGradingSet;
181
+ /**
182
+ * One grader outcome for one item. `score: null` says the grader ran and
183
+ * declined to decide — the check never tested the item's defect. Any
184
+ * `CandidateScore` from a judge run is already a valid outcome.
185
+ */
186
+ interface PlantOutcome {
187
+ itemId: string;
188
+ score: number | null;
189
+ }
190
+ /**
191
+ * `evaluated` — a rate stands. `incomplete` — a seeded id had no result.
192
+ * `not_evaluated` — nothing was seeded, or nothing seeded was decided.
193
+ */
194
+ type CatchRateStatus = 'evaluated' | 'incomplete' | 'not_evaluated';
195
+ interface PlantKindCounts {
196
+ seeded: number;
197
+ caught: number;
198
+ missed: number;
199
+ indecisive: number;
200
+ /** caught / (caught + missed), or null when the report is not `evaluated`. */
201
+ rate: number | null;
202
+ }
203
+ /**
204
+ * How the same grader treated the unseeded items of the same set. No labels
205
+ * are needed to read it: a grader that refuses everything scores `rate` 1.0
206
+ * on reject-plants, and a `rejectionRate` of 1.0 here is what separates that
207
+ * reflex from discrimination.
208
+ */
209
+ interface UnseededRejection {
210
+ n: number;
211
+ decided: number;
212
+ rejected: number;
213
+ /** rejected / decided, or null when nothing unseeded was decided. */
214
+ rejectionRate: number | null;
215
+ }
216
+ interface CatchRateReport {
217
+ status: CatchRateStatus;
218
+ /** Why the status is not `evaluated`. Absent when it is. */
219
+ reason?: string;
220
+ seeded: number;
221
+ caught: number;
222
+ missed: number;
223
+ /**
224
+ * The grader returned a result and declined to decide. Counted apart: it
225
+ * enters neither side of `rate`.
226
+ */
227
+ indecisive: number;
228
+ /** caught / (caught + missed), or null unless the status is `evaluated`. */
229
+ rate: number | null;
230
+ /** One entry per kind actually seeded. A kind nobody seeded is absent, never zero. */
231
+ byKind: Partial<Record<PlantKind, PlantKindCounts>>;
232
+ /** Plant ids the grader graded as the seed says it must not. */
233
+ missedIds: string[];
234
+ /** Plant ids with no result at all — the reason a status is `incomplete`. */
235
+ missingIds: string[];
236
+ unseeded: UnseededRejection;
237
+ }
238
+ /**
239
+ * Score a grading run against its sealed manifest.
240
+ *
241
+ * Refusals: a manifest whose contents no longer match its seal, a result for
242
+ * an id the manifest never handed out, and a repeated result id. Each says
243
+ * the results and the manifest describe different runs, and a rate computed
244
+ * across two runs is a fabrication.
245
+ */
246
+ declare function catchRate(results: readonly PlantOutcome[], manifest: PlantManifest): CatchRateReport;
247
+ //#endregion
88
248
  //#region src/meta-eval/sentinel.d.ts
89
249
  declare const SENTINEL_METRIC_NAMES: readonly ['irr', 'calibrationKappa', 'sentinelPassRate'];
90
250
  type SentinelMetricName = (typeof SENTINEL_METRIC_NAMES)[number];
@@ -215,5 +375,5 @@ interface EvalHealthStamp {
215
375
  */
216
376
  declare function evalHealthStamp(report: SentinelReport): EvalHealthStamp;
217
377
  //#endregion
218
- export { CalibrationBin, CalibrationOptions, CalibrationPair, CalibrationReport, CorrelationResult, CorrelationStudyOptions, CorrelationStudyResult, DeploymentOutcome, EvalHealthStamp, EvalMetricSpec, FileSystemOutcomeStore, FileSystemOutcomeStoreOptions, InMemoryOutcomeStore, JudgeSentinelOptions, OutcomeFilter, OutcomePair, OutcomeStore, RubricOutcomePair, RubricPredictiveValidityInput, RubricPredictiveValidityReport, RubricRanking, SentinelMetricName, SentinelMetrics, SentinelReport, SentinelSetOptions, SentinelSnapshot, SentinelStore, SentinelThresholds, SentinelTrend, SnapshotMeta, calibrationCurve, correlationStudy, evalHealthStamp, fileSentinelStore, inMemorySentinelStore, judgeSentinelReport, rubricPredictiveValidity, snapshotFromAgreement, snapshotFromCalibration, snapshotFromSentinelSet, validateSentinelSnapshot };
378
+ export { CalibrationBin, CalibrationOptions, CalibrationPair, CalibrationReport, CatchRateReport, CatchRateStatus, CorrelationResult, CorrelationStudyOptions, CorrelationStudyResult, DeploymentOutcome, EvalHealthStamp, EvalMetricSpec, FileSystemOutcomeStore, FileSystemOutcomeStoreOptions, InMemoryOutcomeStore, JudgeSentinelOptions, OutcomeFilter, OutcomePair, OutcomeStore, Plant, PlantExpectation, PlantKind, PlantKindCounts, PlantManifest, PlantOutcome, RubricOutcomePair, RubricPredictiveValidityInput, RubricPredictiveValidityReport, RubricRanking, SeedPlantsOptions, SeededGradingSet, SentinelMetricName, SentinelMetrics, SentinelReport, SentinelSetOptions, SentinelSnapshot, SentinelStore, SentinelThresholds, SentinelTrend, SnapshotMeta, UnseededRejection, calibrationCurve, catchRate, correlationStudy, definePlant, evalHealthStamp, fileSentinelStore, inMemorySentinelStore, judgeSentinelReport, rubricPredictiveValidity, seedPlants, snapshotFromAgreement, snapshotFromCalibration, snapshotFromSentinelSet, validateSentinelSnapshot };
219
379
  //# sourceMappingURL=index.d.ts.map
@@ -1 +1 @@
1
- {"version":3,"file":"index.d.ts","names":[],"sources":["../../src/meta-eval/correlation-study.ts","../../src/meta-eval/calibration.ts","../../src/meta-eval/sentinel.ts"],"mappings":";;;;;;;UAkBiB;EACf;;;EAGA,WAAW,KAAK,KAAK,OAAO,eAAe;;UAG5B;EACf;EACA;;UAGe;EACf;EACA;EACA;EACA;EACA;;EAEA;IAAe;IAAe;;;EAE9B;;UAGe;EACf,OAAO;EACP;EACA;;UAGe;;EAEf;;EAEA,gBAAgB;;EAEhB;;EAEA;;;EAGA;;iBAGoB,iBACpB,YAAY,YACZ,cAAc,cACd,aAAa,kBACb,8BACA,UAAS,0BACR,QAAQ;;;UCtDM;EACf;EACA;EACA;EACA;EACA;;EAEA;;UAGe;EACf;EACA;EACA;EACA,MAAM;;EAEN;;EAEA;;UAGe;EACf;;EAEA;;EAEA;IAAU;IAAY;;;UAGP;EACf;EACA;;iBAGoB,iBACpB,YAAY,YACZ,cAAc,cACd,YAAY,gBACZ,uBACA,UAAS,qBACR,QAAQ;;;cCDL;KAEM,6BAA6B;UAExB;;EAEf;;EAEA;;EAEA;;UAGe;;EAEf;EACA;;EAEA;;;;;;;EAOA,SAAS;;;UAIM;EACf;EACA;EACA;;;iBAcc,yBAAyB,UAAU,kBAAkB;;;;;;iBAiCrD,wBACd,QAAQ,oBAAoB,6BAC5B,MAAM,eACL;;;;;;;;iBAmBa,sBACd,QAAQ,wBAAwB,qBAChC,MAAM,eACL;UAYc;;EAEf;;;;;;;;;iBAUc,wBACd,QAAQ,kBACR,QAAQ,cACR,MAAM,cACN,UAAS,qBACR;UA4Cc;EACf,OAAO,UAAU,mBAAmB;;EAEpC,QAAQ,mBAAmB,QAAQ;;iBAGrB,sBAAsB,UAAS,qBAA0B;;;;;;;iBAqBzD,kBAAkB,eAAe;UA0ChC;;EAEf;;EAEA;;EAEA;;EAEA;;UAGe;;EAEf;EACA,aAAa;;EAEb,cAAc;;UAGC;EACf;EACA,QAAQ;;EAER,OAAO;;EAEP;;EAEA;;EAEA;EACA;EACA;;UAGe;EACf,UAAU;EACV;EACA;;;;;;;EAOA;;iBAKc,oBACd,SAAS,oBACT,MAAM,uBACL;UA6Gc;EACf;EACA;;;;;;;;;;iBAWc,gBAAgB,QAAQ,iBAAiB"}
1
+ {"version":3,"file":"index.d.ts","names":[],"sources":["../../src/meta-eval/correlation-study.ts","../../src/meta-eval/calibration.ts","../../src/meta-eval/plants.ts","../../src/meta-eval/sentinel.ts"],"mappings":";;;;;;;;UAkBiB;EACf;;;EAGA,WAAW,KAAK,KAAK,OAAO,eAAe;;UAG5B;EACf;EACA;;UAGe;EACf;EACA;EACA;EACA;EACA;;EAEA;IAAe;IAAe;;;EAE9B;;UAGe;EACf,OAAO;EACP;EACA;;UAGe;;EAEf;;EAEA,gBAAgB;;EAEhB;;EAEA;;;EAGA;;iBAGoB,iBACpB,YAAY,YACZ,cAAc,cACd,aAAa,kBACb,8BACA,UAAS,0BACR,QAAQ;;;UCtDM;EACf;EACA;EACA;EACA;EACA;;EAEA;;UAGe;EACf;EACA;EACA;EACA,MAAM;;EAEN;;EAEA;;UAGe;EACf;;EAEA;;EAEA;IAAU;IAAY;;;UAGP;EACf;EACA;;iBAGoB,iBACpB,YAAY,YACZ,cAAc,cACd,YAAY,gBACZ,uBACA,UAAS,qBACR,QAAQ;;;;;;;;KCDC;;;;;;;;;;KAkBA;UAOK;;EAEf;EACA,MAAM;;EAEN,MAAM;;EAEN,iBAAiB;;;;;;;;;;;;;;iBAeH,YAAY;EAC1B;EACA,MAAM;EACN,MAAM;EACN,iBAAiB;IACf;UAiDa;;;;;;EAMf,MAAM;;EAEN;;EAEA;;EAEA;;EAEA,QAAQ;;UAGO;;;;;;EAMf,OAAO;EACP,UAAU;;UAGK;;;;;EAKf;;;;;;EAMA;;;;;;;;;;;iBAYc,WACd,kBAAkB,cAClB,iBAAiB,SACjB,UAAS,oBACR;;;;;;UAqGc;EACf;EACA;;;;;;KAOU;UAEK;EACf;EACA;EACA;EACA;;EAEA;;;;;;;;UASe;EACf;EACA;EACA;;EAEA;;UAGe;EACf,QAAQ;;EAER;EACA;EACA;EACA;;;;;EAKA;;EAEA;;EAEA,QAAQ,QAAQ,OAAO,WAAW;;EAElC;;EAEA;EACA,UAAU;;;;;;;;;;iBAWI,UACd,kBAAkB,gBAClB,UAAU,gBACT;;;cCpUG;KAEM,6BAA6B;UAExB;;EAEf;;EAEA;;EAEA;;UAGe;;EAEf;EACA;;EAEA;;;;;;;EAOA,SAAS;;;UAIM;EACf;EACA;EACA;;;iBAcc,yBAAyB,UAAU,kBAAkB;;;;;;iBAiCrD,wBACd,QAAQ,oBAAoB,6BAC5B,MAAM,eACL;;;;;;;;iBAmBa,sBACd,QAAQ,wBAAwB,qBAChC,MAAM,eACL;UAYc;;EAEf;;;;;;;;;iBAUc,wBACd,QAAQ,kBACR,QAAQ,cACR,MAAM,cACN,UAAS,qBACR;UA4Cc;EACf,OAAO,UAAU,mBAAmB;;EAEpC,QAAQ,mBAAmB,QAAQ;;iBAGrB,sBAAsB,UAAS,qBAA0B;;;;;;;iBAqBzD,kBAAkB,eAAe;UA0ChC;;EAEf;;EAEA;;EAEA;;EAEA;;UAGe;;EAEf;EACA,aAAa;;EAEb,cAAc;;UAGC;EACf;EACA,QAAQ;;EAER,OAAO;;EAEP;;EAEA;;EAEA;EACA;EACA;;UAGe;EACf,UAAU;EACV;EACA;;;;;;;EAOA;;iBAKc,oBACd,SAAS,oBACT,MAAM,uBACL;UA6Gc;EACf;EACA;;;;;;;;;;iBAWc,gBAAgB,QAAQ,iBAAiB"}
@@ -1,5 +1,7 @@
1
- import { s as ValidationError } from "../errors-Dngq5h35.js";
1
+ import { n as CaptureIntegrityError, s as ValidationError } from "../errors-Dngq5h35.js";
2
+ import { a as hashCanonical } from "../canonical-DPyQ_rpt.js";
2
3
  import { i as makeRng } from "../internal-BMFSR8Ns.js";
4
+ import { t as mulberry32 } from "../random-Dn5fPWkt.js";
3
5
  import { a as spearmanR, r as pearsonR } from "../descriptive-1V17A-qa.js";
4
6
  import { u as runMetricExtractor } from "../query-BPGMVlbM.js";
5
7
  import { t as analyzeSeries } from "../series-convergence-CjO2QdRW.js";
@@ -226,6 +228,289 @@ function bootstrapPearsonCi(xs, ys, iterations, seed) {
226
228
  };
227
229
  }
228
230
  //#endregion
231
+ //#region src/meta-eval/plants.ts
232
+ /**
233
+ * Plants — seeded known-wrong items that measure the grader, not the work.
234
+ *
235
+ * A grading run reports how the work scored. It cannot report whether the
236
+ * grader would have noticed a wrong answer, because every item it saw was
237
+ * authored in good faith. A plant closes that hole: an item authored wrong by
238
+ * construction is mixed into the live set, graded by the same path as
239
+ * everything else, and the share of plants the grader refused is the catch
240
+ * rate.
241
+ *
242
+ * Measured motive: a sibling lab ran a deliverable gate that accepted any
243
+ * non-empty submission. It produced six false certifications in seventeen
244
+ * deliveries, and no agent lied — the gate never asked a question the format
245
+ * could fail. A catch rate is the number that would have shown it on day one.
246
+ *
247
+ * This module composes existing primitives rather than adding parallel ones:
248
+ *
249
+ * - A plant IS a {@link GoldenItem} from `../judge-calibration`. Its
250
+ * `humanScore` is the grade a working grader owes the item, so the same
251
+ * array feeds `calibrateJudge` unchanged.
252
+ * - The grader's output is `CandidateScore[]`, the array `calibrateJudge` and
253
+ * `snapshotFromSentinelSet` already consume.
254
+ * - "Caught" is `snapshotFromSentinelSet`'s join with the labels inverted:
255
+ * the grade lands on the side of `acceptThreshold` the seed demands.
256
+ * - The manifest is sealed with `hashCanonical` from `../ledger-core/canonical`,
257
+ * the digest the sealed-experiment path uses, so the answer key cannot be
258
+ * revised once the results are in.
259
+ *
260
+ * Blindness has two halves, and this module owns one. It never puts a plant
261
+ * flag on a graded item: `seedPlants` returns the mixed set and a manifest,
262
+ * and only the manifest knows which ids are seeded. Keeping the manifest out
263
+ * of the graded workspace and publishing its `seal` before grading is the
264
+ * caller's half; {@link catchRate} refuses a manifest whose contents no longer
265
+ * match its seal.
266
+ *
267
+ * Refusals, because a catch rate that cannot refuse is not a measurement:
268
+ *
269
+ * - a seeded id with no result makes the report `incomplete`, never a rate
270
+ * over the results that did come back;
271
+ * - zero seeded plants makes it `not_evaluated`, never 1.0;
272
+ * - a result for an id the manifest never handed out is refused outright.
273
+ */
274
+ const PLANT_KINDS = [
275
+ "wrong-value",
276
+ "self-certifying",
277
+ "unreachable-input",
278
+ "duplicate"
279
+ ];
280
+ const PLANT_EXPECTATIONS = ["reject", "accept"];
281
+ /** The label boundary a plant record is checked against at definition time. */
282
+ const RECORD_LABEL_BOUNDARY = .5;
283
+ /**
284
+ * Build one plant record and refuse an incoherent one.
285
+ *
286
+ * The refusal that matters is the last: a record whose `expectedVerdict`
287
+ * disagrees with `item.humanScore` inverts the measurement silently, because
288
+ * the same item then reads as wrong here and as correct to every calibration
289
+ * instrument that joins on the id.
290
+ *
291
+ * `item` is copied field by field so a later mutation of the caller's object
292
+ * cannot change what the manifest sealed, and `group` is dropped when it is
293
+ * absent so the record always has a canonical JSON form.
294
+ */
295
+ function definePlant(input) {
296
+ const { id, kind, item, expectedVerdict } = input;
297
+ if (typeof id !== "string" || id.trim() === "") throw new ValidationError("definePlant: id must be a non-empty string");
298
+ if (!PLANT_KINDS.includes(kind)) throw new ValidationError(`definePlant: plant "${id}" has kind ${JSON.stringify(kind)}; expected one of ${PLANT_KINDS.join(", ")}`);
299
+ if (!PLANT_EXPECTATIONS.includes(expectedVerdict)) throw new ValidationError(`definePlant: plant "${id}" has expectedVerdict ${JSON.stringify(expectedVerdict)}; expected one of ${PLANT_EXPECTATIONS.join(", ")}`);
300
+ if (typeof item.itemId !== "string" || item.itemId.trim() === "") throw new ValidationError(`definePlant: plant "${id}" has an empty item.itemId`);
301
+ if (!Number.isFinite(item.humanScore) || item.humanScore < 0 || item.humanScore > 1) throw new ValidationError(`definePlant: plant "${id}" has humanScore ${item.humanScore}; expected a finite number in [0, 1]`);
302
+ if (item.group !== void 0 && typeof item.group !== "string") throw new ValidationError(`definePlant: plant "${id}" has a non-string item.group`);
303
+ if (item.humanScore === RECORD_LABEL_BOUNDARY) throw new ValidationError(`definePlant: plant "${id}" has humanScore ${RECORD_LABEL_BOUNDARY}, which states neither a rejection nor an acceptance`);
304
+ const labelSays = item.humanScore < RECORD_LABEL_BOUNDARY ? "reject" : "accept";
305
+ if (labelSays !== expectedVerdict) throw new ValidationError(`definePlant: plant "${id}" expects the grader to ${expectedVerdict} it, but humanScore ${item.humanScore} says ${labelSays}`);
306
+ return {
307
+ id,
308
+ kind,
309
+ item: {
310
+ itemId: item.itemId,
311
+ humanScore: item.humanScore,
312
+ ...item.group === void 0 ? {} : { group: item.group }
313
+ },
314
+ expectedVerdict
315
+ };
316
+ }
317
+ /**
318
+ * Mix plants into a grading set and seal which items they are.
319
+ *
320
+ * Refusals: a duplicate id anywhere in the mixed set (the join is by id, so a
321
+ * repeat makes one of the two unscoreable), a plant whose item id collides
322
+ * with a dataset item (it would shadow real work and grade it as a plant), a
323
+ * dataset item with no usable label, and a plant whose expectation the run's
324
+ * `acceptThreshold` contradicts.
325
+ */
326
+ function seedPlants(dataset, plants, options = {}) {
327
+ const seed = options.seed ?? 7;
328
+ const acceptThreshold = options.acceptThreshold ?? .5;
329
+ if (!Number.isFinite(seed)) throw new ValidationError(`seedPlants: seed must be a finite number, got ${seed}`);
330
+ if (!Number.isFinite(acceptThreshold) || acceptThreshold <= 0 || acceptThreshold > 1) throw new ValidationError(`seedPlants: acceptThreshold must be a finite number in (0, 1], got ${acceptThreshold}`);
331
+ const datasetItems = [];
332
+ const seen = /* @__PURE__ */ new Set();
333
+ for (const item of dataset) {
334
+ if (typeof item.itemId !== "string" || item.itemId.trim() === "") throw new ValidationError("seedPlants: a dataset item has an empty itemId");
335
+ if (!Number.isFinite(item.humanScore)) throw new ValidationError(`seedPlants: dataset item "${item.itemId}" has a non-finite humanScore`);
336
+ if (seen.has(item.itemId)) throw new ValidationError(`seedPlants: duplicate dataset itemId "${item.itemId}"`);
337
+ seen.add(item.itemId);
338
+ datasetItems.push({
339
+ itemId: item.itemId,
340
+ humanScore: item.humanScore,
341
+ ...item.group === void 0 ? {} : { group: item.group }
342
+ });
343
+ }
344
+ const sealedPlants = [];
345
+ const plantIds = /* @__PURE__ */ new Set();
346
+ for (const plant of plants) {
347
+ const record = definePlant(plant);
348
+ if (plantIds.has(record.id)) throw new ValidationError(`seedPlants: duplicate plant id "${record.id}"`);
349
+ if (seen.has(record.item.itemId)) throw new ValidationError(`seedPlants: plant "${record.id}" reuses itemId "${record.item.itemId}", which is already in the set`);
350
+ const thresholdSays = record.item.humanScore >= acceptThreshold ? "accept" : "reject";
351
+ if (thresholdSays !== record.expectedVerdict) throw new ValidationError(`seedPlants: plant "${record.id}" expects the grader to ${record.expectedVerdict} it, but humanScore ${record.item.humanScore} is on the ${thresholdSays} side of acceptThreshold ${acceptThreshold}`);
352
+ plantIds.add(record.id);
353
+ seen.add(record.item.itemId);
354
+ sealedPlants.push(record);
355
+ }
356
+ const items = shuffled([...datasetItems, ...sealedPlants.map((plant) => plant.item)], mulberry32(seed));
357
+ const itemIds = items.map((item) => item.itemId);
358
+ return {
359
+ items,
360
+ manifest: {
361
+ seal: sealManifest({
362
+ seed,
363
+ acceptThreshold,
364
+ itemIds,
365
+ plants: sealedPlants
366
+ }),
367
+ seed,
368
+ acceptThreshold,
369
+ itemIds,
370
+ plants: sealedPlants
371
+ }
372
+ };
373
+ }
374
+ /**
375
+ * Order by one independent uniform key per item, which is a uniform
376
+ * permutation and a pure function of the seed. A comparator that returns a
377
+ * fresh random sign instead is neither: it is not a consistent ordering, so
378
+ * the permutation it produces is biased and depends on the sort algorithm.
379
+ */
380
+ function shuffled(items, random) {
381
+ return items.map((item) => ({
382
+ item,
383
+ key: random()
384
+ })).sort((left, right) => left.key - right.key).map((entry) => entry.item);
385
+ }
386
+ function sealManifest(contents) {
387
+ return hashCanonical({
388
+ scheme: "agent-eval.plant-manifest.v1",
389
+ seed: contents.seed,
390
+ acceptThreshold: contents.acceptThreshold,
391
+ itemIds: contents.itemIds,
392
+ plants: contents.plants
393
+ });
394
+ }
395
+ /**
396
+ * Score a grading run against its sealed manifest.
397
+ *
398
+ * Refusals: a manifest whose contents no longer match its seal, a result for
399
+ * an id the manifest never handed out, and a repeated result id. Each says
400
+ * the results and the manifest describe different runs, and a rate computed
401
+ * across two runs is a fabrication.
402
+ */
403
+ function catchRate(results, manifest) {
404
+ const expectedSeal = sealManifest({
405
+ seed: manifest.seed,
406
+ acceptThreshold: manifest.acceptThreshold,
407
+ itemIds: manifest.itemIds,
408
+ plants: manifest.plants
409
+ });
410
+ if (expectedSeal !== manifest.seal) throw new CaptureIntegrityError(`catchRate: manifest contents hash to ${expectedSeal} but the manifest carries seal ${manifest.seal} — the plant set changed after it was sealed`);
411
+ const handedOut = new Set(manifest.itemIds);
412
+ const scoreByItemId = /* @__PURE__ */ new Map();
413
+ for (const result of results) {
414
+ if (!handedOut.has(result.itemId)) throw new ValidationError(`catchRate: result for "${result.itemId}", which this manifest never handed out`);
415
+ if (scoreByItemId.has(result.itemId)) throw new ValidationError(`catchRate: duplicate result for "${result.itemId}"`);
416
+ if (result.score !== null && !Number.isFinite(result.score)) throw new ValidationError(`catchRate: result for "${result.itemId}" has score ${result.score}; expected a finite number or null`);
417
+ scoreByItemId.set(result.itemId, result.score);
418
+ }
419
+ const byKind = {};
420
+ const missedIds = [];
421
+ const missingIds = [];
422
+ let caught = 0;
423
+ let missed = 0;
424
+ let indecisive = 0;
425
+ for (const plant of manifest.plants) {
426
+ let counts = byKind[plant.kind];
427
+ if (counts === void 0) {
428
+ counts = {
429
+ seeded: 0,
430
+ caught: 0,
431
+ missed: 0,
432
+ indecisive: 0,
433
+ rate: null
434
+ };
435
+ byKind[plant.kind] = counts;
436
+ }
437
+ counts.seeded += 1;
438
+ if (!scoreByItemId.has(plant.item.itemId)) {
439
+ missingIds.push(plant.id);
440
+ continue;
441
+ }
442
+ const score = scoreByItemId.get(plant.item.itemId) ?? null;
443
+ if (score === null) {
444
+ indecisive += 1;
445
+ counts.indecisive += 1;
446
+ continue;
447
+ }
448
+ if ((score >= manifest.acceptThreshold ? "accept" : "reject") === plant.expectedVerdict) {
449
+ caught += 1;
450
+ counts.caught += 1;
451
+ } else {
452
+ missed += 1;
453
+ counts.missed += 1;
454
+ missedIds.push(plant.id);
455
+ }
456
+ }
457
+ const seeded = manifest.plants.length;
458
+ const decided = caught + missed;
459
+ const report = {
460
+ status: "evaluated",
461
+ seeded,
462
+ caught,
463
+ missed,
464
+ indecisive,
465
+ rate: null,
466
+ byKind,
467
+ missedIds,
468
+ missingIds,
469
+ unseeded: unseededRejection(manifest, scoreByItemId)
470
+ };
471
+ if (seeded === 0) return {
472
+ ...report,
473
+ status: "not_evaluated",
474
+ reason: "no plant was seeded, so the grader was never asked a question it could fail"
475
+ };
476
+ if (missingIds.length > 0) return {
477
+ ...report,
478
+ status: "incomplete",
479
+ reason: `${missingIds.length} of ${seeded} seeded plants have no result: ${missingIds.join(", ")}`
480
+ };
481
+ if (decided === 0) return {
482
+ ...report,
483
+ status: "not_evaluated",
484
+ reason: `all ${seeded} seeded plants are indecisive: no check tested the seeded defect`
485
+ };
486
+ for (const counts of Object.values(byKind)) {
487
+ const kindDecided = counts.caught + counts.missed;
488
+ counts.rate = kindDecided === 0 ? null : counts.caught / kindDecided;
489
+ }
490
+ report.rate = caught / decided;
491
+ return report;
492
+ }
493
+ function unseededRejection(manifest, scoreByItemId) {
494
+ const plantItemIds = new Set(manifest.plants.map((plant) => plant.item.itemId));
495
+ let n = 0;
496
+ let decided = 0;
497
+ let rejected = 0;
498
+ for (const itemId of manifest.itemIds) {
499
+ if (plantItemIds.has(itemId)) continue;
500
+ n += 1;
501
+ const score = scoreByItemId.get(itemId);
502
+ if (score === void 0 || score === null) continue;
503
+ decided += 1;
504
+ if (score < manifest.acceptThreshold) rejected += 1;
505
+ }
506
+ return {
507
+ n,
508
+ decided,
509
+ rejected,
510
+ rejectionRate: decided === 0 ? null : rejected / decided
511
+ };
512
+ }
513
+ //#endregion
229
514
  //#region src/meta-eval/sentinel.ts
230
515
  /**
231
516
  * Judge sentinel — eval trustworthiness as a continuously measured,
@@ -499,6 +784,6 @@ function evalHealthStamp(report) {
499
784
  };
500
785
  }
501
786
  //#endregion
502
- export { FileSystemOutcomeStore, InMemoryOutcomeStore, calibrationCurve, correlationStudy, evalHealthStamp, fileSentinelStore, inMemorySentinelStore, judgeSentinelReport, rubricPredictiveValidity, snapshotFromAgreement, snapshotFromCalibration, snapshotFromSentinelSet, validateSentinelSnapshot };
787
+ export { FileSystemOutcomeStore, InMemoryOutcomeStore, calibrationCurve, catchRate, correlationStudy, definePlant, evalHealthStamp, fileSentinelStore, inMemorySentinelStore, judgeSentinelReport, rubricPredictiveValidity, seedPlants, snapshotFromAgreement, snapshotFromCalibration, snapshotFromSentinelSet, validateSentinelSnapshot };
503
788
 
504
789
  //# sourceMappingURL=index.js.map