@tangle-network/agent-eval 0.171.0 → 0.172.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +85 -0
- package/README.md +3 -0
- package/dist/adapters/http.d.ts +2 -2
- package/dist/analyst/index.d.ts +5 -5
- package/dist/analyst/index.js +2 -2
- package/dist/{experiment-tracker-Dm8yQMqb.d.ts → attestation-CJBGmMVh.d.ts} +78 -2
- package/dist/attestation-CJBGmMVh.d.ts.map +1 -0
- package/dist/{experiment-tracker-BKEumQug.js → attestation-XSUpbc4o.js} +96 -2
- package/dist/attestation-XSUpbc4o.js.map +1 -0
- package/dist/{benchmark-command-D8k3Gf0J.js → benchmark-command--qeZUHbu.js} +2 -2
- package/dist/{benchmark-command-D8k3Gf0J.js.map → benchmark-command--qeZUHbu.js.map} +1 -1
- package/dist/benchmarks/index.d.ts +4 -4
- package/dist/benchmarks/index.js +2 -2
- package/dist/bounded-process-VIi0KSL2.js +212 -0
- package/dist/bounded-process-VIi0KSL2.js.map +1 -0
- package/dist/builder-eval/index.js +39 -98
- package/dist/builder-eval/index.js.map +1 -1
- package/dist/campaign/index.d.ts +8 -7
- package/dist/campaign/index.js +4 -4
- package/dist/{campaign-B72njjHj.js → campaign-Dp35pBbS.js} +4 -4
- package/dist/{campaign-B72njjHj.js.map → campaign-Dp35pBbS.js.map} +1 -1
- package/dist/canonical-CFpojCN5.d.ts +31 -0
- package/dist/canonical-CFpojCN5.d.ts.map +1 -0
- package/dist/{chat-client-DEtybj5i.js → chat-client-DI79OPye.js} +2 -2
- package/dist/{chat-client-DEtybj5i.js.map → chat-client-DI79OPye.js.map} +1 -1
- package/dist/cli.js +1 -1
- package/dist/{client-CDtcZ3p9.d.ts → client-Df7wdslk.d.ts} +2 -2
- package/dist/{client-CDtcZ3p9.d.ts.map → client-Df7wdslk.d.ts.map} +1 -1
- package/dist/contract/index.d.ts +9 -9
- package/dist/contract/index.js +4 -4
- package/dist/{default-registry-B0s2zU-s.d.ts → default-registry-XxedTLwu.d.ts} +3 -3
- package/dist/{default-registry-B0s2zU-s.d.ts.map → default-registry-XxedTLwu.d.ts.map} +1 -1
- package/dist/{define-agent-eval-Cjy2yhqP.d.ts → define-agent-eval-0wW7gFhr.d.ts} +4 -4
- package/dist/{define-agent-eval-Cjy2yhqP.d.ts.map → define-agent-eval-0wW7gFhr.d.ts.map} +1 -1
- package/dist/{define-agent-eval-Dy8QgxAI.js → define-agent-eval-jS8xj_Q_.js} +2 -2
- package/dist/{define-agent-eval-Dy8QgxAI.js.map → define-agent-eval-jS8xj_Q_.js.map} +1 -1
- package/dist/descriptive-B2iPaT9J.d.ts +89 -0
- package/dist/descriptive-B2iPaT9J.d.ts.map +1 -0
- package/dist/{engine-CAmTUk52.d.ts → engine-BfRay1qD.d.ts} +2 -2
- package/dist/{engine-CAmTUk52.d.ts.map → engine-BfRay1qD.d.ts.map} +1 -1
- package/dist/experiment/index.d.ts +3 -54
- package/dist/experiment/index.d.ts.map +1 -1
- package/dist/experiment/index.js +1 -95
- package/dist/experiment/index.js.map +1 -1
- package/dist/{heldout-gate-Dh2b62w8.d.ts → heldout-gate-JgNRDZwZ.d.ts} +3 -3
- package/dist/{heldout-gate-Dh2b62w8.d.ts.map → heldout-gate-JgNRDZwZ.d.ts.map} +1 -1
- package/dist/hosted/index.d.ts +1 -1
- package/dist/{index-lfaSeKSD.d.ts → index-D-UdhAmg.d.ts} +3 -31
- package/dist/index-D-UdhAmg.d.ts.map +1 -0
- package/dist/{index-DT73JraI.d.ts → index-DDAPhUJJ.d.ts} +4 -4
- package/dist/{index-DT73JraI.d.ts.map → index-DDAPhUJJ.d.ts.map} +1 -1
- package/dist/{index-fNXZMCzX.d.ts → index-DnglhM0A.d.ts} +9 -9
- package/dist/{index-fNXZMCzX.d.ts.map → index-DnglhM0A.d.ts.map} +1 -1
- package/dist/{index-8VIogTyS.d.ts → index-_vPrVMRX.d.ts} +6 -6
- package/dist/{index-8VIogTyS.d.ts.map → index-_vPrVMRX.d.ts.map} +1 -1
- package/dist/index.d.ts +154 -15
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +7 -6
- package/dist/index.js.map +1 -1
- package/dist/{integrity-CyWSSoQS.js → integrity-BWywb34E.js} +34 -11
- package/dist/{integrity-CyWSSoQS.js.map → integrity-BWywb34E.js.map} +1 -1
- package/dist/ledger-core/index.d.ts +2 -1
- package/dist/{llm-judge-BtJ2Sfk_.js → llm-judge-aQHIk5_-.js} +16 -10
- package/dist/llm-judge-aQHIk5_-.js.map +1 -0
- package/dist/{matrix-DiHmUobV.d.ts → matrix-Ch8JO1pG.d.ts} +2 -2
- package/dist/{matrix-DiHmUobV.d.ts.map → matrix-Ch8JO1pG.d.ts.map} +1 -1
- package/dist/meta-eval/index.d.ts +162 -2
- package/dist/meta-eval/index.d.ts.map +1 -1
- package/dist/meta-eval/index.js +287 -2
- package/dist/meta-eval/index.js.map +1 -1
- package/dist/multishot/golden/index.d.ts +1 -1
- package/dist/multishot/index.d.ts +2 -2
- package/dist/openapi.json +1 -1
- package/dist/{produced-state-Cm6DU_Ao.js → produced-state-CxmbFxFd.js} +2 -2
- package/dist/{produced-state-Cm6DU_Ao.js.map → produced-state-CxmbFxFd.js.map} +1 -1
- package/dist/{promotion-policy-WSXtBgBb.d.ts → promotion-policy-BBBcz5_3.d.ts} +2 -2
- package/dist/{promotion-policy-WSXtBgBb.d.ts.map → promotion-policy-BBBcz5_3.d.ts.map} +1 -1
- package/dist/{provenance-CafMdZKM.d.ts → provenance-Dp-vvyrU.d.ts} +13 -5
- package/dist/provenance-Dp-vvyrU.d.ts.map +1 -0
- package/dist/rl.d.ts +1 -1
- package/dist/sandbox-harness-BlSOu4LX.d.ts.map +1 -1
- package/dist/{skillopt-optimization-method-DzlF2RM7.js → skillopt-optimization-method-LHi02MzH.js} +2 -2
- package/dist/{skillopt-optimization-method-DzlF2RM7.js.map → skillopt-optimization-method-LHi02MzH.js.map} +1 -1
- package/dist/{statistical-heldout-UhiexnjU.d.ts → statistical-heldout-Yldkntvy.d.ts} +2 -2
- package/dist/{statistical-heldout-UhiexnjU.d.ts.map → statistical-heldout-Yldkntvy.d.ts.map} +1 -1
- package/dist/{store-tool-spans-D_qMl2__.d.ts → store-tool-spans-BvdUbeOB.d.ts} +3 -3
- package/dist/{store-tool-spans-D_qMl2__.d.ts.map → store-tool-spans-BvdUbeOB.d.ts.map} +1 -1
- package/dist/supervisor-run/index.d.ts +35 -2
- package/dist/supervisor-run/index.d.ts.map +1 -1
- package/dist/supervisor-run/index.js +83 -38
- package/dist/supervisor-run/index.js.map +1 -1
- package/dist/{tool-groups-RGYfVWpc.d.ts → tool-groups-DjwlMBvW.d.ts} +2 -2
- package/dist/tool-groups-DjwlMBvW.d.ts.map +1 -0
- package/dist/trace-repair/index.d.ts +1 -1
- package/dist/traces.d.ts +2 -2
- package/dist/{types-CMyW4GnH.d.ts → types-CCZ34qmV.d.ts} +2 -2
- package/dist/{types-CMyW4GnH.d.ts.map → types-CCZ34qmV.d.ts.map} +1 -1
- package/dist/{types-DeIUdzNd.d.ts → types-CoPUTiXb.d.ts} +23 -92
- package/dist/types-CoPUTiXb.d.ts.map +1 -0
- package/dist/{types-JHMOqZI4.d.ts → types-nokrtr7M.d.ts} +11 -1
- package/dist/types-nokrtr7M.d.ts.map +1 -0
- package/docs/eval-surface-map.md +36 -0
- package/docs/insight-report.md +19 -0
- package/docs/plants.md +123 -0
- package/docs/public-api.md +45 -20
- package/package.json +1 -1
- package/dist/experiment-tracker-BKEumQug.js.map +0 -1
- package/dist/experiment-tracker-Dm8yQMqb.d.ts.map +0 -1
- package/dist/index-lfaSeKSD.d.ts.map +0 -1
- package/dist/llm-judge-BtJ2Sfk_.js.map +0 -1
- package/dist/provenance-CafMdZKM.d.ts.map +0 -1
- package/dist/tool-groups-RGYfVWpc.d.ts.map +0 -1
- package/dist/types-DeIUdzNd.d.ts.map +0 -1
- package/dist/types-JHMOqZI4.d.ts.map +0 -1
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import { p as CostProvenance } from "./cost-ledger-DbQdN3nO.js";
|
|
2
|
-
import { w as JudgeScore } from "./types-
|
|
2
|
+
import { w as JudgeScore } from "./types-nokrtr7M.js";
|
|
3
3
|
import { o as MatrixResult } from "./index-DNgf5gyG.js";
|
|
4
4
|
import { AgentProfile } from "@tangle-network/agent-interface";
|
|
5
5
|
//#region src/multishot/types.d.ts
|
|
@@ -398,4 +398,4 @@ interface RunMultishotMatrixResult {
|
|
|
398
398
|
declare function runMultishotMatrix<TPersona extends MultishotPersona>(opts: RunMultishotMatrixOptions<TPersona>): Promise<RunMultishotMatrixResult>;
|
|
399
399
|
//#endregion
|
|
400
400
|
export { MultishotToolExecutor as A, MultishotFatalToolError as C, MultishotShape as D, MultishotResult as E, assertMultishotShotResult as F, MultishotTransportRequest as M, MultishotTransportResponse as N, MultishotShotResultError as O, MultishotTransportToolCall as P, MultishotDriverEmptyError as S, MultishotPersona as T, JudgeRunResult as _, MultishotCellOutput as a, runJudge as b, RunMultishotMatrixResult as c, MultishotShot as d, RunMultishotOptions as f, JudgeDimension as g, JudgeConfig as h, ConversationJudgeInput as i, MultishotTransport as j, MultishotToolDefinition as k, computeCellComposite as l, DEFAULT_JUDGE_MODEL as m, CellCompositeInput as n, MultishotJudges as o, runMultishot as p, CellCompositeScore as r, RunMultishotMatrixOptions as s, ArtifactJudgeInput as t, runMultishotMatrix as u, renderDimensions as v, MultishotMessage as w, MultishotArtifact as x, renderJsonFooter as y };
|
|
401
|
-
//# sourceMappingURL=matrix-
|
|
401
|
+
//# sourceMappingURL=matrix-Ch8JO1pG.d.ts.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"matrix-
|
|
1
|
+
{"version":3,"file":"matrix-Ch8JO1pG.d.ts","names":[],"sources":["../src/multishot/types.ts","../src/multishot/judges.ts","../src/multishot/multishot.ts","../src/multishot/matrix.ts"],"mappings":";;;;;UAIiB;EACf;EACA;EACA;EACA,YAAY;IAAQ;IAAY;IAAc,MAAM;;;UAGrC;EACf;EACA;EACA;IAAc;IAAc,MAAM;;EAClC;;UAGe;EACf,YAAY;EACZ,WAAW;EACX;EACA;;;EAGA;;;;;;;;EAQA,iBAAiB;;UAGF;EACf;EACA;IACE;IACA;IACA,YAAY;;;;;;UAOC;EACf;EACA,UAAU,MAAM;EAChB,QAAQ;EACR;EACA;EACA,SAAS;;UAGM;EACf;EACA;EACA;IAAY;IAAc;;;UAGX;EACf;IAAW;IAAyB,aAAa;;EACjD;IAAU;IAAwB;;;;EAGlC;;;;EAIA;;;;;;;;KASU,sBACV,KAAK,8BACF,QAAQ;KAED,yBACV,MAAM,yBACN;;EAEE,WAAW;EACX,SAAS;MAER;EAAU;EAAiB;;UAEf;;EAEf;;GAEC;;;;;;;;UASc,eAAe,iBAAiB;;EAE/C,eAAe,SAAS;;;EAGxB,2BAA2B,SAAS;;cAGzB,kCAAkC;WACjB;EAA5B,YAA4B;;cAMjB,gCAAgC;EAC3C,YAAY;;cAMD,iCAAiC;EAC5C,YAAY;;;;;;;;;;;;;;;;;;;iBAyBE,0BAA0B,yBAAyB,SAAS;;;cCvI/D;UAEI;;EAEf;;EAEA;;UAGe,YAAY;;EAE3B;;;EAGA,WAAW;;EAEX;;EAEA,YAAY;;EAEZ;;;EAGA,cAAc,OAAO;;EAErB;;UAGe;;EAEf,OAAO;;EAEP,MAAM;;iBAGc,SAAS,QAC7B,OAAO,YAAY,SACnB,OAAO,SACN,QAAQ;;;iBAsIK,iBAAiB,eAAe;;iBAKhC,iBAAiB,eAAe;;;UCxK/B,oBAAoB,iBAAiB;EACpD,SAAS;EACT,SAAS;;;EAGT,QAAQ,eAAe;;EAEvB,QAAQ;;EAER,gBAAgB,eAAe;;;;EAI/B,mBAAmB;EACnB;EACA;EACA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;;EAGA,gBAAgB;;;;EAIhB,iBAAiB;;;EAGjB,gBAAgB;EAChB,SAAS;;;;;;;;;;;KAYC,cAAc,iBAAiB,qBACzC,MAAM,oBAAoB,cACvB,QAAQ;;;;;;;;;;;;;;;;;iBA2BS,aAAa,iBAAiB,kBAClD,MAAM,oBAAoB,YACzB,QAAQ;;;UClFM,uBAAuB,iBAAiB;EACvD,YAAY;EACZ,SAAS;;UAGM,mBAAmB,iBAAiB;EACnD,UAAU;EACV,SAAS;;UAGM,gBAAgB,iBAAiB;;EAEhD,cAAc,YAAY,uBAAuB;;EAEjD,aAAa,YAAY,mBAAmB;;EAE5C,iBAAiB,YAAY,mBAAmB;;EAEhD;;EAEA;;UAGe;EACf;EACA,cAAc;EACd;IACE,aAAa,MAAM;MAAe;MAAc;;IAChD;;EAEF;IACE,aAAa,MAAM;MAAe;MAAc;;IAChD;;;UAIa,0BAA0B,iBAAiB;;EAE1D,UAAU;IAAQ;IAAY,OAAO;;;EAErC,UAAU;;;EAGV,QAAQ,eAAe;;EAEvB,QAAQ,gBAAgB;;EAExB,QAAQ;;EAER,gBAAgB,eAAe;;EAE/B,mBAAmB;;EAEnB;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;;;EAIA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;;;EAIA,gBAAgB;;EAEhB,iBAAiB;;;EAGjB,gBAAgB;;;;;;;;;;;;;;;EAehB,UAAU,cAAc;;;;;UAMT;EACf;EACA;EACA;;UAoBe;EACf,cAAc;;EAEd,cAAc,cAAc;;EAE5B,iBAAiB,cAAc;;;;;;;iBAQjB,qBAAqB,OAAO;EAC1C;EACA;EACA;EACA;;UAuBe;EACf,QAAQ,aAAa;;iBAGD,mBAAmB,iBAAiB,kBACxD,MAAM,0BAA0B,YAC/B,QAAQ"}
|
|
@@ -1,7 +1,8 @@
|
|
|
1
1
|
import { f as Run } from "../schema-DID1Cqct.js";
|
|
2
|
-
import { s as TraceStore } from "../store-Cq9oOrI1.js";
|
|
3
2
|
import { a as ContinuousCalibrationResult, n as CandidateScore, o as GoldenItem, r as ContinuousAgreement, t as CalibrationResult } from "../judge-calibration-C5CbMYce.js";
|
|
4
3
|
import { n as SeriesConvergenceResult, o as CorpusAgreementReport, t as SeriesConvergenceOptions } from "../series-convergence-D9WgpXGi.js";
|
|
4
|
+
import { r as LedgerHash } from "../canonical-CFpojCN5.js";
|
|
5
|
+
import { s as TraceStore } from "../store-Cq9oOrI1.js";
|
|
5
6
|
import { a as OutcomeFilter, i as InMemoryOutcomeStore, n as FileSystemOutcomeStore, o as OutcomeStore, r as FileSystemOutcomeStoreOptions, t as DeploymentOutcome } from "../outcome-store-BYHIuO0e.js";
|
|
6
7
|
import { a as rubricPredictiveValidity, i as RubricRanking, n as RubricPredictiveValidityInput, r as RubricPredictiveValidityReport, t as RubricOutcomePair } from "../rubric-predictive-validity-DluJLCKQ.js";
|
|
7
8
|
//#region src/meta-eval/correlation-study.d.ts
|
|
@@ -85,6 +86,165 @@ interface CalibrationPair {
|
|
|
85
86
|
}
|
|
86
87
|
declare function calibrationCurve(traceStore: TraceStore, outcomeStore: OutcomeStore, evalMetric: EvalMetricSpec, outcomeMetric: string, options?: CalibrationOptions): Promise<CalibrationReport | null>;
|
|
87
88
|
//#endregion
|
|
89
|
+
//#region src/meta-eval/plants.d.ts
|
|
90
|
+
/**
|
|
91
|
+
* How a plant item was authored wrong. The class is reported separately in
|
|
92
|
+
* {@link CatchRateReport.byKind} because a grader is routinely sharp on one
|
|
93
|
+
* and blind to another.
|
|
94
|
+
*/
|
|
95
|
+
type PlantKind =
|
|
96
|
+
/** A load-bearing value is altered: a number off by one, a comparison flipped. */
|
|
97
|
+
'wrong-value' |
|
|
98
|
+
/** The item carries its own check, and that check passes without testing the claim. */
|
|
99
|
+
'self-certifying' |
|
|
100
|
+
/** The check names an input that does not exist, so it cannot run at all. */
|
|
101
|
+
'unreachable-input' |
|
|
102
|
+
/** A copy of an item already in the set, which is owed a duplicate flag rather than a second grade. */
|
|
103
|
+
'duplicate';
|
|
104
|
+
/** What a working grader owes a seeded item. */
|
|
105
|
+
type PlantExpectation = 'reject' | 'accept';
|
|
106
|
+
interface Plant {
|
|
107
|
+
/** Name of the plant record. Reported in `missedIds` and `missingIds`. */
|
|
108
|
+
id: string;
|
|
109
|
+
kind: PlantKind;
|
|
110
|
+
/** The seeded item, indistinguishable from a real one once mixed. */
|
|
111
|
+
item: GoldenItem;
|
|
112
|
+
/** The verdict a working grader owes this item. */
|
|
113
|
+
expectedVerdict: PlantExpectation;
|
|
114
|
+
}
|
|
115
|
+
/**
|
|
116
|
+
* Build one plant record and refuse an incoherent one.
|
|
117
|
+
*
|
|
118
|
+
* The refusal that matters is the last: a record whose `expectedVerdict`
|
|
119
|
+
* disagrees with `item.humanScore` inverts the measurement silently, because
|
|
120
|
+
* the same item then reads as wrong here and as correct to every calibration
|
|
121
|
+
* instrument that joins on the id.
|
|
122
|
+
*
|
|
123
|
+
* `item` is copied field by field so a later mutation of the caller's object
|
|
124
|
+
* cannot change what the manifest sealed, and `group` is dropped when it is
|
|
125
|
+
* absent so the record always has a canonical JSON form.
|
|
126
|
+
*/
|
|
127
|
+
declare function definePlant(input: {
|
|
128
|
+
id: string;
|
|
129
|
+
kind: PlantKind;
|
|
130
|
+
item: GoldenItem;
|
|
131
|
+
expectedVerdict: PlantExpectation;
|
|
132
|
+
}): Plant;
|
|
133
|
+
interface PlantManifest {
|
|
134
|
+
/**
|
|
135
|
+
* Digest over the seeded order, the plants, and the threshold. Publish it
|
|
136
|
+
* before the grading run: a manifest edited afterwards to match the results
|
|
137
|
+
* no longer matches its seal, and {@link catchRate} refuses it.
|
|
138
|
+
*/
|
|
139
|
+
seal: LedgerHash;
|
|
140
|
+
/** The seed that fixed the mix order. */
|
|
141
|
+
seed: number;
|
|
142
|
+
/** The grade at or above which the graded policy's own gate accepts an item. */
|
|
143
|
+
acceptThreshold: number;
|
|
144
|
+
/** Every item id in the seeded set, in the order handed out. */
|
|
145
|
+
itemIds: string[];
|
|
146
|
+
/** The seeded plants. This is the answer key; keep it out of the graded workspace. */
|
|
147
|
+
plants: Plant[];
|
|
148
|
+
}
|
|
149
|
+
interface SeededGradingSet {
|
|
150
|
+
/**
|
|
151
|
+
* The mixed set in seeded order. Operator-side: it still carries every
|
|
152
|
+
* item's `humanScore`, so hand the graded policy the payload each `itemId`
|
|
153
|
+
* names, never this array.
|
|
154
|
+
*/
|
|
155
|
+
items: GoldenItem[];
|
|
156
|
+
manifest: PlantManifest;
|
|
157
|
+
}
|
|
158
|
+
interface SeedPlantsOptions {
|
|
159
|
+
/**
|
|
160
|
+
* Fixes the mix order. The same dataset, plants, threshold, and seed always
|
|
161
|
+
* produce the same seeded order and the same seal.
|
|
162
|
+
*/
|
|
163
|
+
seed?: number;
|
|
164
|
+
/**
|
|
165
|
+
* The grade at or above which the graded policy's own gate accepts an item.
|
|
166
|
+
* Supply the threshold your gate uses; the default suits a judge scoring in
|
|
167
|
+
* [0, 1] with a pass at the midpoint.
|
|
168
|
+
*/
|
|
169
|
+
acceptThreshold?: number;
|
|
170
|
+
}
|
|
171
|
+
/**
|
|
172
|
+
* Mix plants into a grading set and seal which items they are.
|
|
173
|
+
*
|
|
174
|
+
* Refusals: a duplicate id anywhere in the mixed set (the join is by id, so a
|
|
175
|
+
* repeat makes one of the two unscoreable), a plant whose item id collides
|
|
176
|
+
* with a dataset item (it would shadow real work and grade it as a plant), a
|
|
177
|
+
* dataset item with no usable label, and a plant whose expectation the run's
|
|
178
|
+
* `acceptThreshold` contradicts.
|
|
179
|
+
*/
|
|
180
|
+
declare function seedPlants(dataset: readonly GoldenItem[], plants: readonly Plant[], options?: SeedPlantsOptions): SeededGradingSet;
|
|
181
|
+
/**
|
|
182
|
+
* One grader outcome for one item. `score: null` says the grader ran and
|
|
183
|
+
* declined to decide — the check never tested the item's defect. Any
|
|
184
|
+
* `CandidateScore` from a judge run is already a valid outcome.
|
|
185
|
+
*/
|
|
186
|
+
interface PlantOutcome {
|
|
187
|
+
itemId: string;
|
|
188
|
+
score: number | null;
|
|
189
|
+
}
|
|
190
|
+
/**
|
|
191
|
+
* `evaluated` — a rate stands. `incomplete` — a seeded id had no result.
|
|
192
|
+
* `not_evaluated` — nothing was seeded, or nothing seeded was decided.
|
|
193
|
+
*/
|
|
194
|
+
type CatchRateStatus = 'evaluated' | 'incomplete' | 'not_evaluated';
|
|
195
|
+
interface PlantKindCounts {
|
|
196
|
+
seeded: number;
|
|
197
|
+
caught: number;
|
|
198
|
+
missed: number;
|
|
199
|
+
indecisive: number;
|
|
200
|
+
/** caught / (caught + missed), or null when the report is not `evaluated`. */
|
|
201
|
+
rate: number | null;
|
|
202
|
+
}
|
|
203
|
+
/**
|
|
204
|
+
* How the same grader treated the unseeded items of the same set. No labels
|
|
205
|
+
* are needed to read it: a grader that refuses everything scores `rate` 1.0
|
|
206
|
+
* on reject-plants, and a `rejectionRate` of 1.0 here is what separates that
|
|
207
|
+
* reflex from discrimination.
|
|
208
|
+
*/
|
|
209
|
+
interface UnseededRejection {
|
|
210
|
+
n: number;
|
|
211
|
+
decided: number;
|
|
212
|
+
rejected: number;
|
|
213
|
+
/** rejected / decided, or null when nothing unseeded was decided. */
|
|
214
|
+
rejectionRate: number | null;
|
|
215
|
+
}
|
|
216
|
+
interface CatchRateReport {
|
|
217
|
+
status: CatchRateStatus;
|
|
218
|
+
/** Why the status is not `evaluated`. Absent when it is. */
|
|
219
|
+
reason?: string;
|
|
220
|
+
seeded: number;
|
|
221
|
+
caught: number;
|
|
222
|
+
missed: number;
|
|
223
|
+
/**
|
|
224
|
+
* The grader returned a result and declined to decide. Counted apart: it
|
|
225
|
+
* enters neither side of `rate`.
|
|
226
|
+
*/
|
|
227
|
+
indecisive: number;
|
|
228
|
+
/** caught / (caught + missed), or null unless the status is `evaluated`. */
|
|
229
|
+
rate: number | null;
|
|
230
|
+
/** One entry per kind actually seeded. A kind nobody seeded is absent, never zero. */
|
|
231
|
+
byKind: Partial<Record<PlantKind, PlantKindCounts>>;
|
|
232
|
+
/** Plant ids the grader graded as the seed says it must not. */
|
|
233
|
+
missedIds: string[];
|
|
234
|
+
/** Plant ids with no result at all — the reason a status is `incomplete`. */
|
|
235
|
+
missingIds: string[];
|
|
236
|
+
unseeded: UnseededRejection;
|
|
237
|
+
}
|
|
238
|
+
/**
|
|
239
|
+
* Score a grading run against its sealed manifest.
|
|
240
|
+
*
|
|
241
|
+
* Refusals: a manifest whose contents no longer match its seal, a result for
|
|
242
|
+
* an id the manifest never handed out, and a repeated result id. Each says
|
|
243
|
+
* the results and the manifest describe different runs, and a rate computed
|
|
244
|
+
* across two runs is a fabrication.
|
|
245
|
+
*/
|
|
246
|
+
declare function catchRate(results: readonly PlantOutcome[], manifest: PlantManifest): CatchRateReport;
|
|
247
|
+
//#endregion
|
|
88
248
|
//#region src/meta-eval/sentinel.d.ts
|
|
89
249
|
declare const SENTINEL_METRIC_NAMES: readonly ['irr', 'calibrationKappa', 'sentinelPassRate'];
|
|
90
250
|
type SentinelMetricName = (typeof SENTINEL_METRIC_NAMES)[number];
|
|
@@ -215,5 +375,5 @@ interface EvalHealthStamp {
|
|
|
215
375
|
*/
|
|
216
376
|
declare function evalHealthStamp(report: SentinelReport): EvalHealthStamp;
|
|
217
377
|
//#endregion
|
|
218
|
-
export { CalibrationBin, CalibrationOptions, CalibrationPair, CalibrationReport, CorrelationResult, CorrelationStudyOptions, CorrelationStudyResult, DeploymentOutcome, EvalHealthStamp, EvalMetricSpec, FileSystemOutcomeStore, FileSystemOutcomeStoreOptions, InMemoryOutcomeStore, JudgeSentinelOptions, OutcomeFilter, OutcomePair, OutcomeStore, RubricOutcomePair, RubricPredictiveValidityInput, RubricPredictiveValidityReport, RubricRanking, SentinelMetricName, SentinelMetrics, SentinelReport, SentinelSetOptions, SentinelSnapshot, SentinelStore, SentinelThresholds, SentinelTrend, SnapshotMeta, calibrationCurve, correlationStudy, evalHealthStamp, fileSentinelStore, inMemorySentinelStore, judgeSentinelReport, rubricPredictiveValidity, snapshotFromAgreement, snapshotFromCalibration, snapshotFromSentinelSet, validateSentinelSnapshot };
|
|
378
|
+
export { CalibrationBin, CalibrationOptions, CalibrationPair, CalibrationReport, CatchRateReport, CatchRateStatus, CorrelationResult, CorrelationStudyOptions, CorrelationStudyResult, DeploymentOutcome, EvalHealthStamp, EvalMetricSpec, FileSystemOutcomeStore, FileSystemOutcomeStoreOptions, InMemoryOutcomeStore, JudgeSentinelOptions, OutcomeFilter, OutcomePair, OutcomeStore, Plant, PlantExpectation, PlantKind, PlantKindCounts, PlantManifest, PlantOutcome, RubricOutcomePair, RubricPredictiveValidityInput, RubricPredictiveValidityReport, RubricRanking, SeedPlantsOptions, SeededGradingSet, SentinelMetricName, SentinelMetrics, SentinelReport, SentinelSetOptions, SentinelSnapshot, SentinelStore, SentinelThresholds, SentinelTrend, SnapshotMeta, UnseededRejection, calibrationCurve, catchRate, correlationStudy, definePlant, evalHealthStamp, fileSentinelStore, inMemorySentinelStore, judgeSentinelReport, rubricPredictiveValidity, seedPlants, snapshotFromAgreement, snapshotFromCalibration, snapshotFromSentinelSet, validateSentinelSnapshot };
|
|
219
379
|
//# sourceMappingURL=index.d.ts.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"index.d.ts","names":[],"sources":["../../src/meta-eval/correlation-study.ts","../../src/meta-eval/calibration.ts","../../src/meta-eval/sentinel.ts"],"mappings":"
|
|
1
|
+
{"version":3,"file":"index.d.ts","names":[],"sources":["../../src/meta-eval/correlation-study.ts","../../src/meta-eval/calibration.ts","../../src/meta-eval/plants.ts","../../src/meta-eval/sentinel.ts"],"mappings":";;;;;;;;UAkBiB;EACf;;;EAGA,WAAW,KAAK,KAAK,OAAO,eAAe;;UAG5B;EACf;EACA;;UAGe;EACf;EACA;EACA;EACA;EACA;;EAEA;IAAe;IAAe;;;EAE9B;;UAGe;EACf,OAAO;EACP;EACA;;UAGe;;EAEf;;EAEA,gBAAgB;;EAEhB;;EAEA;;;EAGA;;iBAGoB,iBACpB,YAAY,YACZ,cAAc,cACd,aAAa,kBACb,8BACA,UAAS,0BACR,QAAQ;;;UCtDM;EACf;EACA;EACA;EACA;EACA;;EAEA;;UAGe;EACf;EACA;EACA;EACA,MAAM;;EAEN;;EAEA;;UAGe;EACf;;EAEA;;EAEA;IAAU;IAAY;;;UAGP;EACf;EACA;;iBAGoB,iBACpB,YAAY,YACZ,cAAc,cACd,YAAY,gBACZ,uBACA,UAAS,qBACR,QAAQ;;;;;;;;KCDC;;;;;;;;;;KAkBA;UAOK;;EAEf;EACA,MAAM;;EAEN,MAAM;;EAEN,iBAAiB;;;;;;;;;;;;;;iBAeH,YAAY;EAC1B;EACA,MAAM;EACN,MAAM;EACN,iBAAiB;IACf;UAiDa;;;;;;EAMf,MAAM;;EAEN;;EAEA;;EAEA;;EAEA,QAAQ;;UAGO;;;;;;EAMf,OAAO;EACP,UAAU;;UAGK;;;;;EAKf;;;;;;EAMA;;;;;;;;;;;iBAYc,WACd,kBAAkB,cAClB,iBAAiB,SACjB,UAAS,oBACR;;;;;;UAqGc;EACf;EACA;;;;;;KAOU;UAEK;EACf;EACA;EACA;EACA;;EAEA;;;;;;;;UASe;EACf;EACA;EACA;;EAEA;;UAGe;EACf,QAAQ;;EAER;EACA;EACA;EACA;;;;;EAKA;;EAEA;;EAEA,QAAQ,QAAQ,OAAO,WAAW;;EAElC;;EAEA;EACA,UAAU;;;;;;;;;;iBAWI,UACd,kBAAkB,gBAClB,UAAU,gBACT;;;cCpUG;KAEM,6BAA6B;UAExB;;EAEf;;EAEA;;EAEA;;UAGe;;EAEf;EACA;;EAEA;;;;;;;EAOA,SAAS;;;UAIM;EACf;EACA;EACA;;;iBAcc,yBAAyB,UAAU,kBAAkB;;;;;;iBAiCrD,wBACd,QAAQ,oBAAoB,6BAC5B,MAAM,eACL;;;;;;;;iBAmBa,sBACd,QAAQ,wBAAwB,qBAChC,MAAM,eACL;UAYc;;EAEf;;;;;;;;;iBAUc,wBACd,QAAQ,kBACR,QAAQ,cACR,MAAM,cACN,UAAS,qBACR;UA4Cc;EACf,OAAO,UAAU,mBAAmB;;EAEpC,QAAQ,mBAAmB,QAAQ;;iBAGrB,sBAAsB,UAAS,qBAA0B;;;;;;;iBAqBzD,kBAAkB,eAAe;UA0ChC;;EAEf;;EAEA;;EAEA;;EAEA;;UAGe;;EAEf;EACA,aAAa;;EAEb,cAAc;;UAGC;EACf;EACA,QAAQ;;EAER,OAAO;;EAEP;;EAEA;;EAEA;EACA;EACA;;UAGe;EACf,UAAU;EACV;EACA;;;;;;;EAOA;;iBAKc,oBACd,SAAS,oBACT,MAAM,uBACL;UA6Gc;EACf;EACA;;;;;;;;;;iBAWc,gBAAgB,QAAQ,iBAAiB"}
|
package/dist/meta-eval/index.js
CHANGED
|
@@ -1,5 +1,7 @@
|
|
|
1
|
-
import { s as ValidationError } from "../errors-Dngq5h35.js";
|
|
1
|
+
import { n as CaptureIntegrityError, s as ValidationError } from "../errors-Dngq5h35.js";
|
|
2
|
+
import { a as hashCanonical } from "../canonical-DPyQ_rpt.js";
|
|
2
3
|
import { i as makeRng } from "../internal-BMFSR8Ns.js";
|
|
4
|
+
import { t as mulberry32 } from "../random-Dn5fPWkt.js";
|
|
3
5
|
import { a as spearmanR, r as pearsonR } from "../descriptive-1V17A-qa.js";
|
|
4
6
|
import { u as runMetricExtractor } from "../query-BPGMVlbM.js";
|
|
5
7
|
import { t as analyzeSeries } from "../series-convergence-CjO2QdRW.js";
|
|
@@ -226,6 +228,289 @@ function bootstrapPearsonCi(xs, ys, iterations, seed) {
|
|
|
226
228
|
};
|
|
227
229
|
}
|
|
228
230
|
//#endregion
|
|
231
|
+
//#region src/meta-eval/plants.ts
|
|
232
|
+
/**
|
|
233
|
+
* Plants — seeded known-wrong items that measure the grader, not the work.
|
|
234
|
+
*
|
|
235
|
+
* A grading run reports how the work scored. It cannot report whether the
|
|
236
|
+
* grader would have noticed a wrong answer, because every item it saw was
|
|
237
|
+
* authored in good faith. A plant closes that hole: an item authored wrong by
|
|
238
|
+
* construction is mixed into the live set, graded by the same path as
|
|
239
|
+
* everything else, and the share of plants the grader refused is the catch
|
|
240
|
+
* rate.
|
|
241
|
+
*
|
|
242
|
+
* Measured motive: a sibling lab ran a deliverable gate that accepted any
|
|
243
|
+
* non-empty submission. It produced six false certifications in seventeen
|
|
244
|
+
* deliveries, and no agent lied — the gate never asked a question the format
|
|
245
|
+
* could fail. A catch rate is the number that would have shown it on day one.
|
|
246
|
+
*
|
|
247
|
+
* This module composes existing primitives rather than adding parallel ones:
|
|
248
|
+
*
|
|
249
|
+
* - A plant IS a {@link GoldenItem} from `../judge-calibration`. Its
|
|
250
|
+
* `humanScore` is the grade a working grader owes the item, so the same
|
|
251
|
+
* array feeds `calibrateJudge` unchanged.
|
|
252
|
+
* - The grader's output is `CandidateScore[]`, the array `calibrateJudge` and
|
|
253
|
+
* `snapshotFromSentinelSet` already consume.
|
|
254
|
+
* - "Caught" is `snapshotFromSentinelSet`'s join with the labels inverted:
|
|
255
|
+
* the grade lands on the side of `acceptThreshold` the seed demands.
|
|
256
|
+
* - The manifest is sealed with `hashCanonical` from `../ledger-core/canonical`,
|
|
257
|
+
* the digest the sealed-experiment path uses, so the answer key cannot be
|
|
258
|
+
* revised once the results are in.
|
|
259
|
+
*
|
|
260
|
+
* Blindness has two halves, and this module owns one. It never puts a plant
|
|
261
|
+
* flag on a graded item: `seedPlants` returns the mixed set and a manifest,
|
|
262
|
+
* and only the manifest knows which ids are seeded. Keeping the manifest out
|
|
263
|
+
* of the graded workspace and publishing its `seal` before grading is the
|
|
264
|
+
* caller's half; {@link catchRate} refuses a manifest whose contents no longer
|
|
265
|
+
* match its seal.
|
|
266
|
+
*
|
|
267
|
+
* Refusals, because a catch rate that cannot refuse is not a measurement:
|
|
268
|
+
*
|
|
269
|
+
* - a seeded id with no result makes the report `incomplete`, never a rate
|
|
270
|
+
* over the results that did come back;
|
|
271
|
+
* - zero seeded plants makes it `not_evaluated`, never 1.0;
|
|
272
|
+
* - a result for an id the manifest never handed out is refused outright.
|
|
273
|
+
*/
|
|
274
|
+
const PLANT_KINDS = [
|
|
275
|
+
"wrong-value",
|
|
276
|
+
"self-certifying",
|
|
277
|
+
"unreachable-input",
|
|
278
|
+
"duplicate"
|
|
279
|
+
];
|
|
280
|
+
const PLANT_EXPECTATIONS = ["reject", "accept"];
|
|
281
|
+
/** The label boundary a plant record is checked against at definition time. */
|
|
282
|
+
const RECORD_LABEL_BOUNDARY = .5;
|
|
283
|
+
/**
|
|
284
|
+
* Build one plant record and refuse an incoherent one.
|
|
285
|
+
*
|
|
286
|
+
* The refusal that matters is the last: a record whose `expectedVerdict`
|
|
287
|
+
* disagrees with `item.humanScore` inverts the measurement silently, because
|
|
288
|
+
* the same item then reads as wrong here and as correct to every calibration
|
|
289
|
+
* instrument that joins on the id.
|
|
290
|
+
*
|
|
291
|
+
* `item` is copied field by field so a later mutation of the caller's object
|
|
292
|
+
* cannot change what the manifest sealed, and `group` is dropped when it is
|
|
293
|
+
* absent so the record always has a canonical JSON form.
|
|
294
|
+
*/
|
|
295
|
+
function definePlant(input) {
|
|
296
|
+
const { id, kind, item, expectedVerdict } = input;
|
|
297
|
+
if (typeof id !== "string" || id.trim() === "") throw new ValidationError("definePlant: id must be a non-empty string");
|
|
298
|
+
if (!PLANT_KINDS.includes(kind)) throw new ValidationError(`definePlant: plant "${id}" has kind ${JSON.stringify(kind)}; expected one of ${PLANT_KINDS.join(", ")}`);
|
|
299
|
+
if (!PLANT_EXPECTATIONS.includes(expectedVerdict)) throw new ValidationError(`definePlant: plant "${id}" has expectedVerdict ${JSON.stringify(expectedVerdict)}; expected one of ${PLANT_EXPECTATIONS.join(", ")}`);
|
|
300
|
+
if (typeof item.itemId !== "string" || item.itemId.trim() === "") throw new ValidationError(`definePlant: plant "${id}" has an empty item.itemId`);
|
|
301
|
+
if (!Number.isFinite(item.humanScore) || item.humanScore < 0 || item.humanScore > 1) throw new ValidationError(`definePlant: plant "${id}" has humanScore ${item.humanScore}; expected a finite number in [0, 1]`);
|
|
302
|
+
if (item.group !== void 0 && typeof item.group !== "string") throw new ValidationError(`definePlant: plant "${id}" has a non-string item.group`);
|
|
303
|
+
if (item.humanScore === RECORD_LABEL_BOUNDARY) throw new ValidationError(`definePlant: plant "${id}" has humanScore ${RECORD_LABEL_BOUNDARY}, which states neither a rejection nor an acceptance`);
|
|
304
|
+
const labelSays = item.humanScore < RECORD_LABEL_BOUNDARY ? "reject" : "accept";
|
|
305
|
+
if (labelSays !== expectedVerdict) throw new ValidationError(`definePlant: plant "${id}" expects the grader to ${expectedVerdict} it, but humanScore ${item.humanScore} says ${labelSays}`);
|
|
306
|
+
return {
|
|
307
|
+
id,
|
|
308
|
+
kind,
|
|
309
|
+
item: {
|
|
310
|
+
itemId: item.itemId,
|
|
311
|
+
humanScore: item.humanScore,
|
|
312
|
+
...item.group === void 0 ? {} : { group: item.group }
|
|
313
|
+
},
|
|
314
|
+
expectedVerdict
|
|
315
|
+
};
|
|
316
|
+
}
|
|
317
|
+
/**
|
|
318
|
+
* Mix plants into a grading set and seal which items they are.
|
|
319
|
+
*
|
|
320
|
+
* Refusals: a duplicate id anywhere in the mixed set (the join is by id, so a
|
|
321
|
+
* repeat makes one of the two unscoreable), a plant whose item id collides
|
|
322
|
+
* with a dataset item (it would shadow real work and grade it as a plant), a
|
|
323
|
+
* dataset item with no usable label, and a plant whose expectation the run's
|
|
324
|
+
* `acceptThreshold` contradicts.
|
|
325
|
+
*/
|
|
326
|
+
function seedPlants(dataset, plants, options = {}) {
|
|
327
|
+
const seed = options.seed ?? 7;
|
|
328
|
+
const acceptThreshold = options.acceptThreshold ?? .5;
|
|
329
|
+
if (!Number.isFinite(seed)) throw new ValidationError(`seedPlants: seed must be a finite number, got ${seed}`);
|
|
330
|
+
if (!Number.isFinite(acceptThreshold) || acceptThreshold <= 0 || acceptThreshold > 1) throw new ValidationError(`seedPlants: acceptThreshold must be a finite number in (0, 1], got ${acceptThreshold}`);
|
|
331
|
+
const datasetItems = [];
|
|
332
|
+
const seen = /* @__PURE__ */ new Set();
|
|
333
|
+
for (const item of dataset) {
|
|
334
|
+
if (typeof item.itemId !== "string" || item.itemId.trim() === "") throw new ValidationError("seedPlants: a dataset item has an empty itemId");
|
|
335
|
+
if (!Number.isFinite(item.humanScore)) throw new ValidationError(`seedPlants: dataset item "${item.itemId}" has a non-finite humanScore`);
|
|
336
|
+
if (seen.has(item.itemId)) throw new ValidationError(`seedPlants: duplicate dataset itemId "${item.itemId}"`);
|
|
337
|
+
seen.add(item.itemId);
|
|
338
|
+
datasetItems.push({
|
|
339
|
+
itemId: item.itemId,
|
|
340
|
+
humanScore: item.humanScore,
|
|
341
|
+
...item.group === void 0 ? {} : { group: item.group }
|
|
342
|
+
});
|
|
343
|
+
}
|
|
344
|
+
const sealedPlants = [];
|
|
345
|
+
const plantIds = /* @__PURE__ */ new Set();
|
|
346
|
+
for (const plant of plants) {
|
|
347
|
+
const record = definePlant(plant);
|
|
348
|
+
if (plantIds.has(record.id)) throw new ValidationError(`seedPlants: duplicate plant id "${record.id}"`);
|
|
349
|
+
if (seen.has(record.item.itemId)) throw new ValidationError(`seedPlants: plant "${record.id}" reuses itemId "${record.item.itemId}", which is already in the set`);
|
|
350
|
+
const thresholdSays = record.item.humanScore >= acceptThreshold ? "accept" : "reject";
|
|
351
|
+
if (thresholdSays !== record.expectedVerdict) throw new ValidationError(`seedPlants: plant "${record.id}" expects the grader to ${record.expectedVerdict} it, but humanScore ${record.item.humanScore} is on the ${thresholdSays} side of acceptThreshold ${acceptThreshold}`);
|
|
352
|
+
plantIds.add(record.id);
|
|
353
|
+
seen.add(record.item.itemId);
|
|
354
|
+
sealedPlants.push(record);
|
|
355
|
+
}
|
|
356
|
+
const items = shuffled([...datasetItems, ...sealedPlants.map((plant) => plant.item)], mulberry32(seed));
|
|
357
|
+
const itemIds = items.map((item) => item.itemId);
|
|
358
|
+
return {
|
|
359
|
+
items,
|
|
360
|
+
manifest: {
|
|
361
|
+
seal: sealManifest({
|
|
362
|
+
seed,
|
|
363
|
+
acceptThreshold,
|
|
364
|
+
itemIds,
|
|
365
|
+
plants: sealedPlants
|
|
366
|
+
}),
|
|
367
|
+
seed,
|
|
368
|
+
acceptThreshold,
|
|
369
|
+
itemIds,
|
|
370
|
+
plants: sealedPlants
|
|
371
|
+
}
|
|
372
|
+
};
|
|
373
|
+
}
|
|
374
|
+
/**
|
|
375
|
+
* Order by one independent uniform key per item, which is a uniform
|
|
376
|
+
* permutation and a pure function of the seed. A comparator that returns a
|
|
377
|
+
* fresh random sign instead is neither: it is not a consistent ordering, so
|
|
378
|
+
* the permutation it produces is biased and depends on the sort algorithm.
|
|
379
|
+
*/
|
|
380
|
+
function shuffled(items, random) {
|
|
381
|
+
return items.map((item) => ({
|
|
382
|
+
item,
|
|
383
|
+
key: random()
|
|
384
|
+
})).sort((left, right) => left.key - right.key).map((entry) => entry.item);
|
|
385
|
+
}
|
|
386
|
+
function sealManifest(contents) {
|
|
387
|
+
return hashCanonical({
|
|
388
|
+
scheme: "agent-eval.plant-manifest.v1",
|
|
389
|
+
seed: contents.seed,
|
|
390
|
+
acceptThreshold: contents.acceptThreshold,
|
|
391
|
+
itemIds: contents.itemIds,
|
|
392
|
+
plants: contents.plants
|
|
393
|
+
});
|
|
394
|
+
}
|
|
395
|
+
/**
|
|
396
|
+
* Score a grading run against its sealed manifest.
|
|
397
|
+
*
|
|
398
|
+
* Refusals: a manifest whose contents no longer match its seal, a result for
|
|
399
|
+
* an id the manifest never handed out, and a repeated result id. Each says
|
|
400
|
+
* the results and the manifest describe different runs, and a rate computed
|
|
401
|
+
* across two runs is a fabrication.
|
|
402
|
+
*/
|
|
403
|
+
function catchRate(results, manifest) {
|
|
404
|
+
const expectedSeal = sealManifest({
|
|
405
|
+
seed: manifest.seed,
|
|
406
|
+
acceptThreshold: manifest.acceptThreshold,
|
|
407
|
+
itemIds: manifest.itemIds,
|
|
408
|
+
plants: manifest.plants
|
|
409
|
+
});
|
|
410
|
+
if (expectedSeal !== manifest.seal) throw new CaptureIntegrityError(`catchRate: manifest contents hash to ${expectedSeal} but the manifest carries seal ${manifest.seal} — the plant set changed after it was sealed`);
|
|
411
|
+
const handedOut = new Set(manifest.itemIds);
|
|
412
|
+
const scoreByItemId = /* @__PURE__ */ new Map();
|
|
413
|
+
for (const result of results) {
|
|
414
|
+
if (!handedOut.has(result.itemId)) throw new ValidationError(`catchRate: result for "${result.itemId}", which this manifest never handed out`);
|
|
415
|
+
if (scoreByItemId.has(result.itemId)) throw new ValidationError(`catchRate: duplicate result for "${result.itemId}"`);
|
|
416
|
+
if (result.score !== null && !Number.isFinite(result.score)) throw new ValidationError(`catchRate: result for "${result.itemId}" has score ${result.score}; expected a finite number or null`);
|
|
417
|
+
scoreByItemId.set(result.itemId, result.score);
|
|
418
|
+
}
|
|
419
|
+
const byKind = {};
|
|
420
|
+
const missedIds = [];
|
|
421
|
+
const missingIds = [];
|
|
422
|
+
let caught = 0;
|
|
423
|
+
let missed = 0;
|
|
424
|
+
let indecisive = 0;
|
|
425
|
+
for (const plant of manifest.plants) {
|
|
426
|
+
let counts = byKind[plant.kind];
|
|
427
|
+
if (counts === void 0) {
|
|
428
|
+
counts = {
|
|
429
|
+
seeded: 0,
|
|
430
|
+
caught: 0,
|
|
431
|
+
missed: 0,
|
|
432
|
+
indecisive: 0,
|
|
433
|
+
rate: null
|
|
434
|
+
};
|
|
435
|
+
byKind[plant.kind] = counts;
|
|
436
|
+
}
|
|
437
|
+
counts.seeded += 1;
|
|
438
|
+
if (!scoreByItemId.has(plant.item.itemId)) {
|
|
439
|
+
missingIds.push(plant.id);
|
|
440
|
+
continue;
|
|
441
|
+
}
|
|
442
|
+
const score = scoreByItemId.get(plant.item.itemId) ?? null;
|
|
443
|
+
if (score === null) {
|
|
444
|
+
indecisive += 1;
|
|
445
|
+
counts.indecisive += 1;
|
|
446
|
+
continue;
|
|
447
|
+
}
|
|
448
|
+
if ((score >= manifest.acceptThreshold ? "accept" : "reject") === plant.expectedVerdict) {
|
|
449
|
+
caught += 1;
|
|
450
|
+
counts.caught += 1;
|
|
451
|
+
} else {
|
|
452
|
+
missed += 1;
|
|
453
|
+
counts.missed += 1;
|
|
454
|
+
missedIds.push(plant.id);
|
|
455
|
+
}
|
|
456
|
+
}
|
|
457
|
+
const seeded = manifest.plants.length;
|
|
458
|
+
const decided = caught + missed;
|
|
459
|
+
const report = {
|
|
460
|
+
status: "evaluated",
|
|
461
|
+
seeded,
|
|
462
|
+
caught,
|
|
463
|
+
missed,
|
|
464
|
+
indecisive,
|
|
465
|
+
rate: null,
|
|
466
|
+
byKind,
|
|
467
|
+
missedIds,
|
|
468
|
+
missingIds,
|
|
469
|
+
unseeded: unseededRejection(manifest, scoreByItemId)
|
|
470
|
+
};
|
|
471
|
+
if (seeded === 0) return {
|
|
472
|
+
...report,
|
|
473
|
+
status: "not_evaluated",
|
|
474
|
+
reason: "no plant was seeded, so the grader was never asked a question it could fail"
|
|
475
|
+
};
|
|
476
|
+
if (missingIds.length > 0) return {
|
|
477
|
+
...report,
|
|
478
|
+
status: "incomplete",
|
|
479
|
+
reason: `${missingIds.length} of ${seeded} seeded plants have no result: ${missingIds.join(", ")}`
|
|
480
|
+
};
|
|
481
|
+
if (decided === 0) return {
|
|
482
|
+
...report,
|
|
483
|
+
status: "not_evaluated",
|
|
484
|
+
reason: `all ${seeded} seeded plants are indecisive: no check tested the seeded defect`
|
|
485
|
+
};
|
|
486
|
+
for (const counts of Object.values(byKind)) {
|
|
487
|
+
const kindDecided = counts.caught + counts.missed;
|
|
488
|
+
counts.rate = kindDecided === 0 ? null : counts.caught / kindDecided;
|
|
489
|
+
}
|
|
490
|
+
report.rate = caught / decided;
|
|
491
|
+
return report;
|
|
492
|
+
}
|
|
493
|
+
function unseededRejection(manifest, scoreByItemId) {
|
|
494
|
+
const plantItemIds = new Set(manifest.plants.map((plant) => plant.item.itemId));
|
|
495
|
+
let n = 0;
|
|
496
|
+
let decided = 0;
|
|
497
|
+
let rejected = 0;
|
|
498
|
+
for (const itemId of manifest.itemIds) {
|
|
499
|
+
if (plantItemIds.has(itemId)) continue;
|
|
500
|
+
n += 1;
|
|
501
|
+
const score = scoreByItemId.get(itemId);
|
|
502
|
+
if (score === void 0 || score === null) continue;
|
|
503
|
+
decided += 1;
|
|
504
|
+
if (score < manifest.acceptThreshold) rejected += 1;
|
|
505
|
+
}
|
|
506
|
+
return {
|
|
507
|
+
n,
|
|
508
|
+
decided,
|
|
509
|
+
rejected,
|
|
510
|
+
rejectionRate: decided === 0 ? null : rejected / decided
|
|
511
|
+
};
|
|
512
|
+
}
|
|
513
|
+
//#endregion
|
|
229
514
|
//#region src/meta-eval/sentinel.ts
|
|
230
515
|
/**
|
|
231
516
|
* Judge sentinel — eval trustworthiness as a continuously measured,
|
|
@@ -499,6 +784,6 @@ function evalHealthStamp(report) {
|
|
|
499
784
|
};
|
|
500
785
|
}
|
|
501
786
|
//#endregion
|
|
502
|
-
export { FileSystemOutcomeStore, InMemoryOutcomeStore, calibrationCurve, correlationStudy, evalHealthStamp, fileSentinelStore, inMemorySentinelStore, judgeSentinelReport, rubricPredictiveValidity, snapshotFromAgreement, snapshotFromCalibration, snapshotFromSentinelSet, validateSentinelSnapshot };
|
|
787
|
+
export { FileSystemOutcomeStore, InMemoryOutcomeStore, calibrationCurve, catchRate, correlationStudy, definePlant, evalHealthStamp, fileSentinelStore, inMemorySentinelStore, judgeSentinelReport, rubricPredictiveValidity, seedPlants, snapshotFromAgreement, snapshotFromCalibration, snapshotFromSentinelSet, validateSentinelSnapshot };
|
|
503
788
|
|
|
504
789
|
//# sourceMappingURL=index.js.map
|