@tangle-network/agent-eval 0.144.6 → 0.144.8
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +30 -0
- package/README.md +2 -0
- package/dist/{benchmark-J9Qe6j2_.d.ts → agent-profile-CgDTo40f.d.ts} +91 -4
- package/dist/agent-profile-CgDTo40f.d.ts.map +1 -0
- package/dist/analyst/index.d.ts +470 -88
- package/dist/analyst/index.d.ts.map +1 -1
- package/dist/analyst/index.js +24 -5
- package/dist/analyst/index.js.map +1 -1
- package/dist/{analyze-runs-DWIvOAGk.js → analyze-runs-BNNK7irB.js} +3 -3
- package/dist/{analyze-runs-DWIvOAGk.js.map → analyze-runs-BNNK7irB.js.map} +1 -1
- package/dist/{baseline-D_fT6277.d.ts → baseline-CavEbRyH.d.ts} +2 -23
- package/dist/baseline-CavEbRyH.d.ts.map +1 -0
- package/dist/{benchmark-command-CQd78YHt.js → benchmark-command-BKENp2s5.js} +662 -605
- package/dist/benchmark-command-BKENp2s5.js.map +1 -0
- package/dist/benchmarks/index.d.ts +1 -1
- package/dist/benchmarks/index.js +1 -1
- package/dist/{benchmarks-BEOkuvIg.js → benchmarks-BlPmjd88.js} +4 -4
- package/dist/{benchmarks-BEOkuvIg.js.map → benchmarks-BlPmjd88.js.map} +1 -1
- package/dist/campaign/index.d.ts +7 -5
- package/dist/campaign/index.js +5 -3
- package/dist/{campaign-CXsdyym7.js → campaign--HVSuvV0.js} +17 -301
- package/dist/campaign--HVSuvV0.js.map +1 -0
- package/dist/cli.js +2 -2
- package/dist/{client-0JI64ovJ.d.ts → client-DjXROWpx.d.ts} +3 -3
- package/dist/{client-0JI64ovJ.d.ts.map → client-DjXROWpx.d.ts.map} +1 -1
- package/dist/{completion-verifier-CBiee74w.d.ts → completion-verifier-foUCLif_.d.ts} +5 -5
- package/dist/{completion-verifier-CBiee74w.d.ts.map → completion-verifier-foUCLif_.d.ts.map} +1 -1
- package/dist/contract/index.d.ts +9 -8
- package/dist/contract/index.d.ts.map +1 -1
- package/dist/contract/index.js +7 -6
- package/dist/contract/index.js.map +1 -1
- package/dist/control.d.ts +2 -2
- package/dist/control.js +1 -1
- package/dist/counterfactual-CWPTrMH7.js +126 -0
- package/dist/counterfactual-CWPTrMH7.js.map +1 -0
- package/dist/counterfactual-CxmxAONP.d.ts +72 -0
- package/dist/counterfactual-CxmxAONP.d.ts.map +1 -0
- package/dist/{default-registry-J9m-_tya.d.ts → default-registry-BZhStdjl.d.ts} +6 -5
- package/dist/default-registry-BZhStdjl.d.ts.map +1 -0
- package/dist/{default-registry-Dta70shL.js → default-registry-BaQXW1Ow.js} +2 -2
- package/dist/{default-registry-Dta70shL.js.map → default-registry-BaQXW1Ow.js.map} +1 -1
- package/dist/{dspy-rlm-engine-19FQEMBK.js → dspy-rlm-engine-BiN49gK6.js} +4 -3
- package/dist/{dspy-rlm-engine-19FQEMBK.js.map → dspy-rlm-engine-BiN49gK6.js.map} +1 -1
- package/dist/{tool-groups-CK0JCkqO.d.ts → engine-nB64f48I.d.ts} +18 -31
- package/dist/engine-nB64f48I.d.ts.map +1 -0
- package/dist/{eval-campaign-CfLQQs9B.js → eval-campaign-DNjCvAm-.js} +7 -6
- package/dist/{eval-campaign-CfLQQs9B.js.map → eval-campaign-DNjCvAm-.js.map} +1 -1
- package/dist/{exact-types-CBYF5MGd.d.ts → exact-types-Djvzosly.d.ts} +2 -2
- package/dist/{exact-types-CBYF5MGd.d.ts.map → exact-types-Djvzosly.d.ts.map} +1 -1
- package/dist/exec-BLtYZdWo.js +49 -0
- package/dist/exec-BLtYZdWo.js.map +1 -0
- package/dist/experiment/index.d.ts +802 -0
- package/dist/experiment/index.d.ts.map +1 -0
- package/dist/experiment/index.js +1108 -0
- package/dist/experiment/index.js.map +1 -0
- package/dist/experiment-tracker-CnRICnMl.js +500 -0
- package/dist/experiment-tracker-CnRICnMl.js.map +1 -0
- package/dist/experiment-tracker-IMntXr6J.d.ts +299 -0
- package/dist/experiment-tracker-IMntXr6J.d.ts.map +1 -0
- package/dist/{external-optimizer-contracts-Q9c0L3eY.d.ts → external-optimizer-contracts-CZuJNcT5.d.ts} +2 -2
- package/dist/{external-optimizer-contracts-Q9c0L3eY.d.ts.map → external-optimizer-contracts-CZuJNcT5.d.ts.map} +1 -1
- package/dist/{feedback-trajectory-GgoS0-MK.d.ts → feedback-trajectory-Rh280oXo.d.ts} +3 -3
- package/dist/{feedback-trajectory-GgoS0-MK.d.ts.map → feedback-trajectory-Rh280oXo.d.ts.map} +1 -1
- package/dist/fuzz.d.ts +1 -1
- package/dist/hosted/index.d.ts +3 -3
- package/dist/{index-BZUe-ODI.d.ts → index-BZ3-y4YL.d.ts} +3 -3
- package/dist/{index-BZUe-ODI.d.ts.map → index-BZ3-y4YL.d.ts.map} +1 -1
- package/dist/{index-DSC51roc.d.ts → index-C3ssXVLv.d.ts} +2 -2
- package/dist/{index-DSC51roc.d.ts.map → index-C3ssXVLv.d.ts.map} +1 -1
- package/dist/{index-4XwggC10.d.ts → index-C5HOo4ZF2.d.ts} +4 -4
- package/dist/index-C5HOo4ZF2.d.ts.map +1 -0
- package/dist/{index-B6-B0zTB.d.ts → index-CvXXlyz7.d.ts} +2 -2
- package/dist/{index-B6-B0zTB.d.ts.map → index-CvXXlyz7.d.ts.map} +1 -1
- package/dist/{index-Dx1kF3Ez.d.ts → index-CwDrUMe0.d.ts} +2 -2
- package/dist/{index-Dx1kF3Ez.d.ts.map → index-CwDrUMe0.d.ts.map} +1 -1
- package/dist/{index-BIL5vxxt.d.ts → index-Sh2I0DRc.d.ts} +11 -645
- package/dist/index-Sh2I0DRc.d.ts.map +1 -0
- package/dist/index.d.ts +214 -404
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +233 -649
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-DqEsugpr.d.ts → insight-report-C6h6F_4L.d.ts} +3 -3
- package/dist/{insight-report-DqEsugpr.d.ts.map → insight-report-C6h6F_4L.d.ts.map} +1 -1
- package/dist/{integrity-CNGUaGBY.d.ts → integrity-BuqEKu-x.d.ts} +2 -2
- package/dist/{integrity-CNGUaGBY.d.ts.map → integrity-BuqEKu-x.d.ts.map} +1 -1
- package/dist/integrity-MLzHOfV9.js +141 -0
- package/dist/integrity-MLzHOfV9.js.map +1 -0
- package/dist/kind-factory-BHIgPmzS.js.map +1 -1
- package/dist/{llm-client-Dv5BiKLE.js → llm-client-DzvMUsS_.js} +24 -6
- package/dist/llm-client-DzvMUsS_.js.map +1 -0
- package/dist/matrix/index.d.ts +2 -2
- package/dist/meta-eval/index.d.ts +1 -1
- package/dist/{mint-DD-0oQTA.js → mint-CGEkzPLf.js} +2 -2
- package/dist/{mint-DD-0oQTA.js.map → mint-CGEkzPLf.js.map} +1 -1
- package/dist/{multi-layer-verifier-BHY1gWAc.d.ts → multi-layer-verifier-DnAqwl0h.d.ts} +2 -2
- package/dist/{multi-layer-verifier-BHY1gWAc.d.ts.map → multi-layer-verifier-DnAqwl0h.d.ts.map} +1 -1
- package/dist/multishot/index.d.ts +2 -2
- package/dist/openapi.json +1 -1
- package/dist/paired-promotion-decision-B6zJ3gYM.d.ts +114 -0
- package/dist/paired-promotion-decision-B6zJ3gYM.d.ts.map +1 -0
- package/dist/pipelines/index.d.ts +2 -1
- package/dist/pipelines/index.d.ts.map +1 -1
- package/dist/pre-registration-DakwTRXk.js +96 -0
- package/dist/pre-registration-DakwTRXk.js.map +1 -0
- package/dist/prime-bridge-transport-6feEglLf.d.ts +49 -0
- package/dist/prime-bridge-transport-6feEglLf.d.ts.map +1 -0
- package/dist/prime-protocol-BfSalTfR.js +453 -0
- package/dist/prime-protocol-BfSalTfR.js.map +1 -0
- package/dist/profile-cell.js +242 -1
- package/dist/profile-cell.js.map +1 -0
- package/dist/promotion-policy-ChWhTDBH.d.ts +289 -0
- package/dist/promotion-policy-ChWhTDBH.d.ts.map +1 -0
- package/dist/promotion-policy-CrLrmys8.js +682 -0
- package/dist/promotion-policy-CrLrmys8.js.map +1 -0
- package/dist/{propose-review-control-BciUCZoh.js → propose-review-control-qgWLA6E9.js} +2 -2
- package/dist/{propose-review-control-BciUCZoh.js.map → propose-review-control-qgWLA6E9.js.map} +1 -1
- package/dist/{release-report-ChOgpIoQ.d.ts → release-report-CI8uisI1.d.ts} +2 -2
- package/dist/{release-report-ChOgpIoQ.d.ts.map → release-report-CI8uisI1.d.ts.map} +1 -1
- package/dist/{release-report-Sl0xfkFv.js → release-report-Dy39cbFF.js} +3 -3
- package/dist/{release-report-Sl0xfkFv.js.map → release-report-Dy39cbFF.js.map} +1 -1
- package/dist/{replay-Krvb114g.d.ts → replay-DFf-teiC.d.ts} +5 -4
- package/dist/replay-DFf-teiC.d.ts.map +1 -0
- package/dist/{replay-GW61ezMW.js → replay-DyBLaKFc.js} +3 -3
- package/dist/{replay-GW61ezMW.js.map → replay-DyBLaKFc.js.map} +1 -1
- package/dist/reporting.d.ts +3 -3
- package/dist/reporting.js +2 -2
- package/dist/{researcher-xLeNcpKX.d.ts → researcher-BoaxeCzP.d.ts} +4 -4
- package/dist/{researcher-xLeNcpKX.d.ts.map → researcher-BoaxeCzP.d.ts.map} +1 -1
- package/dist/{reward-hacking-CyuzxKly.js → reward-hacking-B2tPL9a3.js} +12 -3
- package/dist/reward-hacking-B2tPL9a3.js.map +1 -0
- package/dist/{reward-hacking-RZgnGWlx.d.ts → reward-hacking-Cf1PtEOz.d.ts} +33 -3
- package/dist/reward-hacking-Cf1PtEOz.d.ts.map +1 -0
- package/dist/rl.d.ts +17 -7
- package/dist/rl.d.ts.map +1 -1
- package/dist/rl.js +16 -7
- package/dist/rl.js.map +1 -1
- package/dist/rollout/index.d.ts +2 -2
- package/dist/rollout/index.js +2 -2
- package/dist/{rollout-C2fD1cf4.js → rollout-2ECTXb2N.js} +2 -2
- package/dist/{rollout-C2fD1cf4.js.map → rollout-2ECTXb2N.js.map} +1 -1
- package/dist/{run-evidence-C6G41MSI.d.ts → run-evidence-BDFFai9R.d.ts} +2 -2
- package/dist/{run-evidence-C6G41MSI.d.ts.map → run-evidence-BDFFai9R.d.ts.map} +1 -1
- package/dist/{run-record-CWN8-VsV.js → run-record-DqOw5X6_.js} +2 -2
- package/dist/{run-record-CWN8-VsV.js.map → run-record-DqOw5X6_.js.map} +1 -1
- package/dist/{schema-Cef2cFmb.d.ts → schema-Cef2cFmb2.d.ts} +1 -1
- package/dist/schema-Cef2cFmb2.d.ts.map +1 -0
- package/dist/{semantic-concept-judge-DwF6n05O.js → semantic-concept-judge-Bmrq6yqU.js} +2 -2
- package/dist/{semantic-concept-judge-DwF6n05O.js.map → semantic-concept-judge-Bmrq6yqU.js.map} +1 -1
- package/dist/sequential-D-BLJBKU.js +299 -0
- package/dist/sequential-D-BLJBKU.js.map +1 -0
- package/dist/{server-D6XJQHw7.js → server-iu0ede49.js} +2 -2
- package/dist/{server-D6XJQHw7.js.map → server-iu0ede49.js.map} +1 -1
- package/dist/{single-run-lock-D5iN0Xzb.js → single-run-lock-BMQEv1wG.js} +7 -161
- package/dist/single-run-lock-BMQEv1wG.js.map +1 -0
- package/dist/{skill-usage-GlOphAhX.d.ts → skill-usage-CJlWEUFt.d.ts} +10 -10
- package/dist/{skill-usage-GlOphAhX.d.ts.map → skill-usage-CJlWEUFt.d.ts.map} +1 -1
- package/dist/{skillopt-optimization-method-7S43rbDB.d.ts → skillopt-optimization-method-BO7NAl3b.d.ts} +8 -294
- package/dist/skillopt-optimization-method-BO7NAl3b.d.ts.map +1 -0
- package/dist/{skillopt-optimization-method-Bfb-vBKe.js → skillopt-optimization-method-CkaI2ly4.js} +19 -654
- package/dist/skillopt-optimization-method-CkaI2ly4.js.map +1 -0
- package/dist/statistical-heldout-Dn9ruizm.d.ts +555 -0
- package/dist/statistical-heldout-Dn9ruizm.d.ts.map +1 -0
- package/dist/{statistics-C-dm-J6H.d.ts → statistics-D6Uebe_4.d.ts} +2 -2
- package/dist/{statistics-C-dm-J6H.d.ts.map → statistics-D6Uebe_4.d.ts.map} +1 -1
- package/dist/steps-BArUxhna.d.ts +51 -0
- package/dist/steps-BArUxhna.d.ts.map +1 -0
- package/dist/{integrity-fdt8XPAv.js → store-DNe_Uv1Q.js} +2 -140
- package/dist/store-DNe_Uv1Q.js.map +1 -0
- package/dist/{summary-report-B0cAyA7N.d.ts → summary-report-DuUS_i7W.d.ts} +3 -114
- package/dist/summary-report-DuUS_i7W.d.ts.map +1 -0
- package/dist/{summary-report-9A5y7EsK.js → summary-report-Lf-5I7xh.js} +2 -2
- package/dist/{summary-report-9A5y7EsK.js.map → summary-report-Lf-5I7xh.js.map} +1 -1
- package/dist/supervisor-run/index.d.ts +2 -2
- package/dist/supervisor-run/index.js +1 -1
- package/dist/{supervisor-run-DiyQVczd.js → supervisor-run-D_sokXcO.js} +32 -8
- package/dist/{supervisor-run-DiyQVczd.js.map → supervisor-run-D_sokXcO.js.map} +1 -1
- package/dist/tool-groups-DIVBnJyl.d.ts +28 -0
- package/dist/tool-groups-DIVBnJyl.d.ts.map +1 -0
- package/dist/trace-repair/index.d.ts +2102 -0
- package/dist/trace-repair/index.d.ts.map +1 -0
- package/dist/trace-repair/index.js +3878 -0
- package/dist/trace-repair/index.js.map +1 -0
- package/dist/traces.d.ts +6 -6
- package/dist/traces.js +3 -2
- package/dist/trajectory-YC15QDYQ.d.ts +24 -0
- package/dist/trajectory-YC15QDYQ.d.ts.map +1 -0
- package/dist/trajectory-replay/index.d.ts +781 -0
- package/dist/trajectory-replay/index.d.ts.map +1 -0
- package/dist/trajectory-replay/index.js +2103 -0
- package/dist/trajectory-replay/index.js.map +1 -0
- package/dist/{types-XMVEdrE_.d.ts → types-D216SgwM.d.ts} +24 -6
- package/dist/types-D216SgwM.d.ts.map +1 -0
- package/dist/{types-yLK8gXE9.d.ts → types-D4mog56g.d.ts} +2 -2
- package/dist/{types-yLK8gXE9.d.ts.map → types-D4mog56g.d.ts.map} +1 -1
- package/dist/{types-BhP9q0Fq.d.ts → types-DF_Udrp-.d.ts} +52 -3
- package/dist/{types-BhP9q0Fq.d.ts.map → types-DF_Udrp-.d.ts.map} +1 -1
- package/dist/{types-DOZyvsFU.d.ts → types-DYuNHo9R.d.ts} +3 -3
- package/dist/{types-DOZyvsFU.d.ts.map → types-DYuNHo9R.d.ts.map} +1 -1
- package/dist/usage-receipt-EVI8B8Xu.js.map +1 -1
- package/dist/verdict-DExhxfgR.d.ts +201 -0
- package/dist/verdict-DExhxfgR.d.ts.map +1 -0
- package/dist/verdict-cache-BCcOh0kF.js +159 -0
- package/dist/verdict-cache-BCcOh0kF.js.map +1 -0
- package/dist/wire/index.d.ts +2 -2
- package/dist/wire/index.js +1 -1
- package/docs/campaign-proposers.md +5 -0
- package/docs/charter.md +112 -0
- package/docs/experiment.md +104 -0
- package/docs/prime-analyst.md +1 -0
- package/docs/trace-analysis.md +26 -0
- package/docs/trace-repair-admission.md +194 -0
- package/docs/trace-repair-analyst-arms.md +121 -0
- package/docs/trace-repair-continuation.md +107 -0
- package/docs/trace-repair-grader.md +163 -0
- package/docs/trajectory-replay.md +110 -0
- package/docs/verification-strategies.md +103 -0
- package/package.json +19 -2
- package/dist/agent-profile-cell-CbfBm2g6.js +0 -335
- package/dist/agent-profile-cell-CbfBm2g6.js.map +0 -1
- package/dist/baseline-D_fT6277.d.ts.map +0 -1
- package/dist/benchmark-J9Qe6j2_.d.ts.map +0 -1
- package/dist/benchmark-command-CQd78YHt.js.map +0 -1
- package/dist/campaign-CXsdyym7.js.map +0 -1
- package/dist/default-registry-J9m-_tya.d.ts.map +0 -1
- package/dist/index-4XwggC10.d.ts.map +0 -1
- package/dist/index-BIL5vxxt.d.ts.map +0 -1
- package/dist/integrity-fdt8XPAv.js.map +0 -1
- package/dist/llm-client-Dv5BiKLE.js.map +0 -1
- package/dist/replay-Krvb114g.d.ts.map +0 -1
- package/dist/reward-hacking-CyuzxKly.js.map +0 -1
- package/dist/reward-hacking-RZgnGWlx.d.ts.map +0 -1
- package/dist/schema-Cef2cFmb.d.ts.map +0 -1
- package/dist/single-run-lock-D5iN0Xzb.js.map +0 -1
- package/dist/skillopt-optimization-method-7S43rbDB.d.ts.map +0 -1
- package/dist/skillopt-optimization-method-Bfb-vBKe.js.map +0 -1
- package/dist/summary-report-B0cAyA7N.d.ts.map +0 -1
- package/dist/tool-groups-CK0JCkqO.d.ts.map +0 -1
- package/dist/types-XMVEdrE_.d.ts.map +0 -1
- package/dist/verdict-Dps8_okt.d.ts +0 -37
- package/dist/verdict-Dps8_okt.d.ts.map +0 -1
package/dist/index.js
CHANGED
|
@@ -1,12 +1,13 @@
|
|
|
1
1
|
import { t as __exportAll } from "./rolldown-runtime-8H4AJuhK.js";
|
|
2
|
-
import { _ as observeAll, a as jsonlReviewStore, c as scoreFromEvals, d as runAgentControlLoop, f as stopOnNoProgress, g as noProgressDetector, h as errorStreakDetector, i as inMemoryReviewStore, l as allCriticalPassed, m as subjectiveEval, n as runProposeReviewAsControlLoop, o as runProposeReview, p as stopOnRepeatedAction, r as createLlmReviewer, s as controlRunToRunRecord, t as controlFailureClassFromVerification, u as objectiveEval, v as repeatedActionDetector, y as evaluateActionPolicy } from "./propose-review-control-
|
|
2
|
+
import { _ as observeAll, a as jsonlReviewStore, c as scoreFromEvals, d as runAgentControlLoop, f as stopOnNoProgress, g as noProgressDetector, h as errorStreakDetector, i as inMemoryReviewStore, l as allCriticalPassed, m as subjectiveEval, n as runProposeReviewAsControlLoop, o as runProposeReview, p as stopOnRepeatedAction, r as createLlmReviewer, s as controlRunToRunRecord, t as controlFailureClassFromVerification, u as objectiveEval, v as repeatedActionDetector, y as evaluateActionPolicy } from "./propose-review-control-qgWLA6E9.js";
|
|
3
3
|
import { a as LimitExceededError, c as ValidationError, i as JudgeError, l as VerificationError, n as CaptureIntegrityError, o as NotFoundError, r as ConfigError, s as ReplayError, t as AgentEvalError } from "./errors-D-LKuDhb.js";
|
|
4
|
-
import {
|
|
4
|
+
import { a as verifyManifest, i as signManifest, n as evaluateHypothesis, r as hashJson, t as canonicalize } from "./pre-registration-DakwTRXk.js";
|
|
5
|
+
import { AGENT_PROFILE_KINDS, AgentProfileCellValidationError, agentProfileCellHashMaterial, agentProfileCellKey, assertRunAgentProfileCell, buildAgentInterfaceProfileCell, buildAgentProfileCell, groupRunsByAgentProfileCell, requireAgentProfileCell, toAgentProfileJson, validateAgentProfileCell, verifyAgentProfileCell } from "./profile-cell.js";
|
|
5
6
|
import { a as estimateTokens, i as estimateCost, n as MetricsCollector, o as isModelPriced, r as TokenCounter, s as resolveModelPricing, t as MODEL_PRICING } from "./metrics-C9YY1OcL.js";
|
|
6
7
|
import { a as CostLedgerPersistenceError, c as costForTokenPricing, i as CostLedger, l as costForUsage, n as CostCallConflictError, o as CostReceiptCaptureError, r as CostCeilingReachedError, s as CostReservationExceededError, t as CostAccountingIncompleteError, u as modelPriceKey } from "./cost-ledger-DMFxsLKr.js";
|
|
7
|
-
import { C as
|
|
8
|
+
import { C as servedModelAcceptable, E as judgeFamily, S as normalizeModelId, T as assertCrossFamily, _ as ServedCrossFamilyError, a as assertLlmRoute, b as assertServedModels, c as callLlmJson, d as isTransientLlmError, f as maximumChargeForLlmRequest, g as PROBE_MAX_TOKENS, h as ModelSubstitutionError, i as LlmRouteAssertionError, l as costReceiptFromLlm, m as stripFencedJson, n as LlmClient, o as backoffMs, p as probeLlm, r as LlmResponseError, s as callLlm, t as LlmCallError, u as costReceiptFromLlmError, v as assertCrossFamilyServed, w as CrossFamilyError, x as checkServedModel, y as assertServedModel } from "./llm-client-DzvMUsS_.js";
|
|
8
9
|
import { a as providerFromBaseUrl, i as defaultProviderRedactor, n as InMemoryRawProviderSink, r as NoopRawProviderSink, t as FileSystemRawProviderSink } from "./raw-provider-sink-BQd7mzyT.js";
|
|
9
|
-
import { C as createChatClient, S as computeTraceMetrics, _ as CONTROL_INTEGRITY_ANALYST, a as analystFindingDigest, c as completedAnalystReviewQuality, d as validateAnalystReviewDecisions, f as DEFAULT_TRACE_ANALYST_KINDS, g as FAILURE_MODE_KIND_SPEC, h as IMPROVEMENT_KIND_SPEC, i as assertExactRegistryRunOpts, l as readAnalystReview, m as KNOWLEDGE_GAP_KIND_SPEC, n as AnalystRegistry, o as analystRunDigest, p as KNOWLEDGE_POISONING_KIND_SPEC, r as ExactAnalystRunExecutionError, s as assertUniqueFindingIds, t as buildDefaultAnalystRegistry, u as snapshotAnalystRun, v as ControlIntegrityAnalyst, y as emitControlIntegrityFindings } from "./default-registry-
|
|
10
|
+
import { C as createChatClient, S as computeTraceMetrics, _ as CONTROL_INTEGRITY_ANALYST, a as analystFindingDigest, c as completedAnalystReviewQuality, d as validateAnalystReviewDecisions, f as DEFAULT_TRACE_ANALYST_KINDS, g as FAILURE_MODE_KIND_SPEC, h as IMPROVEMENT_KIND_SPEC, i as assertExactRegistryRunOpts, l as readAnalystReview, m as KNOWLEDGE_GAP_KIND_SPEC, n as AnalystRegistry, o as analystRunDigest, p as KNOWLEDGE_POISONING_KIND_SPEC, r as ExactAnalystRunExecutionError, s as assertUniqueFindingIds, t as buildDefaultAnalystRegistry, u as snapshotAnalystRun, v as ControlIntegrityAnalyst, y as emitControlIntegrityFindings } from "./default-registry-BaQXW1Ow.js";
|
|
10
11
|
import { INPUT_VALUE, LLM_CACHED_TOKENS, LLM_CACHED_TOKEN_ATTR_KEYS, LLM_CACHE_WRITE_TOKENS, LLM_CACHE_WRITE_TOKEN_ATTR_KEYS, LLM_CONTEXT_TOKENS, LLM_COST_ATTR_KEYS, LLM_COST_USD, LLM_INPUT_TOKENS, LLM_INPUT_TOKEN_ATTR_KEYS, LLM_MODEL_ATTR_KEYS, LLM_MODEL_NAME, LLM_OUTPUT_TOKENS, LLM_OUTPUT_TOKEN_ATTR_KEYS, LLM_REASONING_TOKENS, LLM_REASONING_TOKEN_ATTR_KEYS, OPENINFERENCE_SPAN_KIND, OUTPUT_VALUE, RUN_COST_ATTR_KEYS, SPAN_KIND_ATTR_KEYS, TOOL_ARGS_CAPTURED, TOOL_LATENCY_MS, TOOL_NAME, TOOL_NAME_ATTR_KEYS, applyLlmSpanOtlpAttributes, asNumber, contextInputTokens, firstNumberAttr } from "./trace-attributes.js";
|
|
11
12
|
import { C as TraceNotFoundError, G as resolveTraceAnalystLimits, J as extractOtlpAttributes, K as asString, Q as readOtlpStatus, S as TraceFileTooLargeError, W as DEFAULT_TRACE_ANALYST_LIMITS, X as inferOtlpKind, Y as firstStringAttr, Z as projectOtlpFlatLine, _ as TraceAnalysisLimitError, b as TraceFileMalformedError, c as traceAnalystFunctionGroup, et as stringField, g as SpanNotFoundError, h as TRACE_ANALYSIS_LIMITS, i as runTraceAnalyst, it as traceSpanKindToOpenInferenceKind, l as createBoundedTraceAnalysisStore, m as TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, n as renderPriorFindings, nt as classifyOtlpSpanRole, o as TRACE_ANALYST_TOOL_NAMESPACE, p as DEFAULT_TRACE_ANALYST_BUDGETS, r as renderUpstreamFindings, rt as isOtlpModelCall, s as buildTraceAnalysisToolDescriptors, t as createTraceAnalyst, tt as applyToolSpanOtlpAttributes, v as TraceAnalysisStoreContractError, x as TraceFileMissingError, y as TraceAnalysisValidationError } from "./kind-factory-BHIgPmzS.js";
|
|
12
13
|
import { a as computeFindingId, o as makeFinding, s as makeProposalFinding } from "./usage-receipt-EVI8B8Xu.js";
|
|
@@ -14,42 +15,46 @@ import { c as SUPERVISOR_RUN_INTEGRITY_SCHEMA, n as supervisorRunRolloutLines, t
|
|
|
14
15
|
import { c as assertMintedLines, f as isRolloutLine, i as ROLLOUT_SCHEMA, m as validateRolloutLine, p as isTrainableSplit, s as assertMinted, u as assertRolloutLine } from "./schema-C6DW4ZHR.js";
|
|
15
16
|
import { a as scoreOrigin, n as observedScore, o as trainingReward, r as observedSplitScore, s as trainingScore, t as isRealnessGated } from "./reward-nw2xZGZG.js";
|
|
16
17
|
import { a as clamp01, i as aggregateRunScore, r as DEFAULT_RUN_SCORE_WEIGHTS } from "./proposal-findings-2GIUo1et.js";
|
|
17
|
-
import { a as RunCritic, d as defaultIsMaterial, f as diffFindings, h as defineTraceAnalyst, i as runSemanticConceptJudge, n as SEMANTIC_CONCEPT_JUDGE_VERSION, o as SKILL_USAGE_ANALYST, p as LockedJsonlAppender, r as createSemanticConceptJudge, s as SkillUsageAnalyst, t as DEFAULT_COMPLEXITY_WEIGHTS, u as FindingsStore } from "./semantic-concept-judge-
|
|
18
|
-
import {
|
|
19
|
-
import { t as createDspyRlmTraceEngine } from "./dspy-rlm-engine-
|
|
18
|
+
import { a as RunCritic, d as defaultIsMaterial, f as diffFindings, h as defineTraceAnalyst, i as runSemanticConceptJudge, n as SEMANTIC_CONCEPT_JUDGE_VERSION, o as SKILL_USAGE_ANALYST, p as LockedJsonlAppender, r as createSemanticConceptJudge, s as SkillUsageAnalyst, t as DEFAULT_COMPLEXITY_WEIGHTS, u as FindingsStore } from "./semantic-concept-judge-Bmrq6yqU.js";
|
|
19
|
+
import { a as inMemoryVerdictCache, i as fileVerdictCache, n as canonicalJson, r as contentHash, t as cachedJudge } from "./verdict-cache-BCcOh0kF.js";
|
|
20
|
+
import { t as createDspyRlmTraceEngine } from "./dspy-rlm-engine-BiN49gK6.js";
|
|
20
21
|
import { f as Mutex, p as mapConcurrent } from "./ledger-core-DXZIqu17.js";
|
|
21
|
-
import {
|
|
22
|
+
import { $ as runReferenceEquivalenceJudge, L as DEFAULT_MUTATION_PRIMITIVES, M as surfaceContentHash, Q as createReferenceEquivalenceJudge, R as buildReflectionPrompt, X as REFERENCE_EQUIVALENCE_INPUT_LIMITS, Z as REFERENCE_EQUIVALENCE_JUDGE_VERSION, _t as assertRealBackend, at as Dataset, bt as JudgeParseError, ct as llmJudge, dt as paretoFrontier, et as DEFAULT_RED_TEAM_CORPUS, ft as paretoFrontierWithCrowding, gt as assertRealAgentReceipts, ht as BackendIntegrityError, it as toolNamesForRun, lt as crowdingDistance, nt as redTeamReport, ot as HoldoutLockedError, pt as scalarScore, rt as scoreRedTeamOutput, st as hashScenarios, tt as redTeamDataset, ut as dominates, vt as summarizeAgentReceiptIntegrity, y as runCanaries, yt as summarizeBackendIntegrity, z as parseReflectionResponse } from "./skillopt-optimization-method-CkaI2ly4.js";
|
|
22
23
|
import { $ as selfPreference, A as pairedRiskDifference, B as requiredSampleSize, C as mulberry32, D as pairedCohensDz, E as pairedBootstrap, F as partialCredit, G as wilson, H as weightedComposite, I as passAtK, J as normalCdf, K as studentTCdf, L as pearsonR, M as pairedRiskDifferenceScore, N as pairedSignTest, O as pairedDeltaTieFraction, P as pairedTTest, Q as positionalBias, R as ranks, S as mcnemarRequiredN, T as pairedBinaryScale, U as weightedMean, V as spearmanR, W as wilcoxonSignedRank, X as calibrateJudgeContinuous, Y as calibrateJudge, Z as continuousAgreement, _ as interpretCliffs, a as MANN_WHITNEY_EXACT_MAX_WORK, b as mcnemar, c as bonferroni, d as confidenceInterval, et as verbosityBias, f as corpusInterRaterAgreement, g as interRaterReliability, h as holm, i as MANN_WHITNEY_EXACT_MAX_STATES, j as pairedRiskDifferenceExact, k as pairedMde, l as cliffsDelta, m as eProcess, n as DECISION_PAIRED_DELTA_STATISTIC, o as WILCOXON_EXACT_MAX_N, p as corpusInterRaterAgreementFromJudgeScores, r as DEFAULT_PERMUTATIONS, s as benjaminiHochberg, t as BOOTSTRAP_GATE_MIN_N, u as cohensD, v as isBinaryOutcomeVector, w as normalizeScores, x as mcnemarPower, y as mannWhitneyU, z as requiredPairedSampleSize } from "./statistics-ByxzSiOM.js";
|
|
23
24
|
import { n as pairArms, r as pairRunRecords, t as comparePairedArms } from "./paired-arms-iZ08VFMN.js";
|
|
25
|
+
import { a as improvementVerdict, i as gitProvenanceReader, n as computeExperimentStats, o as inMemoryExperimentStore, r as fileExperimentStore, s as clusteredPairedBinary, t as ExperimentTracker } from "./experiment-tracker-CnRICnMl.js";
|
|
24
26
|
import { n as llmSpanFromProvider, t as TraceEmitter } from "./emitter-CPBAhxum.js";
|
|
25
27
|
import { a as isRetrievalSpan, i as isLlmSpan, n as TRACE_SCHEMA_VERSION, o as isSandboxSpan, r as isJudgeSpan, s as isToolSpan, t as FAILURE_CLASSES } from "./schema-CRhEY1SO.js";
|
|
26
|
-
import { a as roundTripRunRecord, i as parseRunRecordSafe, n as isRunRecord, o as runTaskScore, r as modelHasSnapshot, s as validateRunRecord, t as RunRecordValidationError } from "./run-record-
|
|
27
|
-
import { a as evaluateReleaseConfidence, i as assertReleaseConfidence, n as bootstrapCi, r as judgeReplayGate, t as renderReleaseReport } from "./release-report-
|
|
28
|
+
import { a as roundTripRunRecord, i as parseRunRecordSafe, n as isRunRecord, o as runTaskScore, r as modelHasSnapshot, s as validateRunRecord, t as RunRecordValidationError } from "./run-record-DqOw5X6_.js";
|
|
29
|
+
import { a as evaluateReleaseConfidence, i as assertReleaseConfidence, n as bootstrapCi, r as judgeReplayGate, t as renderReleaseReport } from "./release-report-Dy39cbFF.js";
|
|
30
|
+
import { d as pairedDecisionShape, f as minimumPairsForPairedDeltaTest, p as pairedDeltaTest, u as decidePairedPromotion } from "./promotion-policy-CrLrmys8.js";
|
|
28
31
|
import { i as toRewardRows, r as toJsonl, s as toSftRows } from "./exporters-q9iL-2Jf.js";
|
|
29
|
-
import { a as toHarborTrajectories, i as relabelImportedSplit, n as HARBOR_IMPORT_GAP, o as toHarborTrajectory, r as fromHarborTrajectory, t as ATIF_SCHEMA_VERSION } from "./rollout-
|
|
32
|
+
import { a as toHarborTrajectories, i as relabelImportedSplit, n as HARBOR_IMPORT_GAP, o as toHarborTrajectory, r as fromHarborTrajectory, t as ATIF_SCHEMA_VERSION } from "./rollout-2ECTXb2N.js";
|
|
30
33
|
import { t as buildTrajectory } from "./trajectory-D_7rLrvE.js";
|
|
31
|
-
import { n as unmintableReasons, t as mintRolloutRows } from "./mint-
|
|
32
|
-
import { C as rollupSupervisorRuns, D as isUnavailable, E as SUPERVISOR_RUN_SCHEMA, O as showMeasured, b as readClaudeCodeSupervisorRun, c as writeSupervisorRunReport, d as readRuntimeSupervisorRun, f as runtimeSupervisorRunReader, h as renderSupervisorRunMarkdown, m as renderSupervisorRunHeadline, t as analyzeSupervisorRun, u as isRuntimeSupervisorRunDir, x as analyzeSupervisorRunSources, y as claudeCodeSupervisorRunReader } from "./supervisor-run-
|
|
33
|
-
import { A as TRACE_ANALYST_ACTOR_DESCRIPTION, C as domainEvidencePattern, D as tokenizeDomainWords, E as scoreTraceInsightReadiness, O as traceAnalystOnRunComplete, S as describeTraceInsightScope, T as planTraceInsightQuestions, _ as otlpToTraceRunRecords, a as convertTraceStoresToOtlp, b as buildTraceInsightPrompt, c as otelRunCompleteHook, d as captureFetchToRawSink, f as ToolTraceMissingError, g as otlpToRunRecords, h as otlpRowsToTraceRunRecords, i as iterateRawCalls, j as TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION, k as analyzeTraces, l as OTEL_AGENT_EVAL_SCOPE, m as otlpRowsToRunRecords, n as ReplayCacheMissError, o as createOtelExporter, p as toolSpansToTraceAnalysisStore, r as createReplayFetch, s as createOtelTracingStore, t as ReplayCache, u as exportRunAsOtlp, v as flattenOtlpExportToNdjson, w as inferDomainKeywords, x as defaultTraceInsightPanel, y as buildTraceInsightContext } from "./replay-
|
|
34
|
+
import { n as unmintableReasons, t as mintRolloutRows } from "./mint-CGEkzPLf.js";
|
|
35
|
+
import { C as rollupSupervisorRuns, D as isUnavailable, E as SUPERVISOR_RUN_SCHEMA, O as showMeasured, b as readClaudeCodeSupervisorRun, c as writeSupervisorRunReport, d as readRuntimeSupervisorRun, f as runtimeSupervisorRunReader, h as renderSupervisorRunMarkdown, m as renderSupervisorRunHeadline, t as analyzeSupervisorRun, u as isRuntimeSupervisorRunDir, x as analyzeSupervisorRunSources, y as claudeCodeSupervisorRunReader } from "./supervisor-run-D_sokXcO.js";
|
|
36
|
+
import { A as TRACE_ANALYST_ACTOR_DESCRIPTION, C as domainEvidencePattern, D as tokenizeDomainWords, E as scoreTraceInsightReadiness, O as traceAnalystOnRunComplete, S as describeTraceInsightScope, T as planTraceInsightQuestions, _ as otlpToTraceRunRecords, a as convertTraceStoresToOtlp, b as buildTraceInsightPrompt, c as otelRunCompleteHook, d as captureFetchToRawSink, f as ToolTraceMissingError, g as otlpToRunRecords, h as otlpRowsToTraceRunRecords, i as iterateRawCalls, j as TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION, k as analyzeTraces, l as OTEL_AGENT_EVAL_SCOPE, m as otlpRowsToRunRecords, n as ReplayCacheMissError, o as createOtelExporter, p as toolSpansToTraceAnalysisStore, r as createReplayFetch, s as createOtelTracingStore, t as ReplayCache, u as exportRunAsOtlp, v as flattenOtlpExportToNdjson, w as inferDomainKeywords, x as defaultTraceInsightPanel, y as buildTraceInsightContext } from "./replay-DyBLaKFc.js";
|
|
34
37
|
import { i as otlpTextToTraceAnalysisStore, n as OtlpFileTraceStore } from "./store-otlp-CKtTpRhv.js";
|
|
35
38
|
import { n as extractUsageFromResponse, r as extractUsageFromSse, t as extractUsage } from "./extract-usage-BW27f3XW.js";
|
|
36
|
-
import { B as
|
|
39
|
+
import { B as harnessAxisOf, F as HARNESS_NATIVE_MODEL, G as parseCorrectnessResponse, H as completionVerdict, I as agentProfileHash, K as verifyCompletion, L as agentProfileId, P as CODING_HARNESSES, R as agentProfileModelId, U as createLlmCorrectnessChecker, V as extractProducedState, W as createTokenRecallChecker, z as expandProfileAxes } from "./campaign--HVSuvV0.js";
|
|
37
40
|
import { n as iqr, r as welchsTTest, t as compareToBaseline } from "./baseline-C-GocmIW.js";
|
|
38
41
|
import { a as judgeSpans, c as runsForScenario, i as hasCapturedToolArgs, l as toolSpans, n as argHash, o as llmSpans, r as groupBy, s as runFailureClass, t as aggregateLlm } from "./query-Di7eEQ79.js";
|
|
39
|
-
import { a as checkBehavioralCanary, i as canaryLeakView, o as checkCanaries, r as HoldoutAuditor, s as runBehavioralCanaries, t as analyzeRuns } from "./analyze-runs-
|
|
40
|
-
import { a as summaryTable, i as researchReport, n as gainHistogram, r as paretoChart, t as RESEARCH_REPORT_HARD_PAIR_FLOOR } from "./summary-report-
|
|
42
|
+
import { a as checkBehavioralCanary, i as canaryLeakView, o as checkCanaries, r as HoldoutAuditor, s as runBehavioralCanaries, t as analyzeRuns } from "./analyze-runs-BNNK7irB.js";
|
|
43
|
+
import { a as summaryTable, i as researchReport, n as gainHistogram, r as paretoChart, t as RESEARCH_REPORT_HARD_PAIR_FLOOR } from "./summary-report-Lf-5I7xh.js";
|
|
41
44
|
import { a as composeParsers, c as vitestTestParser, i as SubprocessSandboxDriver, n as DockerSandboxDriver, o as jestTestParser, r as SandboxHarness, s as pytestTestParser, t as runTestGradedScenario } from "./test-graded-scenario-JHcKQNpq.js";
|
|
42
|
-
import {
|
|
45
|
+
import { n as InMemoryTraceStore, t as FileSystemTraceStore } from "./store-DNe_Uv1Q.js";
|
|
46
|
+
import { n as assertRunCaptured, r as throwIfRunIncomplete, t as RunIntegrityError } from "./integrity-MLzHOfV9.js";
|
|
43
47
|
import { i as redactValue, n as REDACTION_VERSION, r as redactString, t as DEFAULT_REDACTION_RULES } from "./redact-7Aq1ukl-.js";
|
|
44
48
|
import { n as DEFAULT_RULES, r as classifyFailure, t as computeToolUseMetrics } from "./tool-use-metrics-DEGMKycK.js";
|
|
45
49
|
import { t as analyzeSeries } from "./series-convergence-CjO2QdRW.js";
|
|
46
|
-
import {
|
|
47
|
-
import { t as
|
|
50
|
+
import { n as runCounterfactual, t as attributeCounterfactuals } from "./counterfactual-CWPTrMH7.js";
|
|
51
|
+
import { _ as deterministicSplit, g as BENCHMARK_SPLIT_SEED, t as benchmarks_exports } from "./benchmarks-BlPmjd88.js";
|
|
52
|
+
import { t as runEvalCampaign } from "./eval-campaign-DNjCvAm-.js";
|
|
48
53
|
import { n as pairedEvalueSequence, t as evaluateInterimReleaseConfidence } from "./sequential-Br0mAPHA.js";
|
|
49
54
|
import { accessSync, appendFileSync, constants, cpSync, existsSync, mkdirSync, promises, readFileSync, readdirSync, statSync, writeFileSync } from "node:fs";
|
|
50
55
|
import { basename, delimiter, dirname, extname, isAbsolute, join, relative, resolve } from "node:path";
|
|
51
56
|
import { createHash } from "node:crypto";
|
|
52
|
-
import {
|
|
57
|
+
import { spawnSync } from "node:child_process";
|
|
53
58
|
import { readFile } from "node:fs/promises";
|
|
54
59
|
import { cpus } from "node:os";
|
|
55
60
|
import { gzipSync } from "node:zlib";
|
|
@@ -423,6 +428,10 @@ async function executeScenario(chat, scenario, config) {
|
|
|
423
428
|
receiptFromError: costReceiptFromLlmError
|
|
424
429
|
});
|
|
425
430
|
if (!paid.succeeded) throw paid.error;
|
|
431
|
+
assertServedModel(model, paid.value.servedModel, {
|
|
432
|
+
allowUnreported: true,
|
|
433
|
+
context: `executeScenario "${scenario.id}" turn ${i}`
|
|
434
|
+
});
|
|
426
435
|
const rawContent = paid.value.content;
|
|
427
436
|
if (typeof rawContent !== "string") throw new CaptureIntegrityError(`chat response for scenario "${scenario.id}" turn ${i} is malformed: expected content to be a string, got ${rawContent === null ? "null" : typeof rawContent}`);
|
|
428
437
|
const content = rawContent;
|
|
@@ -1100,235 +1109,6 @@ async function runE2EWorkflow(client, name, workflow) {
|
|
|
1100
1109
|
};
|
|
1101
1110
|
}
|
|
1102
1111
|
//#endregion
|
|
1103
|
-
//#region src/clustered-paired-binary.ts
|
|
1104
|
-
/**
|
|
1105
|
-
* Paired binary comparison for work items nested inside independent clusters.
|
|
1106
|
-
*
|
|
1107
|
-
* Pairing is delegated to {@link pairArms}; this module adds the cluster-aware
|
|
1108
|
-
* estimands and inference that task-level McNemar/bootstrap utilities cannot
|
|
1109
|
-
* provide. Callers keep their own row shape through accessors, and every
|
|
1110
|
-
* matched or unpaired result returns the original row object unchanged.
|
|
1111
|
-
*/
|
|
1112
|
-
const DEFAULT_BOOTSTRAP_RESAMPLES = 1e4;
|
|
1113
|
-
const DEFAULT_SIGN_FLIP_RESAMPLES = 1e5;
|
|
1114
|
-
const MAX_RESAMPLES = 1e6;
|
|
1115
|
-
const DEFAULT_EXACT_CLUSTER_LIMIT = 20;
|
|
1116
|
-
const SIGN_FLIP_SEED_SALT = 2654435769;
|
|
1117
|
-
/**
|
|
1118
|
-
* Compare binary outcomes on matched work items while respecting independent
|
|
1119
|
-
* clusters. The confidence interval resamples whole clusters and recomputes the
|
|
1120
|
-
* task-weighted risk difference. The sign-flip test flips whole-cluster outcome
|
|
1121
|
-
* totals and tests that same task-weighted estimand.
|
|
1122
|
-
*/
|
|
1123
|
-
function clusteredPairedBinary(rows, options) {
|
|
1124
|
-
const config = validateOptions(options);
|
|
1125
|
-
const paired = pairArms(projectSelectedRows(rows, options), {
|
|
1126
|
-
baselineArm: options.baselineArm,
|
|
1127
|
-
treatmentArm: options.treatmentArm
|
|
1128
|
-
});
|
|
1129
|
-
const matchedPairs = paired.pairs.map((pair) => {
|
|
1130
|
-
const baseline = pair.baseline;
|
|
1131
|
-
const treatment = pair.treatment;
|
|
1132
|
-
if (baseline.clusterKey !== treatment.clusterKey) throw new ValidationError(`clusteredPairedBinary: pairKey '${pair.pairKey}' rep ${pair.repIndex} crosses clusters ('${baseline.clusterKey}' vs '${treatment.clusterKey}')`);
|
|
1133
|
-
return {
|
|
1134
|
-
pairKey: pair.pairKey,
|
|
1135
|
-
repIndex: pair.repIndex,
|
|
1136
|
-
clusterKey: baseline.clusterKey,
|
|
1137
|
-
baseline: baseline.original,
|
|
1138
|
-
treatment: treatment.original,
|
|
1139
|
-
baselinePass: baseline.pass,
|
|
1140
|
-
treatmentPass: treatment.pass
|
|
1141
|
-
};
|
|
1142
|
-
});
|
|
1143
|
-
const unpairedBaseline = paired.unpairedBaseline.map((row) => row.original);
|
|
1144
|
-
const unpairedTreatment = paired.unpairedTreatment.map((row) => row.original);
|
|
1145
|
-
if (matchedPairs.length === 0) return {
|
|
1146
|
-
matchedPairs,
|
|
1147
|
-
unpairedBaseline,
|
|
1148
|
-
unpairedTreatment,
|
|
1149
|
-
statistics: null
|
|
1150
|
-
};
|
|
1151
|
-
const clusters = summarizeClusters(matchedPairs);
|
|
1152
|
-
const b10 = clusters.reduce((sum, cluster) => sum + cluster.b10, 0);
|
|
1153
|
-
const b01 = clusters.reduce((sum, cluster) => sum + cluster.b01, 0);
|
|
1154
|
-
const taskWeightedRiskDifference = (b10 - b01) / matchedPairs.length;
|
|
1155
|
-
const equalClusterMean = mean$4(clusters.map((cluster) => cluster.meanDifference));
|
|
1156
|
-
const bootstrap = clusters.length < 2 ? null : clusterBootstrap(clusters, config);
|
|
1157
|
-
const signFlip = clusterSignFlip(clusters, config);
|
|
1158
|
-
return {
|
|
1159
|
-
matchedPairs,
|
|
1160
|
-
unpairedBaseline,
|
|
1161
|
-
unpairedTreatment,
|
|
1162
|
-
statistics: {
|
|
1163
|
-
nPairs: matchedPairs.length,
|
|
1164
|
-
nClusters: clusters.length,
|
|
1165
|
-
b10,
|
|
1166
|
-
b01,
|
|
1167
|
-
taskWeightedRiskDifference,
|
|
1168
|
-
equalClusterMean,
|
|
1169
|
-
clusters,
|
|
1170
|
-
bootstrap,
|
|
1171
|
-
signFlip
|
|
1172
|
-
}
|
|
1173
|
-
};
|
|
1174
|
-
}
|
|
1175
|
-
function validateOptions(options) {
|
|
1176
|
-
assertNonEmptyString("baselineArm", options.baselineArm);
|
|
1177
|
-
assertNonEmptyString("treatmentArm", options.treatmentArm);
|
|
1178
|
-
if (options.baselineArm === options.treatmentArm) throw new ValidationError(`clusteredPairedBinary: baselineArm and treatmentArm are both '${options.baselineArm}'`);
|
|
1179
|
-
if (!Number.isInteger(options.seed)) throw new ValidationError(`clusteredPairedBinary: seed must be an integer, got ${options.seed}`);
|
|
1180
|
-
const confidence = options.confidence ?? .95;
|
|
1181
|
-
if (!Number.isFinite(confidence) || confidence <= 0 || confidence >= 1) throw new ValidationError(`clusteredPairedBinary: confidence must be in (0,1), got ${confidence}`);
|
|
1182
|
-
const bootstrapResamples = options.bootstrapResamples ?? DEFAULT_BOOTSTRAP_RESAMPLES;
|
|
1183
|
-
assertResampleCount("bootstrapResamples", bootstrapResamples);
|
|
1184
|
-
const rawMinimumBootstrapResamples = 2 / (1 - confidence);
|
|
1185
|
-
const minimumBootstrapResamples = Math.ceil(rawMinimumBootstrapResamples - Number.EPSILON * Math.max(1, rawMinimumBootstrapResamples) * 8);
|
|
1186
|
-
if (bootstrapResamples < minimumBootstrapResamples) throw new ValidationError(`clusteredPairedBinary: bootstrapResamples must be at least ${minimumBootstrapResamples} for confidence ${confidence} so both interval tails are represented, got ${bootstrapResamples}`);
|
|
1187
|
-
const signFlipResamples = options.signFlipResamples ?? DEFAULT_SIGN_FLIP_RESAMPLES;
|
|
1188
|
-
assertResampleCount("signFlipResamples", signFlipResamples);
|
|
1189
|
-
const exactClusterLimit = options.exactClusterLimit ?? DEFAULT_EXACT_CLUSTER_LIMIT;
|
|
1190
|
-
if (!Number.isInteger(exactClusterLimit) || exactClusterLimit < 0 || exactClusterLimit > DEFAULT_EXACT_CLUSTER_LIMIT) throw new ValidationError(`clusteredPairedBinary: exactClusterLimit must be an integer in [0,${DEFAULT_EXACT_CLUSTER_LIMIT}], got ${exactClusterLimit}`);
|
|
1191
|
-
const alternative = options.alternative ?? "two-sided";
|
|
1192
|
-
if (alternative !== "two-sided" && alternative !== "greater" && alternative !== "less") throw new ValidationError(`clusteredPairedBinary: alternative must be 'two-sided', 'greater', or 'less', got ${String(alternative)}`);
|
|
1193
|
-
return {
|
|
1194
|
-
seed: options.seed,
|
|
1195
|
-
confidence,
|
|
1196
|
-
bootstrapResamples,
|
|
1197
|
-
alternative,
|
|
1198
|
-
exactClusterLimit,
|
|
1199
|
-
signFlipResamples
|
|
1200
|
-
};
|
|
1201
|
-
}
|
|
1202
|
-
function assertResampleCount(name, value) {
|
|
1203
|
-
if (!Number.isInteger(value) || value <= 0) throw new ValidationError(`clusteredPairedBinary: ${name} must be a positive integer, got ${value}`);
|
|
1204
|
-
if (value > MAX_RESAMPLES) throw new ValidationError(`clusteredPairedBinary: ${name} must not exceed ${MAX_RESAMPLES}, got ${value}`);
|
|
1205
|
-
}
|
|
1206
|
-
function projectSelectedRows(rows, options) {
|
|
1207
|
-
const projected = [];
|
|
1208
|
-
for (const original of rows) {
|
|
1209
|
-
const arm = options.arm(original);
|
|
1210
|
-
assertNonEmptyString("arm", arm);
|
|
1211
|
-
if (arm !== options.baselineArm && arm !== options.treatmentArm) continue;
|
|
1212
|
-
const pairKey = options.pairKey(original);
|
|
1213
|
-
const clusterKey = options.clusterKey(original);
|
|
1214
|
-
const pass = options.pass(original);
|
|
1215
|
-
const repKey = options.repKey?.(original);
|
|
1216
|
-
assertNonEmptyString("pairKey", pairKey);
|
|
1217
|
-
assertNonEmptyString("clusterKey", clusterKey);
|
|
1218
|
-
if (typeof pass !== "boolean") throw new ValidationError(`clusteredPairedBinary: pass accessor must return boolean for pairKey '${pairKey}'`);
|
|
1219
|
-
if (repKey !== void 0) assertNonEmptyString("repKey", repKey);
|
|
1220
|
-
projected.push({
|
|
1221
|
-
pairKey,
|
|
1222
|
-
clusterKey,
|
|
1223
|
-
arm,
|
|
1224
|
-
pass,
|
|
1225
|
-
repKey,
|
|
1226
|
-
original
|
|
1227
|
-
});
|
|
1228
|
-
}
|
|
1229
|
-
return projected;
|
|
1230
|
-
}
|
|
1231
|
-
function assertNonEmptyString(name, value) {
|
|
1232
|
-
if (typeof value !== "string" || value.trim().length === 0) throw new ValidationError(`clusteredPairedBinary: ${name} accessor must return a non-empty string`);
|
|
1233
|
-
}
|
|
1234
|
-
function summarizeClusters(pairs) {
|
|
1235
|
-
const byCluster = /* @__PURE__ */ new Map();
|
|
1236
|
-
for (const pair of pairs) {
|
|
1237
|
-
const summary = byCluster.get(pair.clusterKey) ?? {
|
|
1238
|
-
nPairs: 0,
|
|
1239
|
-
b10: 0,
|
|
1240
|
-
b01: 0
|
|
1241
|
-
};
|
|
1242
|
-
summary.nPairs++;
|
|
1243
|
-
if (pair.treatmentPass && !pair.baselinePass) summary.b10++;
|
|
1244
|
-
else if (pair.baselinePass && !pair.treatmentPass) summary.b01++;
|
|
1245
|
-
byCluster.set(pair.clusterKey, summary);
|
|
1246
|
-
}
|
|
1247
|
-
return [...byCluster.entries()].sort(([a], [b]) => a < b ? -1 : a > b ? 1 : 0).map(([clusterKey, summary]) => ({
|
|
1248
|
-
clusterKey,
|
|
1249
|
-
...summary,
|
|
1250
|
-
meanDifference: (summary.b10 - summary.b01) / summary.nPairs
|
|
1251
|
-
}));
|
|
1252
|
-
}
|
|
1253
|
-
function clusterBootstrap(clusters, config) {
|
|
1254
|
-
const rng = mulberry32(config.seed);
|
|
1255
|
-
const samples = new Array(config.bootstrapResamples);
|
|
1256
|
-
for (let draw = 0; draw < config.bootstrapResamples; draw++) {
|
|
1257
|
-
let differenceSum = 0;
|
|
1258
|
-
let pairCount = 0;
|
|
1259
|
-
for (let index = 0; index < clusters.length; index++) {
|
|
1260
|
-
const cluster = clusters[Math.floor(rng() * clusters.length)];
|
|
1261
|
-
differenceSum += cluster.b10 - cluster.b01;
|
|
1262
|
-
pairCount += cluster.nPairs;
|
|
1263
|
-
}
|
|
1264
|
-
samples[draw] = differenceSum / pairCount;
|
|
1265
|
-
}
|
|
1266
|
-
samples.sort((a, b) => a - b);
|
|
1267
|
-
const alpha = 1 - config.confidence;
|
|
1268
|
-
const lowerIndex = Math.floor(alpha / 2 * config.bootstrapResamples);
|
|
1269
|
-
const upperIndex = Math.min(config.bootstrapResamples - 1, Math.ceil((1 - alpha / 2) * config.bootstrapResamples) - 1);
|
|
1270
|
-
return {
|
|
1271
|
-
statistic: "task-weighted-risk-difference",
|
|
1272
|
-
lower: samples[lowerIndex],
|
|
1273
|
-
upper: samples[Math.max(lowerIndex, upperIndex)],
|
|
1274
|
-
confidence: config.confidence,
|
|
1275
|
-
resamples: config.bootstrapResamples,
|
|
1276
|
-
seed: config.seed
|
|
1277
|
-
};
|
|
1278
|
-
}
|
|
1279
|
-
function clusterSignFlip(clusters, config) {
|
|
1280
|
-
const clusterTotals = clusters.map((cluster) => cluster.b10 - cluster.b01);
|
|
1281
|
-
const nonZero = clusterTotals.filter((delta) => delta !== 0);
|
|
1282
|
-
const totalPairs = clusters.reduce((sum, cluster) => sum + cluster.nPairs, 0);
|
|
1283
|
-
const statistic = clusterTotals.reduce((sum, delta) => sum + delta, 0) / totalPairs;
|
|
1284
|
-
if (nonZero.length <= config.exactClusterLimit) {
|
|
1285
|
-
const assignments = 2 ** nonZero.length;
|
|
1286
|
-
let extreme = 0;
|
|
1287
|
-
for (let mask = 0; mask < assignments; mask++) {
|
|
1288
|
-
let sum = 0;
|
|
1289
|
-
for (let index = 0; index < nonZero.length; index++) sum += (mask & 2 ** index ? 1 : -1) * nonZero[index];
|
|
1290
|
-
if (isExtreme(sum / totalPairs, statistic, config.alternative)) extreme++;
|
|
1291
|
-
}
|
|
1292
|
-
return {
|
|
1293
|
-
statistic,
|
|
1294
|
-
pValue: extreme / assignments,
|
|
1295
|
-
alternative: config.alternative,
|
|
1296
|
-
method: "exact",
|
|
1297
|
-
assignments,
|
|
1298
|
-
nClusters: clusters.length,
|
|
1299
|
-
nNonZeroClusters: nonZero.length,
|
|
1300
|
-
seed: null
|
|
1301
|
-
};
|
|
1302
|
-
}
|
|
1303
|
-
const signFlipSeed = (config.seed ^ SIGN_FLIP_SEED_SALT) >>> 0;
|
|
1304
|
-
const rng = mulberry32(signFlipSeed);
|
|
1305
|
-
let extreme = 0;
|
|
1306
|
-
for (let draw = 0; draw < config.signFlipResamples; draw++) {
|
|
1307
|
-
let sum = 0;
|
|
1308
|
-
for (const delta of nonZero) sum += (rng() < .5 ? -1 : 1) * delta;
|
|
1309
|
-
if (isExtreme(sum / totalPairs, statistic, config.alternative)) extreme++;
|
|
1310
|
-
}
|
|
1311
|
-
return {
|
|
1312
|
-
statistic,
|
|
1313
|
-
pValue: (extreme + 1) / (config.signFlipResamples + 1),
|
|
1314
|
-
alternative: config.alternative,
|
|
1315
|
-
method: "monte-carlo",
|
|
1316
|
-
assignments: config.signFlipResamples,
|
|
1317
|
-
nClusters: clusters.length,
|
|
1318
|
-
nNonZeroClusters: nonZero.length,
|
|
1319
|
-
seed: signFlipSeed
|
|
1320
|
-
};
|
|
1321
|
-
}
|
|
1322
|
-
function isExtreme(candidate, observed, alternative) {
|
|
1323
|
-
const tolerance = Number.EPSILON * Math.max(1, Math.abs(observed)) * 16;
|
|
1324
|
-
if (alternative === "greater") return candidate >= observed - tolerance;
|
|
1325
|
-
if (alternative === "less") return candidate <= observed + tolerance;
|
|
1326
|
-
return Math.abs(candidate) >= Math.abs(observed) - tolerance;
|
|
1327
|
-
}
|
|
1328
|
-
function mean$4(values) {
|
|
1329
|
-
return values.reduce((sum, value) => sum + value, 0) / values.length;
|
|
1330
|
-
}
|
|
1331
|
-
//#endregion
|
|
1332
1112
|
//#region src/convergence.ts
|
|
1333
1113
|
/**
|
|
1334
1114
|
* ConvergenceTracker — tracks completion percentage over turns.
|
|
@@ -1606,6 +1386,10 @@ async function decideNextUserTurn(chat, opts) {
|
|
|
1606
1386
|
receiptFromError: costReceiptFromLlmError
|
|
1607
1387
|
});
|
|
1608
1388
|
if (!paid.succeeded) throw paid.error;
|
|
1389
|
+
assertServedModel(model, paid.value.servedModel, {
|
|
1390
|
+
allowUnreported: true,
|
|
1391
|
+
context: "decideNextUserTurn"
|
|
1392
|
+
});
|
|
1609
1393
|
return paid.value.content.trim();
|
|
1610
1394
|
}
|
|
1611
1395
|
//#endregion
|
|
@@ -2132,9 +1916,10 @@ function canonicalize$1(value) {
|
|
|
2132
1916
|
* - membership (free): GET `{baseUrl}/models` once; a model is `listed` when
|
|
2133
1917
|
* its id is in the served set.
|
|
2134
1918
|
* - probe (spends a tiny number of tokens): POST `{baseUrl}/chat/completions`
|
|
2135
|
-
* per model with a 1-message,
|
|
2136
|
-
* router
|
|
2137
|
-
* captured in `detail`, and `servedModel` recording
|
|
1919
|
+
* per model with a 1-message, `PROBE_MAX_TOKENS`-token request; `served` is
|
|
1920
|
+
* whether the router reached a provider, with the HTTP `status` and the
|
|
1921
|
+
* body's `error.message` captured in `detail`, and `servedModel` recording
|
|
1922
|
+
* WHICH model answered.
|
|
2138
1923
|
*
|
|
2139
1924
|
* A 2xx is not proof the requested model answered — a gateway can accept one
|
|
2140
1925
|
* id and route to another. The probe therefore compares the echoed id against
|
|
@@ -2147,6 +1932,12 @@ function canonicalize$1(value) {
|
|
|
2147
1932
|
* `assertModelsServed` and it surfaces every dead or substituted id with its
|
|
2148
1933
|
* status + detail instead of silently producing a stub or mislabelled run.
|
|
2149
1934
|
*/
|
|
1935
|
+
/**
|
|
1936
|
+
* Provider signature for "the budget ran out before an answer". The model is
|
|
1937
|
+
* alive — a provider took the request and consumed the budget — so this must
|
|
1938
|
+
* never be scored as a dead id.
|
|
1939
|
+
*/
|
|
1940
|
+
const REASONING_BUDGET_EXHAUSTED = /reasoning[\s_-]?budget[\s_-]?exhausted/i;
|
|
2150
1941
|
function stripSlash$1(url) {
|
|
2151
1942
|
return url.replace(/\/+$/, "");
|
|
2152
1943
|
}
|
|
@@ -2165,13 +1956,19 @@ function errorMessage(body) {
|
|
|
2165
1956
|
* fallbacks.
|
|
2166
1957
|
*
|
|
2167
1958
|
* The membership check (one GET) always runs. When `probe` is true, each model
|
|
2168
|
-
* additionally gets a
|
|
1959
|
+
* additionally gets a small chat probe so a model that is listed but
|
|
2169
1960
|
* unconfigured (a 401 `model_not_found` from the router) is caught.
|
|
2170
1961
|
*/
|
|
2171
1962
|
async function preflightModels(opts) {
|
|
2172
1963
|
const fetchImpl = opts.fetchImpl ?? fetch;
|
|
2173
1964
|
const baseUrl = stripSlash$1(opts.baseUrl);
|
|
2174
1965
|
const authHeaders = { authorization: `Bearer ${opts.apiKey}` };
|
|
1966
|
+
const maxTokens = opts.probeMaxTokens ?? 64;
|
|
1967
|
+
if (!Number.isInteger(maxTokens) || maxTokens <= 0) return {
|
|
1968
|
+
succeeded: false,
|
|
1969
|
+
value: null,
|
|
1970
|
+
error: `preflightModels: probeMaxTokens must be a positive integer, got ${maxTokens}`
|
|
1971
|
+
};
|
|
2175
1972
|
let served;
|
|
2176
1973
|
try {
|
|
2177
1974
|
const res = await fetchImpl(`${baseUrl}/models`, {
|
|
@@ -2206,6 +2003,7 @@ async function preflightModels(opts) {
|
|
|
2206
2003
|
served: null,
|
|
2207
2004
|
status: null,
|
|
2208
2005
|
detail: null,
|
|
2006
|
+
budgetExhausted: false,
|
|
2209
2007
|
substitution: null
|
|
2210
2008
|
});
|
|
2211
2009
|
continue;
|
|
@@ -2223,22 +2021,28 @@ async function preflightModels(opts) {
|
|
|
2223
2021
|
role: "user",
|
|
2224
2022
|
content: "ping"
|
|
2225
2023
|
}],
|
|
2226
|
-
max_tokens:
|
|
2024
|
+
max_tokens: maxTokens
|
|
2227
2025
|
})
|
|
2228
2026
|
});
|
|
2229
2027
|
let detail = null;
|
|
2230
2028
|
let substitution = null;
|
|
2029
|
+
let budgetExhausted = false;
|
|
2231
2030
|
const body = await res.json().catch(() => null);
|
|
2232
2031
|
if (res.ok) {
|
|
2233
2032
|
const echoed = body?.model;
|
|
2234
2033
|
substitution = checkServedModel(model, typeof echoed === "string" && echoed.trim() !== "" ? echoed : null);
|
|
2235
|
-
} else
|
|
2034
|
+
} else {
|
|
2035
|
+
detail = errorMessage(body);
|
|
2036
|
+
budgetExhausted = detail !== null && REASONING_BUDGET_EXHAUSTED.test(detail);
|
|
2037
|
+
if (budgetExhausted) substitution = checkServedModel(model, null);
|
|
2038
|
+
}
|
|
2236
2039
|
results.push({
|
|
2237
2040
|
model,
|
|
2238
2041
|
listed,
|
|
2239
|
-
served: res.ok,
|
|
2042
|
+
served: res.ok || budgetExhausted,
|
|
2240
2043
|
status: res.status,
|
|
2241
2044
|
detail,
|
|
2045
|
+
budgetExhausted,
|
|
2242
2046
|
substitution
|
|
2243
2047
|
});
|
|
2244
2048
|
} catch (err) {
|
|
@@ -2269,6 +2073,7 @@ function describeFailure(r) {
|
|
|
2269
2073
|
return `${r.model}: not in /models${probeNote}`;
|
|
2270
2074
|
}
|
|
2271
2075
|
if (r.served === false) return `${r.model}: listed but probe ${r.status}${r.detail ? ` — ${r.detail}` : ""}`;
|
|
2076
|
+
if (r.budgetExhausted) return `${r.model}: alive but the probe ran out of reasoning budget (status ${r.status}${r.detail ? `: ${r.detail}` : ""}) — it echoed no model id, so identity is unproven. Raise probeMaxTokens, or pass allowUnreported to accept reachability without identity.`;
|
|
2272
2077
|
const s = r.substitution;
|
|
2273
2078
|
if (s?.verdict === "unreported") return `${r.model}: probe answered without echoing a model id — identity unproven`;
|
|
2274
2079
|
return `${r.model}: probe answered by ${s?.served} (${s?.verdict})`;
|
|
@@ -5047,269 +4852,6 @@ var EvalTraceStore = class {
|
|
|
5047
4852
|
}
|
|
5048
4853
|
};
|
|
5049
4854
|
//#endregion
|
|
5050
|
-
//#region src/experiment-tracker.ts
|
|
5051
|
-
/**
|
|
5052
|
-
* Experiment tracker — git-provenanced experiment log with N-rep stats and a
|
|
5053
|
-
* KEEP / REGRESSION / NOISE verdict against a parent.
|
|
5054
|
-
*
|
|
5055
|
-
* Every loop the fleet runs reduces to the same question: "I ran the candidate
|
|
5056
|
-
* N times — is the median measurably better than the parent, or is the delta
|
|
5057
|
-
* inside the noise band?" The hand-rolled copies bake a fixed score scale
|
|
5058
|
-
* (percentage points), a fixed store path (`.evolve/experiments-v2.json`), and
|
|
5059
|
-
* `execSync('git …')` straight into the module. This is the canonical version:
|
|
5060
|
-
* provenance and persistence are injected, thresholds are configurable, and the
|
|
5061
|
-
* stats + verdict are pure functions you can unit-test without a git repo or a
|
|
5062
|
-
* filesystem.
|
|
5063
|
-
*
|
|
5064
|
-
* Stats per experiment: median / mean / min / max / iqr / stddev / passRate /
|
|
5065
|
-
* n, plus a `stable` flag (`iqr < iqrUnstableAbove && stddev < stddevUnstableAbove`).
|
|
5066
|
-
*
|
|
5067
|
-
* Verdict against a parent (both must have `n >= minRepsForVerdict`):
|
|
5068
|
-
* - NOISE — the candidate is too unstable to judge (`!stable`)
|
|
5069
|
-
* - KEEP — `medianDelta > keepThreshold`
|
|
5070
|
-
* - REGRESSION — `medianDelta < -regressionThreshold`
|
|
5071
|
-
* - NOISE — otherwise (delta inside the band)
|
|
5072
|
-
* With no parent (or insufficient reps) the verdict is the neutral ITERATE.
|
|
5073
|
-
*/
|
|
5074
|
-
const DEFAULTS = {
|
|
5075
|
-
keepThreshold: 5,
|
|
5076
|
-
regressionThreshold: 5,
|
|
5077
|
-
iqrUnstableAbove: 10,
|
|
5078
|
-
stddevUnstableAbove: Number.POSITIVE_INFINITY,
|
|
5079
|
-
minRepsForVerdict: 3
|
|
5080
|
-
};
|
|
5081
|
-
function resolveThresholds(t) {
|
|
5082
|
-
const r = {
|
|
5083
|
-
...DEFAULTS,
|
|
5084
|
-
...t ?? {}
|
|
5085
|
-
};
|
|
5086
|
-
if (r.keepThreshold < 0) throw new ValidationError(`experiment-tracker: keepThreshold must be >= 0, got ${r.keepThreshold}`);
|
|
5087
|
-
if (r.regressionThreshold < 0) throw new ValidationError(`experiment-tracker: regressionThreshold must be >= 0, got ${r.regressionThreshold}`);
|
|
5088
|
-
if (r.minRepsForVerdict < 1) throw new ValidationError(`experiment-tracker: minRepsForVerdict must be >= 1, got ${r.minRepsForVerdict}`);
|
|
5089
|
-
return r;
|
|
5090
|
-
}
|
|
5091
|
-
function median$1(sorted) {
|
|
5092
|
-
const n = sorted.length;
|
|
5093
|
-
if (n === 0) return 0;
|
|
5094
|
-
const mid = Math.floor(n / 2);
|
|
5095
|
-
return n % 2 === 0 ? (sorted[mid - 1] + sorted[mid]) / 2 : sorted[mid];
|
|
5096
|
-
}
|
|
5097
|
-
/** Population standard deviation (÷n). 0 for fewer than 2 values. */
|
|
5098
|
-
function stddev(values, mean) {
|
|
5099
|
-
if (values.length < 2) return 0;
|
|
5100
|
-
const variance = values.reduce((acc, v) => acc + (v - mean) ** 2, 0) / values.length;
|
|
5101
|
-
return Math.sqrt(variance);
|
|
5102
|
-
}
|
|
5103
|
-
/**
|
|
5104
|
-
* Compute the N-rep statistics for a set of reps. Pure — no I/O. The `stable`
|
|
5105
|
-
* flag is the trust gate the verdict depends on: a sample whose spread exceeds
|
|
5106
|
-
* the configured bounds can't distinguish a real delta from run-to-run noise.
|
|
5107
|
-
*/
|
|
5108
|
-
function computeExperimentStats(reps, thresholds) {
|
|
5109
|
-
const t = resolveThresholds(thresholds);
|
|
5110
|
-
const n = reps.length;
|
|
5111
|
-
if (n === 0) return {
|
|
5112
|
-
median: 0,
|
|
5113
|
-
mean: 0,
|
|
5114
|
-
min: 0,
|
|
5115
|
-
max: 0,
|
|
5116
|
-
iqr: 0,
|
|
5117
|
-
stddev: 0,
|
|
5118
|
-
passRate: null,
|
|
5119
|
-
n: 0,
|
|
5120
|
-
stable: false
|
|
5121
|
-
};
|
|
5122
|
-
const scores = reps.map((r) => {
|
|
5123
|
-
if (!Number.isFinite(r.score)) throw new ValidationError(`experiment-tracker: rep ${r.rep} has non-finite score ${r.score}`);
|
|
5124
|
-
return r.score;
|
|
5125
|
-
});
|
|
5126
|
-
const sorted = [...scores].sort((a, b) => a - b);
|
|
5127
|
-
const mean = scores.reduce((s, v) => s + v, 0) / n;
|
|
5128
|
-
const sd = stddev(scores, mean);
|
|
5129
|
-
const spread = iqr(scores);
|
|
5130
|
-
const rated = reps.filter((r) => typeof r.passed === "boolean");
|
|
5131
|
-
const passRate = rated.length === 0 ? null : rated.filter((r) => r.passed).length / rated.length;
|
|
5132
|
-
const stable = spread < t.iqrUnstableAbove && sd < t.stddevUnstableAbove;
|
|
5133
|
-
return {
|
|
5134
|
-
median: median$1(sorted),
|
|
5135
|
-
mean,
|
|
5136
|
-
min: sorted[0],
|
|
5137
|
-
max: sorted[n - 1],
|
|
5138
|
-
iqr: spread,
|
|
5139
|
-
stddev: sd,
|
|
5140
|
-
passRate,
|
|
5141
|
-
n,
|
|
5142
|
-
stable
|
|
5143
|
-
};
|
|
5144
|
-
}
|
|
5145
|
-
/**
|
|
5146
|
-
* Verdict for a candidate against its parent. Pure — operates on already-computed
|
|
5147
|
-
* stats. KEEP/REGRESSION require both sides to have `>= minRepsForVerdict` reps
|
|
5148
|
-
* AND the candidate to be `stable`; otherwise the result is NOISE (unstable) or
|
|
5149
|
-
* ITERATE (not enough reps / no parent).
|
|
5150
|
-
*/
|
|
5151
|
-
function improvementVerdict(candidate, parent, thresholds) {
|
|
5152
|
-
const t = resolveThresholds(thresholds);
|
|
5153
|
-
if (!parent) return {
|
|
5154
|
-
verdict: "ITERATE",
|
|
5155
|
-
medianDelta: null,
|
|
5156
|
-
reason: "no parent experiment to compare against"
|
|
5157
|
-
};
|
|
5158
|
-
if (candidate.n < t.minRepsForVerdict || parent.n < t.minRepsForVerdict) return {
|
|
5159
|
-
verdict: "ITERATE",
|
|
5160
|
-
medianDelta: null,
|
|
5161
|
-
reason: `need >= ${t.minRepsForVerdict} reps on both sides (candidate n=${candidate.n}, parent n=${parent.n})`
|
|
5162
|
-
};
|
|
5163
|
-
if (!candidate.stable) return {
|
|
5164
|
-
verdict: "NOISE",
|
|
5165
|
-
medianDelta: candidate.median - parent.median,
|
|
5166
|
-
reason: `candidate unstable (iqr=${candidate.iqr}, stddev=${candidate.stddev.toFixed(2)})`
|
|
5167
|
-
};
|
|
5168
|
-
const medianDelta = candidate.median - parent.median;
|
|
5169
|
-
if (medianDelta > t.keepThreshold) return {
|
|
5170
|
-
verdict: "KEEP",
|
|
5171
|
-
medianDelta,
|
|
5172
|
-
reason: `median +${medianDelta} > +${t.keepThreshold}`
|
|
5173
|
-
};
|
|
5174
|
-
if (medianDelta < -t.regressionThreshold) return {
|
|
5175
|
-
verdict: "REGRESSION",
|
|
5176
|
-
medianDelta,
|
|
5177
|
-
reason: `median ${medianDelta} < -${t.regressionThreshold}`
|
|
5178
|
-
};
|
|
5179
|
-
return {
|
|
5180
|
-
verdict: "NOISE",
|
|
5181
|
-
medianDelta,
|
|
5182
|
-
reason: `median delta ${medianDelta} inside noise band [-${t.regressionThreshold}, +${t.keepThreshold}]`
|
|
5183
|
-
};
|
|
5184
|
-
}
|
|
5185
|
-
/**
|
|
5186
|
-
* Default provenance reader: `git rev-parse HEAD`, the subject line, and the
|
|
5187
|
-
* files changed vs `HEAD~1`. Fail-loud — a tracker that silently logs
|
|
5188
|
-
* `commit: 'unknown'` corrupts the provenance the whole point of the log is to
|
|
5189
|
-
* carry. When the working tree genuinely has no parent commit, pass an override.
|
|
5190
|
-
*/
|
|
5191
|
-
const gitProvenanceReader = () => {
|
|
5192
|
-
const run = (cmd) => execSync(cmd, { encoding: "utf8" }).trim();
|
|
5193
|
-
const commit = run("git rev-parse --short HEAD");
|
|
5194
|
-
const message = run("git log -1 --format=%s");
|
|
5195
|
-
const changedRaw = run("git diff --name-only HEAD~1");
|
|
5196
|
-
return {
|
|
5197
|
-
commit,
|
|
5198
|
-
message,
|
|
5199
|
-
changedFiles: changedRaw.length === 0 ? [] : changedRaw.split("\n").filter(Boolean)
|
|
5200
|
-
};
|
|
5201
|
-
};
|
|
5202
|
-
/** In-memory store — the default when no persistence is wanted (tests, ephemeral
|
|
5203
|
-
* runs). State lives on the instance. */
|
|
5204
|
-
function inMemoryExperimentStore(initial = []) {
|
|
5205
|
-
let state = initial.map((e) => structuredClone(e));
|
|
5206
|
-
return {
|
|
5207
|
-
async load() {
|
|
5208
|
-
return state.map((e) => structuredClone(e));
|
|
5209
|
-
},
|
|
5210
|
-
async save(experiments) {
|
|
5211
|
-
state = experiments.map((e) => structuredClone(e));
|
|
5212
|
-
}
|
|
5213
|
-
};
|
|
5214
|
-
}
|
|
5215
|
-
/** Filesystem store — a single JSON array at `path`, created on first save. */
|
|
5216
|
-
function fileExperimentStore(path) {
|
|
5217
|
-
return {
|
|
5218
|
-
async load() {
|
|
5219
|
-
const fs = await import("node:fs/promises");
|
|
5220
|
-
try {
|
|
5221
|
-
const raw = await fs.readFile(path, "utf8");
|
|
5222
|
-
const parsed = JSON.parse(raw);
|
|
5223
|
-
if (!Array.isArray(parsed)) throw new ValidationError(`experiment-tracker: store at ${path} is not a JSON array`);
|
|
5224
|
-
return parsed;
|
|
5225
|
-
} catch (err) {
|
|
5226
|
-
if (err.code === "ENOENT") return [];
|
|
5227
|
-
throw err;
|
|
5228
|
-
}
|
|
5229
|
-
},
|
|
5230
|
-
async save(experiments) {
|
|
5231
|
-
const fs = await import("node:fs/promises");
|
|
5232
|
-
const pathMod = await import("node:path");
|
|
5233
|
-
await fs.mkdir(pathMod.dirname(path), { recursive: true });
|
|
5234
|
-
await fs.writeFile(path, JSON.stringify(experiments, null, 2), "utf8");
|
|
5235
|
-
}
|
|
5236
|
-
};
|
|
5237
|
-
}
|
|
5238
|
-
/**
|
|
5239
|
-
* Stateful tracker over an `ExperimentStore`. Create an experiment (provenance
|
|
5240
|
-
* is captured once), append reps as they complete (stats + verdict recompute on
|
|
5241
|
-
* every append), and read the log back for a dashboard. All persistence and git
|
|
5242
|
-
* access flow through the injected seams, so the tracker is fully testable
|
|
5243
|
-
* without a repo or disk.
|
|
5244
|
-
*/
|
|
5245
|
-
var ExperimentTracker = class {
|
|
5246
|
-
store;
|
|
5247
|
-
provenanceReader;
|
|
5248
|
-
thresholds;
|
|
5249
|
-
now;
|
|
5250
|
-
constructor(options = {}) {
|
|
5251
|
-
this.store = options.store ?? inMemoryExperimentStore();
|
|
5252
|
-
this.provenanceReader = options.provenanceReader ?? gitProvenanceReader;
|
|
5253
|
-
this.thresholds = resolveThresholds(options.thresholds);
|
|
5254
|
-
this.now = options.now ?? Date.now;
|
|
5255
|
-
}
|
|
5256
|
-
async create(input) {
|
|
5257
|
-
const experiments = await this.store.load();
|
|
5258
|
-
if (experiments.some((e) => e.id === input.id)) throw new ValidationError(`experiment-tracker: experiment id "${input.id}" already exists`);
|
|
5259
|
-
if (input.parentId && !experiments.some((e) => e.id === input.parentId)) throw new ValidationError(`experiment-tracker: parent experiment "${input.parentId}" not found`);
|
|
5260
|
-
const provenance = input.provenance ?? await this.provenanceReader();
|
|
5261
|
-
const experiment = {
|
|
5262
|
-
id: input.id,
|
|
5263
|
-
label: input.label,
|
|
5264
|
-
provenance,
|
|
5265
|
-
parentId: input.parentId,
|
|
5266
|
-
changeSummary: input.changeSummary,
|
|
5267
|
-
reps: [],
|
|
5268
|
-
stats: computeExperimentStats([], this.thresholds),
|
|
5269
|
-
verdict: "ITERATE",
|
|
5270
|
-
createdAt: new Date(this.now()).toISOString()
|
|
5271
|
-
};
|
|
5272
|
-
experiments.push(experiment);
|
|
5273
|
-
await this.store.save(experiments);
|
|
5274
|
-
return structuredClone(experiment);
|
|
5275
|
-
}
|
|
5276
|
-
/** Append a rep (its `rep` index defaults to the current rep count) and
|
|
5277
|
-
* recompute stats + verdict. Returns the updated experiment. */
|
|
5278
|
-
async addRep(experimentId, rep) {
|
|
5279
|
-
const experiments = await this.store.load();
|
|
5280
|
-
const exp = experiments.find((e) => e.id === experimentId);
|
|
5281
|
-
if (!exp) throw new ValidationError(`experiment-tracker: experiment "${experimentId}" not found`);
|
|
5282
|
-
const fullRep = {
|
|
5283
|
-
rep: rep.rep ?? exp.reps.length,
|
|
5284
|
-
score: rep.score,
|
|
5285
|
-
passed: rep.passed,
|
|
5286
|
-
metrics: rep.metrics,
|
|
5287
|
-
timestamp: rep.timestamp ?? new Date(this.now()).toISOString()
|
|
5288
|
-
};
|
|
5289
|
-
exp.reps.push(fullRep);
|
|
5290
|
-
exp.stats = computeExperimentStats(exp.reps, this.thresholds);
|
|
5291
|
-
const parent = exp.parentId ? experiments.find((e) => e.id === exp.parentId) : void 0;
|
|
5292
|
-
exp.verdict = improvementVerdict(exp.stats, parent?.stats ?? null, this.thresholds).verdict;
|
|
5293
|
-
await this.store.save(experiments);
|
|
5294
|
-
return structuredClone(exp);
|
|
5295
|
-
}
|
|
5296
|
-
async get(experimentId) {
|
|
5297
|
-
const found = (await this.store.load()).find((e) => e.id === experimentId);
|
|
5298
|
-
return found ? structuredClone(found) : void 0;
|
|
5299
|
-
}
|
|
5300
|
-
async list() {
|
|
5301
|
-
return this.store.load();
|
|
5302
|
-
}
|
|
5303
|
-
/** Full verdict (not just the enum) for an experiment vs its parent. */
|
|
5304
|
-
async verdictFor(experimentId) {
|
|
5305
|
-
const experiments = await this.store.load();
|
|
5306
|
-
const exp = experiments.find((e) => e.id === experimentId);
|
|
5307
|
-
if (!exp) throw new ValidationError(`experiment-tracker: experiment "${experimentId}" not found`);
|
|
5308
|
-
const parent = exp.parentId ? experiments.find((e) => e.id === exp.parentId) : void 0;
|
|
5309
|
-
return improvementVerdict(exp.stats, parent?.stats ?? null, this.thresholds);
|
|
5310
|
-
}
|
|
5311
|
-
};
|
|
5312
|
-
//#endregion
|
|
5313
4855
|
//#region src/leaderboard.ts
|
|
5314
4856
|
function mean$1(xs) {
|
|
5315
4857
|
return xs.length ? xs.reduce((a, b) => a + b, 0) / xs.length : null;
|
|
@@ -6228,6 +5770,162 @@ function statusAdvanced(key, progression) {
|
|
|
6228
5770
|
};
|
|
6229
5771
|
}
|
|
6230
5772
|
//#endregion
|
|
5773
|
+
//#region src/verification-strategy.ts
|
|
5774
|
+
/**
|
|
5775
|
+
* The family registry. `Record` over the union keeps it exhaustive: adding
|
|
5776
|
+
* a member to `VerificationStrategySource` without a profile here fails to
|
|
5777
|
+
* compile. The failure mode travels with the taxonomy so a reader of a
|
|
5778
|
+
* certification can surface it without this package's docs at hand.
|
|
5779
|
+
*/
|
|
5780
|
+
const VERIFICATION_STRATEGIES = {
|
|
5781
|
+
compile: {
|
|
5782
|
+
determinism: "deterministic",
|
|
5783
|
+
failureMode: "code that compiles is not code that is correct"
|
|
5784
|
+
},
|
|
5785
|
+
test: {
|
|
5786
|
+
determinism: "deterministic",
|
|
5787
|
+
failureMode: "assumes an answer key; certifies nothing outside suite coverage, and a stubbed integration reports green"
|
|
5788
|
+
},
|
|
5789
|
+
schema: {
|
|
5790
|
+
determinism: "deterministic",
|
|
5791
|
+
failureMode: "shape is not meaning; a well-formed wrong answer passes"
|
|
5792
|
+
},
|
|
5793
|
+
sandbox: {
|
|
5794
|
+
determinism: "deterministic",
|
|
5795
|
+
failureMode: "an exit code compresses the run to one bit; a faked success exits 0"
|
|
5796
|
+
},
|
|
5797
|
+
judge: {
|
|
5798
|
+
determinism: "probabilistic",
|
|
5799
|
+
failureMode: "drifts across model versions and is Goodhart-gameable by the graded policy"
|
|
5800
|
+
},
|
|
5801
|
+
composite: {
|
|
5802
|
+
determinism: "inherited",
|
|
5803
|
+
failureMode: "scalar collapse: the blend hides which member carried the score"
|
|
5804
|
+
},
|
|
5805
|
+
"proof-kernel": {
|
|
5806
|
+
determinism: "deterministic",
|
|
5807
|
+
failureMode: "the formalization gap: the kernel certifies the formal statement, never that it matches the informal claim"
|
|
5808
|
+
},
|
|
5809
|
+
invariant: {
|
|
5810
|
+
determinism: "deterministic",
|
|
5811
|
+
failureMode: "weak invariants pass everything; a set uncalibrated by seeded bugs is a rubber stamp"
|
|
5812
|
+
},
|
|
5813
|
+
replication: {
|
|
5814
|
+
determinism: "deterministic",
|
|
5815
|
+
failureMode: "re-runs the method, so it catches drift and nondeterminism, never an error the method itself carries"
|
|
5816
|
+
},
|
|
5817
|
+
agreement: {
|
|
5818
|
+
determinism: "probabilistic",
|
|
5819
|
+
failureMode: "the shared blind spot: derivers with common corpora or priors agree for the same wrong reason"
|
|
5820
|
+
}
|
|
5821
|
+
};
|
|
5822
|
+
/** Every family member, derived from the registry so it cannot drift. */
|
|
5823
|
+
const VERIFICATION_STRATEGY_SOURCES = Object.keys(VERIFICATION_STRATEGIES);
|
|
5824
|
+
//#endregion
|
|
5825
|
+
//#region src/equivalence-check.ts
|
|
5826
|
+
/** A refused equivalence check. `code` names the exact refusal for programmatic handling. */
|
|
5827
|
+
var EquivalenceProtocolError = class extends Error {
|
|
5828
|
+
code;
|
|
5829
|
+
constructor(code, message) {
|
|
5830
|
+
super(message);
|
|
5831
|
+
this.name = "EquivalenceProtocolError";
|
|
5832
|
+
this.code = code;
|
|
5833
|
+
}
|
|
5834
|
+
};
|
|
5835
|
+
/**
|
|
5836
|
+
* Validate and freeze a two-arm blind equivalence check spec.
|
|
5837
|
+
*
|
|
5838
|
+
* The literal types already refuse a wide design at compile time; the
|
|
5839
|
+
* runtime checks hold the same line for untyped callers. There is no
|
|
5840
|
+
* escape hatch: `arms: 3` or `blind: false` throws, never downgrades.
|
|
5841
|
+
*/
|
|
5842
|
+
function defineEquivalenceCheck(spec) {
|
|
5843
|
+
if (spec.arms !== 2) throw new EquivalenceProtocolError("arm-count", `equivalence check requires exactly 2 arms, received ${String(spec.arms)}`);
|
|
5844
|
+
if (spec.blind !== true) throw new EquivalenceProtocolError("not-blind", "equivalence check requires blind: true — a non-blind run is not a weaker check, it is no check");
|
|
5845
|
+
if (typeof spec.artifact !== "string" || spec.artifact.trim() === "") throw new EquivalenceProtocolError("empty-field", "spec.artifact must identify the claim under verification");
|
|
5846
|
+
if (!VERIFICATION_STRATEGY_SOURCES.includes(spec.source)) throw new EquivalenceProtocolError("unknown-source", `spec.source '${String(spec.source)}' is not a verification-strategy member`);
|
|
5847
|
+
return Object.freeze({ spec: Object.freeze({ ...spec }) });
|
|
5848
|
+
}
|
|
5849
|
+
/**
|
|
5850
|
+
* Assemble the equivalence record, refusing every asymmetry.
|
|
5851
|
+
*
|
|
5852
|
+
* Refusals (each throws `EquivalenceProtocolError`):
|
|
5853
|
+
* - an arm whose `blindness.toOtherArms` is false — it saw the other
|
|
5854
|
+
* statement, so nothing was independently derived;
|
|
5855
|
+
* - an arm whose `blindness.toOutcome` is false — it could steer its
|
|
5856
|
+
* statement toward (or away from) the known result;
|
|
5857
|
+
* - duplicate arm ids, empty statements, empty derivations;
|
|
5858
|
+
* - an obligation whose fields contradict its status (see
|
|
5859
|
+
* `EquivalenceObligation`).
|
|
5860
|
+
*/
|
|
5861
|
+
function buildEquivalenceRecord(definition, arms, obligation) {
|
|
5862
|
+
assertArms(arms);
|
|
5863
|
+
assertObligation(obligation);
|
|
5864
|
+
return Object.freeze({
|
|
5865
|
+
spec: definition.spec,
|
|
5866
|
+
arms: Object.freeze([Object.freeze({ ...arms[0] }), Object.freeze({ ...arms[1] })]),
|
|
5867
|
+
obligation: Object.freeze({ ...obligation })
|
|
5868
|
+
});
|
|
5869
|
+
}
|
|
5870
|
+
/**
|
|
5871
|
+
* Discharge the obligation through an injected checker and return the
|
|
5872
|
+
* record.
|
|
5873
|
+
*
|
|
5874
|
+
* Order matters: every arm refusal fires BEFORE the checker runs — an
|
|
5875
|
+
* invalid check must not spend. A checker whose `strategy` differs from
|
|
5876
|
+
* `spec.source` is refused for the same reason: a judge cannot silently
|
|
5877
|
+
* discharge a proof-kernel obligation.
|
|
5878
|
+
*
|
|
5879
|
+
* A checker failure (`succeeded: false`) is not thrown: it becomes an
|
|
5880
|
+
* `'unresolved'` obligation carrying the full error text, which is the
|
|
5881
|
+
* honest record of an undischarged check.
|
|
5882
|
+
*/
|
|
5883
|
+
async function runEquivalenceCheck(definition, arms, checker) {
|
|
5884
|
+
assertArms(arms);
|
|
5885
|
+
if (checker.strategy !== definition.spec.source) throw new EquivalenceProtocolError("checker-strategy-mismatch", `spec.source is '${definition.spec.source}' but the bound checker declares '${checker.strategy}'`);
|
|
5886
|
+
const outcome = await checker.check({
|
|
5887
|
+
artifact: definition.spec.artifact,
|
|
5888
|
+
statements: [arms[0].statement, arms[1].statement]
|
|
5889
|
+
});
|
|
5890
|
+
if (!outcome.succeeded) return buildEquivalenceRecord(definition, arms, {
|
|
5891
|
+
status: "unresolved",
|
|
5892
|
+
unresolvedReason: outcome.error,
|
|
5893
|
+
checker: checker.identity
|
|
5894
|
+
});
|
|
5895
|
+
const { status, separatingWitness, evidenceDigest } = outcome.value;
|
|
5896
|
+
return buildEquivalenceRecord(definition, arms, {
|
|
5897
|
+
status,
|
|
5898
|
+
...separatingWitness === void 0 ? {} : { separatingWitness },
|
|
5899
|
+
checker: checker.identity,
|
|
5900
|
+
evidenceDigest
|
|
5901
|
+
});
|
|
5902
|
+
}
|
|
5903
|
+
function assertArms(arms) {
|
|
5904
|
+
if (arms.length !== 2) throw new EquivalenceProtocolError("arm-count", `equivalence record requires exactly 2 arms, received ${arms.length}`);
|
|
5905
|
+
if (arms[0].armId === arms[1].armId) throw new EquivalenceProtocolError("duplicate-arm-id", `both arms declare armId '${arms[0].armId}' — two labels for one derivation is one arm`);
|
|
5906
|
+
for (const arm of arms) {
|
|
5907
|
+
if (arm.statement.trim() === "") throw new EquivalenceProtocolError("empty-field", `arm '${arm.armId}' committed an empty statement`);
|
|
5908
|
+
if (arm.derivedFrom.trim() === "") throw new EquivalenceProtocolError("empty-field", `arm '${arm.armId}' declares no derivation provenance`);
|
|
5909
|
+
if (arm.blindness.toOtherArms !== true) throw new EquivalenceProtocolError("arm-saw-other", `arm '${arm.armId}' saw another arm's statement — the derivation is not independent and the check is invalid`);
|
|
5910
|
+
if (arm.blindness.toOutcome !== true) throw new EquivalenceProtocolError("arm-saw-outcome", `arm '${arm.armId}' saw the outcome before committing — the check is invalid`);
|
|
5911
|
+
}
|
|
5912
|
+
}
|
|
5913
|
+
function assertObligation(obligation) {
|
|
5914
|
+
const { status, separatingWitness, unresolvedReason, evidenceDigest } = obligation;
|
|
5915
|
+
if (status === "refuted-with-separating-witness") {
|
|
5916
|
+
if (typeof separatingWitness !== "string" || separatingWitness.trim() === "") throw new EquivalenceProtocolError("witness-missing", "a refuted equivalence must carry the separating witness — a refutation without one is an assertion");
|
|
5917
|
+
if (typeof evidenceDigest !== "string" || evidenceDigest.trim() === "") throw new EquivalenceProtocolError("evidence-missing", "a refuted equivalence must carry its evidence digest");
|
|
5918
|
+
return;
|
|
5919
|
+
}
|
|
5920
|
+
if (status === "proved") {
|
|
5921
|
+
if (separatingWitness !== void 0) throw new EquivalenceProtocolError("witness-on-proved", "a proved equivalence cannot carry a separating witness — the two claims contradict");
|
|
5922
|
+
if (typeof evidenceDigest !== "string" || evidenceDigest.trim() === "") throw new EquivalenceProtocolError("evidence-missing", "a proved equivalence must carry its evidence digest");
|
|
5923
|
+
return;
|
|
5924
|
+
}
|
|
5925
|
+
if (separatingWitness !== void 0) throw new EquivalenceProtocolError("witness-on-unresolved", "an unresolved obligation cannot carry a separating witness — a witness in hand is a refutation");
|
|
5926
|
+
if (typeof unresolvedReason !== "string" || unresolvedReason.trim() === "") throw new EquivalenceProtocolError("reason-missing", "an unresolved obligation must say why — 'unresolved' with no reason erases the diagnostic");
|
|
5927
|
+
}
|
|
5928
|
+
//#endregion
|
|
6231
5929
|
//#region src/ui-finding.ts
|
|
6232
5930
|
/** Frozen tuple of lenses for validation + iteration. */
|
|
6233
5931
|
const UI_LENSES = [
|
|
@@ -7428,126 +7126,6 @@ async function promptBisect(options) {
|
|
|
7428
7126
|
};
|
|
7429
7127
|
}
|
|
7430
7128
|
//#endregion
|
|
7431
|
-
//#region src/counterfactual.ts
|
|
7432
|
-
/**
|
|
7433
|
-
* Counterfactual replay — "what would have happened if we'd changed
|
|
7434
|
-
* exactly one thing at turn N?"
|
|
7435
|
-
*
|
|
7436
|
-
* The framework does NOT drive the agent — it sets up the replay
|
|
7437
|
-
* context (prior spans, prior state, mutation spec) and records the
|
|
7438
|
-
* resulting divergence. Consumers supply an `executeFrom(ctx)` callback
|
|
7439
|
-
* that runs their agent starting from turn N with the mutation applied.
|
|
7440
|
-
*
|
|
7441
|
-
* Counterfactual runs are recorded as a new Run with `layer='meta'` and
|
|
7442
|
-
* `parentRunId = originalRunId`, so downstream diff + correlation
|
|
7443
|
-
* pipelines see them natively.
|
|
7444
|
-
*/
|
|
7445
|
-
async function runCounterfactual(store, originalRunId, mutation, runner) {
|
|
7446
|
-
const originalRun = await store.getRun(originalRunId);
|
|
7447
|
-
if (!originalRun) throw new NotFoundError(`counterfactual: run ${originalRunId} not found`);
|
|
7448
|
-
const trajectory = await buildTrajectory(store, originalRunId);
|
|
7449
|
-
if (mutation.at < 0 || mutation.at >= trajectory.steps.length) throw new ValidationError(`counterfactual: mutation.at=${mutation.at} out of range [0, ${trajectory.steps.length})`);
|
|
7450
|
-
const targetStep = trajectory.steps[mutation.at];
|
|
7451
|
-
const mutatedStep = applyMutation(targetStep, mutation);
|
|
7452
|
-
const cfEmitter = new TraceEmitter(store);
|
|
7453
|
-
await cfEmitter.startRun({
|
|
7454
|
-
scenarioId: originalRun.scenarioId,
|
|
7455
|
-
variantId: originalRun.variantId ? `${originalRun.variantId}+cf:${mutation.kind}@${mutation.at}` : `cf:${mutation.kind}@${mutation.at}`,
|
|
7456
|
-
projectId: originalRun.projectId,
|
|
7457
|
-
parentRunId: originalRunId,
|
|
7458
|
-
layer: "meta",
|
|
7459
|
-
tags: {
|
|
7460
|
-
counterfactual: "true",
|
|
7461
|
-
mutationKind: mutation.kind,
|
|
7462
|
-
mutationAt: String(mutation.at)
|
|
7463
|
-
}
|
|
7464
|
-
});
|
|
7465
|
-
await runner.executeFrom({
|
|
7466
|
-
originalRunId,
|
|
7467
|
-
originalTrajectory: trajectory,
|
|
7468
|
-
prefix: trajectory.steps.slice(0, mutation.at),
|
|
7469
|
-
mutation,
|
|
7470
|
-
mutatedStep
|
|
7471
|
-
}, cfEmitter);
|
|
7472
|
-
const counterfactual = await store.getRun(cfEmitter.runId);
|
|
7473
|
-
const delta = {
|
|
7474
|
-
originalOutcomeScore: originalRun.outcome?.score ?? null,
|
|
7475
|
-
counterfactualOutcomeScore: counterfactual?.outcome?.score ?? null,
|
|
7476
|
-
deltaScore: originalRun.outcome?.score !== void 0 && counterfactual?.outcome?.score !== void 0 ? counterfactual.outcome.score - originalRun.outcome.score : null
|
|
7477
|
-
};
|
|
7478
|
-
return {
|
|
7479
|
-
counterfactualRunId: cfEmitter.runId,
|
|
7480
|
-
originalRunId,
|
|
7481
|
-
mutation,
|
|
7482
|
-
delta
|
|
7483
|
-
};
|
|
7484
|
-
}
|
|
7485
|
-
function applyMutation(step, mutation) {
|
|
7486
|
-
if (mutation.kind === "swap-model" && step.span.kind === "llm") {
|
|
7487
|
-
const llm = step.span;
|
|
7488
|
-
return {
|
|
7489
|
-
...step,
|
|
7490
|
-
span: {
|
|
7491
|
-
...llm,
|
|
7492
|
-
model: mutation.newModel
|
|
7493
|
-
}
|
|
7494
|
-
};
|
|
7495
|
-
}
|
|
7496
|
-
if (mutation.kind === "swap-tool-result" && step.span.kind === "tool") {
|
|
7497
|
-
const tool = step.span;
|
|
7498
|
-
return {
|
|
7499
|
-
...step,
|
|
7500
|
-
span: {
|
|
7501
|
-
...tool,
|
|
7502
|
-
result: mutation.newResult
|
|
7503
|
-
}
|
|
7504
|
-
};
|
|
7505
|
-
}
|
|
7506
|
-
if (mutation.kind === "inject-system-message" && step.span.kind === "llm") {
|
|
7507
|
-
const llm = step.span;
|
|
7508
|
-
return {
|
|
7509
|
-
...step,
|
|
7510
|
-
span: {
|
|
7511
|
-
...llm,
|
|
7512
|
-
messages: [{
|
|
7513
|
-
role: "system",
|
|
7514
|
-
content: mutation.content
|
|
7515
|
-
}, ...llm.messages]
|
|
7516
|
-
}
|
|
7517
|
-
};
|
|
7518
|
-
}
|
|
7519
|
-
if (mutation.kind === "custom") return mutation.apply(step);
|
|
7520
|
-
return step;
|
|
7521
|
-
}
|
|
7522
|
-
/**
|
|
7523
|
-
* Aggregate a batch of counterfactuals into a simple attribution table:
|
|
7524
|
-
* which mutation kinds move outcomes most? (Useful when you run a grid
|
|
7525
|
-
* over the same trajectory — swap-model at every llm span, swap-tool
|
|
7526
|
-
* at every tool span — and want a ranked summary.)
|
|
7527
|
-
*/
|
|
7528
|
-
function attributeCounterfactuals(results) {
|
|
7529
|
-
const grouped = /* @__PURE__ */ new Map();
|
|
7530
|
-
for (const r of results) {
|
|
7531
|
-
const arr = grouped.get(r.mutation.kind) ?? [];
|
|
7532
|
-
arr.push(r);
|
|
7533
|
-
grouped.set(r.mutation.kind, arr);
|
|
7534
|
-
}
|
|
7535
|
-
const out = [];
|
|
7536
|
-
for (const [kind, items] of grouped) {
|
|
7537
|
-
const deltas = items.map((i) => i.delta.deltaScore).filter((d) => typeof d === "number");
|
|
7538
|
-
if (deltas.length === 0) continue;
|
|
7539
|
-
const meanAbs = deltas.reduce((a, b) => a + Math.abs(b), 0) / deltas.length;
|
|
7540
|
-
const meanSigned = deltas.reduce((a, b) => a + b, 0) / deltas.length;
|
|
7541
|
-
out.push({
|
|
7542
|
-
mutationKind: kind,
|
|
7543
|
-
n: deltas.length,
|
|
7544
|
-
meanAbsDelta: meanAbs,
|
|
7545
|
-
meanSignedDelta: meanSigned
|
|
7546
|
-
});
|
|
7547
|
-
}
|
|
7548
|
-
return out.sort((a, b) => b.meanAbsDelta - a.meanAbsDelta);
|
|
7549
|
-
}
|
|
7550
|
-
//#endregion
|
|
7551
7129
|
//#region src/cross-trace-diff.ts
|
|
7552
7130
|
async function crossTraceDiff(store, runA, runB, options = {}) {
|
|
7553
7131
|
const [a, b] = await Promise.all([buildTrajectory(store, runA), buildTrajectory(store, runB)]);
|
|
@@ -11402,14 +10980,20 @@ function attachCostToReport(report, ledger) {
|
|
|
11402
10980
|
/**
|
|
11403
10981
|
* Tier presets — plain data, swap or spread freely.
|
|
11404
10982
|
*
|
|
11405
|
-
*
|
|
11406
|
-
*
|
|
11407
|
-
*
|
|
11408
|
-
*
|
|
11409
|
-
*
|
|
11410
|
-
*
|
|
11411
|
-
*
|
|
11412
|
-
*
|
|
10983
|
+
* A preset names REQUESTED ids, and a requested id is NOT a guarantee of
|
|
10984
|
+
* family, of provider, or of liveness. A routing gateway can accept any id
|
|
10985
|
+
* below and answer from a different model on HTTP 200, with only the response
|
|
10986
|
+
* body's `model` field betraying the swap. The served assertion is the
|
|
10987
|
+
* guarantee: gate a run on `assertModelsServed({ probe: true })` and assert
|
|
10988
|
+
* `assertServedModel` per call. `assertCrossFamily` over these ids proves the
|
|
10989
|
+
* configuration is diverse; only `assertCrossFamilyServed` over the ids that
|
|
10990
|
+
* ANSWERED proves the run was.
|
|
10991
|
+
*
|
|
10992
|
+
* `economy` names ids a live router probe answered from the provider their
|
|
10993
|
+
* name implies, so the judge trio spanned three provider families (deepseek /
|
|
10994
|
+
* zhipu / google) as configured and, at that probe, as served. A preset is
|
|
10995
|
+
* only as good as its last probe: ids go dead and start resolving elsewhere
|
|
10996
|
+
* without notice, so re-probe rather than trusting this list.
|
|
11413
10997
|
*
|
|
11414
10998
|
* `frontier` is deliberately EMPTY: entitled frontier ids vary per router
|
|
11415
10999
|
* account, and a hardcoded claude/gpt-5 id 401s on keys that lack it. Supply
|
|
@@ -12400,6 +11984,6 @@ function assertProductBenchmarkRun(runDir) {
|
|
|
12400
11984
|
return report;
|
|
12401
11985
|
}
|
|
12402
11986
|
//#endregion
|
|
12403
|
-
export { AGENT_PROFILE_KINDS, ATIF_SCHEMA_VERSION, ATTESTATION_ALGORITHM, AgentDriver, AgentEvalError, AgentProfileCellValidationError, AnalystRegistry, BENCHMARK_SPLIT_SEED, BOOTSTRAP_GATE_MIN_N, BackendIntegrityError, BenchmarkRunner, BudgetBreachError, BudgetGuard, CODING_HARNESSES, CONTROL_INTEGRITY_ANALYST, CallbackResearcher, CaptureIntegrityError, ConfigError, ControlIntegrityAnalyst, ConvergenceTracker, CostAccountingIncompleteError, CostCallConflictError, CostCeilingReachedError, CostLedger, CostLedgerPersistenceError, CostReceiptCaptureError, CostReservationExceededError, CostTracker, CrossFamilyError, DECISION_PAIRED_DELTA_STATISTIC, DEFAULT_AGENT_SLOS, DEFAULT_COMPLEXITY_WEIGHTS, DEFAULT_RULES as DEFAULT_FAILURE_RULES, DEFAULT_FINDERS, DEFAULT_HARNESS_OBJECTIVES, DEFAULT_MUTATION_PRIMITIVES, DEFAULT_MUTATORS, DEFAULT_PERMUTATIONS, DEFAULT_PR_REVIEW_SCORE_WEIGHTS, DEFAULT_REDACTION_RULES, DEFAULT_RED_TEAM_CORPUS, DEFAULT_RUN_SCORE_WEIGHTS, DEFAULT_SEVERITY_WEIGHTS, DEFAULT_TRACE_ANALYST_BUDGETS, DEFAULT_TRACE_ANALYST_KINDS, DEFAULT_TRACE_ANALYST_LIMITS, Dataset, DescriptionLengthGate, DockerSandboxDriver, DualAgentBench, ERROR_COUNT_PATTERNS, EvalTraceStore, ExactAnalystRunExecutionError, ExperimentTracker, FAILURE_CLASSES, FAILURE_MODE_KIND_SPEC, FileSystemFeedbackTrajectoryStore, FileSystemRawProviderSink, FileSystemTraceStore, FindingsStore, HARBOR_IMPORT_GAP, HARNESS_BRIEFS, HARNESS_NATIVE_MODEL, HeldOutGate, HoldoutAuditor, HoldoutLockedError, IMPROVEMENT_KIND_SPEC, INPUT_VALUE, INTENT_MATCH_JUDGE_VERSION, InMemoryFeedbackTrajectoryStore, InMemoryRawProviderSink, InMemoryTraceStore, InMemoryWorkspaceInspector, JudgeError, JudgeParseError, JudgeRunner, KNOWLEDGE_GAP_KIND_SPEC, KNOWLEDGE_POISONING_KIND_SPEC, LLM_CACHED_TOKENS, LLM_CACHED_TOKEN_ATTR_KEYS, LLM_CACHE_WRITE_TOKENS, LLM_CACHE_WRITE_TOKEN_ATTR_KEYS, LLM_CONTEXT_TOKENS, LLM_COST_ATTR_KEYS, LLM_COST_USD, LLM_INPUT_TOKENS, LLM_INPUT_TOKEN_ATTR_KEYS, LLM_MODEL_ATTR_KEYS, LLM_MODEL_NAME, LLM_OUTPUT_TOKENS, LLM_OUTPUT_TOKEN_ATTR_KEYS, LLM_REASONING_TOKENS, LLM_REASONING_TOKEN_ATTR_KEYS, LimitExceededError, LlmCallError, LlmClient, LlmResponseError, LlmRouteAssertionError, LockedJsonlAppender, MANN_WHITNEY_EXACT_MAX_STATES, MANN_WHITNEY_EXACT_MAX_WORK, MODEL_PRICING, MetricsCollector, ModelSubstitutionError, ModelsUnreachableError, MultiLayerVerifier, Mutex, NoopRawProviderSink, NoopResearcher, NotFoundError, OPENINFERENCE_SPAN_KIND, OTEL_AGENT_EVAL_SCOPE, OUTPUT_VALUE, OtlpFileTraceStore, PairwiseSteeringOptimizer, ProductClient, PromptRegistry, REDACTION_VERSION, REFERENCE_EQUIVALENCE_INPUT_LIMITS, REFERENCE_EQUIVALENCE_JUDGE_VERSION, RESEARCH_REPORT_HARD_PAIR_FLOOR, ROLLOUT_SCHEMA, RUN_COST_ATTR_KEYS, ReplayCache, ReplayCacheMissError, ReplayError, RunCritic, RunIntegrityError, RunRecordValidationError, SEMANTIC_CONCEPT_JUDGE_VERSION, SKILL_USAGE_ANALYST, SPAN_KIND_ATTR_KEYS, SUPERVISOR_RUN_INTEGRITY_SCHEMA, SUPERVISOR_RUN_SCHEMA, SandboxHarness, ScenarioRegistry, SeatUnsetError, ServedCrossFamilyError, SingleBackendError, SkillUsageAnalyst, SpanNotFoundError, SubprocessSandboxDriver, TOOL_ARGS_CAPTURED, TOOL_LATENCY_MS, TOOL_NAME, TOOL_NAME_ATTR_KEYS, TRACE_ANALYSIS_LIMITS, TRACE_ANALYST_ACTOR_DESCRIPTION, TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION, TRACE_ANALYST_TOOL_NAMESPACE, TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, TRACE_SCHEMA_VERSION, TokenCounter, ToolTraceMissingError, TraceAnalysisLimitError, TraceAnalysisStoreContractError, TraceAnalysisValidationError, TraceContractBuilder, TraceEmitter, TraceFileMalformedError, TraceFileMissingError, TraceFileTooLargeError, TraceNotFoundError, UI_FINDING_SEVERITIES, UI_LENSES, UNIVERSAL_FINDERS, ValidationError, VerificationError, WILCOXON_EXACT_MAX_N, WORKER_DRIVER_DOCTRINE, acquisitionPlansForKnowledgeGaps, agentProfileCellHashMaterial, agentProfileCellKey, agentProfileHash, agentProfileId, agentProfileModelId, agentVisibleFields, aggregateJudgeVerdicts, aggregateLlm, aggregatePrReviewScore, aggregateRunScore, allCriticalPassed, analystFindingDigest, analystRunDigest, analystRunToFeedbackTrajectory, analystRunToReviewRequests, analyzeAntiSlop, analyzeSeries, analyzeSupervisorRun, analyzeSupervisorRunIntegrity, analyzeSupervisorRunSources, analyzeTraces, appendScorecard, applyLlmSpanOtlpAttributes, applyToolSpanOtlpAttributes, argHash, asNumber, asString, assertCapabilityHeadroom, assertCrossFamily, assertCrossFamilyServed, assertExactRegistryRunOpts, assertLlmRoute, assertMinted, assertMintedLines, assertModelsServed, assertNoHiddenLeak, assertProductBenchmarkRun, assertRealAgentReceipts, assertRealBackend, assertReleaseConfidence, assertRolloutLine, assertRunAgentProfileCell, assertRunCaptured, assertServedModel, assertServedModels, assertSingleBackend, assignFeedbackSplit, assignHeldOutTag, attachCostToReport, attest, attributeCounterfactuals, backoffMs, deterministicSplit as benchmarkDeterministicSplit, benchmarks_exports as benchmarks, benjaminiHochberg, bisect, blendHeldout, blockingKnowledgeEval, bonferroni, bootstrapCi, buildAgentInterfaceProfileCell, buildAgentProfileCell, buildDefaultAnalystRegistry, buildDriverSystemPrompt, buildProductBenchmarkManifest, buildReflectionPrompt, buildReviewerPrompt, buildTraceAnalysisToolDescriptors, buildTraceInsightContext, buildTraceInsightPrompt, buildTrajectory, byteLengthRange, cachedJudge, calibrateJudge, calibrateJudgeContinuous, callLlm, callLlmJson, canaryLeakView, canonicalJson, canonicalize, capabilityHeadroom, captureFetchToRawSink, causalAttribution, checkBehavioralCanary, checkCanaries, checkServedModel, checkSlos, checkTraceContracts, clamp01, classifyFailure, classifyOtlpSpanRole, classifyTreatment, claudeCodeSupervisorRunReader, cliffsDelta, clusteredPairedBinary, cohensD, collectionPreserved, commentsForSource, commitBisect, comparePairedArms, compareReferenceReplay, compareToBaseline, compilerJudge, completionVerdict, composeParsers, composeValidators, computeExperimentStats, computeFindingId, computeToolUseMetrics, computeTraceMetrics, confidenceInterval, containsAll, contentHash, contextInputTokens, continuousAgreement, contractJudge, controlFailureClassFromVerification, controlRunToFeedbackTrajectory, controlRunToRunRecord, convertTraceStoresToOtlp, corpusInterRaterAgreement, corpusInterRaterAgreementFromJudgeScores, costForTokenPricing, costForUsage, costReceiptFromLlm, costReceiptFromLlmError, costReport, createAntiSlopJudge, createBoundedTraceAnalysisStore, createChatClient, createDefaultReviewer, createDspyRlmTraceEngine, createFeedbackTrajectory, createIntentMatchJudge, createLlmCorrectnessChecker, createLlmReviewer, createOtelExporter, createOtelTracingStore, createReferenceEquivalenceJudge, createReplayFetch, createSandboxPool, createSemanticConceptJudge, createTokenRecallChecker, createTraceAnalyst, crossTraceDiff, crowdingDistance, dataDescriptionBits, decideNextUserTurn, decidePairedPromotion, decideReferenceReplayPromotion, decideReferenceReplayRunPromotion, defaultBlendWeights, defaultIsMaterial, defaultProviderRedactor, defaultReferenceReplayMatcher, defaultTraceInsightPanel, defineTraceAnalyst, deployGateLayer, describeTraceInsightScope, diffFindings, diffScorecard, discoverPersonas, distillPlaybook, domainEvidencePattern, dominates, eProcess, emitControlIntegrityFindings, ensembleJudge, errorStreakDetector, estimateCost, estimateTokens, evaluateActionPolicy, evaluateContract, evaluateHypothesis, evaluateInterimReleaseConfidence, evaluateOracles, evaluateReleaseConfidence, evaluateTraceContract, executeScenario, expandProfileAxes, expectAgent, exportProductBenchmark, exportProductBenchmarkRuns, exportRewardModel, exportRunAsOtlp, extractAssetUrls, extractErrorCount, extractOtlpAttributes, extractProducedState, extractUsage, extractUsageFromResponse, extractUsageFromSse, feedbackTrajectoriesToDatasetScenarios, feedbackTrajectoriesToOptimizerRows, feedbackTrajectoryToDatasetScenario, feedbackTrajectoryToOptimizerRow, fileContains, fileExists, fileExperimentStore, fileVerdictCache, findAutoMatchNoExpectation, findConstructorCwdDropped, findFallbackToPass, findLiteralTruePass, findProductBenchmarkArtifacts, findSkipCountsAsPass, firstNumberAttr, firstStringAttr, flattenOtlpExportToNdjson, flowLayer, fnv1a32, formatBenchmarkReport, formatDriverReport, formatFindings, formatScorecardDiff, fromHarborTrajectory, gainHistogram, gateTreatmentApplied, gateTreatmentFromMetrics, gateTreatmentFromSpans, gateTreatmentFromToolSpans, ghCliClient, gitProvenanceReader, precision as goldenPrecision, gradeOnHidden, gradeSemanticStatus, groupBy, groupRunsByAgentProfileCell, harnessAxisOf, hasCapturedToolArgs, hashContent, hashJson, hashScenarios, hashToUnit, hiddenGrade, holm, htmlContainsElement, httpGithubClient, improvementVerdict, inMemoryExperimentStore, inMemoryReferenceReplayStore, inMemoryReviewStore, inMemoryRunRecordBackend, inMemoryVerdictCache, inferDomainKeywords, inferOtlpKind, interRaterReliability, interpretCliffs, iqr, isBinaryOutcomeVector, isHiddenDestination, isJudgeSpan, isLlmSpan, isModelPriced, isOtelConfigured, isOtlpModelCall, isRealnessGated, isRetrievalSpan, isRolloutLine, isRunRecord, isRuntimeSupervisorRunDir, isSandboxSpan, isToolSpan, isTrainableSplit, isTransientLlmError, isUnavailable, iterateRawCalls, jestTestParser, jsonHasKeys, jsonShape, jsonlReferenceReplayStore, jsonlReviewStore, jsonlRunRecordBackend, judgeFamily, judgeReplayGate, judgeSpans, keyPreserved, knowledgeReadinessTracePayload, leaderboard, linterJudge, llmJudge, llmSpanFromProvider, llmSpans, loadScorecard, loadScorerFromGrader, localCommandRunner, lowercaseMutator, makeEvalTools, makeFinding, makeProposalFinding, mannWhitneyU, mapConcurrent, matchGoldens, matchSpan, maximumChargeForLlmRequest, mcnemar, mcnemarPower, mcnemarRequiredN, mergeLayerResults, mergeSteeringBundle, minimumPairsForPairedDeltaTest, mintRolloutRows, modelDescriptionBits, modelHasSnapshot, modelPriceKey, mulberry32, multiToolchainLayer, noProgressDetector, normalCdf, normalizeModelId, normalizeScores, notBlocked, objectiveEval, observeAll, observedScore, observedSplitScore, otelRunCompleteHook, otlpRowsToRunRecords, otlpRowsToTraceRunRecords, otlpTextToTraceAnalysisStore, otlpToRunRecords, otlpToTraceRunRecords, pairArms, pairRunRecords, pairedBinaryScale, pairedBootstrap, pairedCohensDz, pairedDecisionShape, pairedDeltaTest, pairedDeltaTieFraction, pairedEvalueSequence, pairedMde, pairedRiskDifference, pairedRiskDifferenceExact, pairedRiskDifferenceScore, pairedSignTest, pairedTTest, paraphraseRobustness, paraphraseRobustnessScenarios, paretoChart, paretoFrontier, paretoFrontierWithCrowding, parseCorrectnessResponse, parseFeedbackTrajectoriesJsonl, parseReflectionResponse, parseRunRecordSafe, parseRuntimeTrajectoryHookEvent, partialCredit, partitionHeldOut, passAtK, passOrthogonality, pearsonR, pixelDeltaRatio, planTraceInsightQuestions, politenessPrefixMutator, positionalBias, preflightModels, printDriverSummary, probeLlm, productBenchmarkIntegrityFailures, productBenchmarkMutableSurfaces, productBenchmarkRepoIdentity, productBenchmarkSplits, profile_exports as profile, projectOtlpFlatLine, projectRuntimeTrajectoryEvidence, promptBisect, proposeSynthesisTargets, providerFromBaseUrl, pytestTestParser, ranks, readClaudeCodeSupervisorRun, readOtlpStatus, readProductBenchmarkManifest, readProductBenchmarkRecords, readRuntimeSupervisorRun, recordRuns, recordRunsToScorecard, redTeamDataset, redTeamReport, redactString, redactValue, referenceReplayRunsToSteeringRows, referenceReplayScenarioToRunScore, regexMatch, regexMatches, relabelImportedSplit, renderMarkdownReport, renderPlaybookMarkdown, renderPreferenceMemoryMarkdown, renderPriorFindings, renderReleaseReport, renderSteeringText, renderSupervisorRunHeadline, renderSupervisorRunMarkdown, renderUpstreamFindings, repeatedActionDetector, replayFeedbackTrajectories, replayFeedbackTrajectory, replayScorerOverCorpus, replayTraceThroughJudge, requireAgentProfileCell, requiredPairedSampleSize, requiredSampleSize, researchReport, resetDeprecationWarnings, resolveModelPricing, resolveSeat, resolveTraceAnalystLimits, rollupSupervisorRuns, roundTripRunRecord, routeFields, rowCount, rowWhere, runAgentControlLoop, runAssertions, runBehavioralCanaries, runCanaries, runCounterfactual, runE2EWorkflow, runEvalCampaign, runExpectations, runFailureClass, runHarnessExperiment, runIntentMatchJudge, runJudgeFleet, runKeywordCoverageJudge, runKeywordCoverageJudgeUrl, runLiveProof, runProposeReview, runProposeReviewAsControlLoop, runRecordToProductBenchmarkRecord, runReferenceEquivalenceJudge, runReferenceReplay, runScore, runSelfPlay, runSemanticConceptJudge, runTaskScore, runTestGradedScenario, runTraceAnalyst, runsForScenario, runtimeSupervisorRunReader, scalarScore, scanForMuffledGates, scoreContinuity, scoreFromEvals, scoreKnowledgeReadiness, scoreOrigin, scorePrReviewComments, scorePrReviewSource, scoreRedTeamOutput, scoreReferenceReplay, scoreTraceInsightReadiness, seatPresets, securityJudge, selectHarnessVariant, selfPreference, sentenceReorderMutator, serializeFeedbackTrajectoriesJsonl, servedModelAcceptable, showMeasured, signManifest, spearmanR, statusAdvanced, stopOnNoProgress, stopOnRepeatedAction, stringField, stripFencedJson, studentTCdf, subjectiveEval, summarizeAgentReceiptIntegrity, summarizeBackendIntegrity, summarizeHarnessResults, summarizePrReviewBenchmark, summarizePreferenceMemory, summaryTable, supervisorRunRolloutLines, testJudge, textInSnapshot, throwIfRunIncomplete, toAgentProfileJson, toHarborTrajectories, toHarborTrajectory, toJsonl, toLangfuseEnvelope, toOpenAiTool, toPrometheusText, toRewardRows, toSftRows, tokenizeDomainWords, toolNamesForRun, toolSpans, toolSpansToTraceAnalysisStore, traceAnalystFunctionGroup, traceAnalystOnRunComplete, traceContract, traceJudge, traceJudgeEnsemble, traceSpanKindToOpenInferenceKind, tracedAnalyzeTraces, trainingReward, trainingScore, typoMutator, unmintableReasons, urlContains, userQuestionsForKnowledgeGaps, validateAgentProfileCell, validateProductBenchmarkManifest, validateProductBenchmarkRecord, validateProductBenchmarkRun, validateRolloutLine, validateRunRecord, verbosityBias, verifyAgentProfileCell, verifyAttestation, verifyCompletion, verifyManifest, visualDiff, viteDeployRunner, vitestTestParser, warnDeprecatedOnce, weightedComposite, weightedMean, weightedRecall, welchsTTest, whitespaceCollapseMutator, wilcoxonSignedRank, wilson, withAssignedFeedbackSplit, withHeldoutBlend, withJudgeRetry, withOtelPipeline, wranglerDeployRunner, writeSupervisorRunReport };
|
|
11987
|
+
export { AGENT_PROFILE_KINDS, ATIF_SCHEMA_VERSION, ATTESTATION_ALGORITHM, AgentDriver, AgentEvalError, AgentProfileCellValidationError, AnalystRegistry, BENCHMARK_SPLIT_SEED, BOOTSTRAP_GATE_MIN_N, BackendIntegrityError, BenchmarkRunner, BudgetBreachError, BudgetGuard, CODING_HARNESSES, CONTROL_INTEGRITY_ANALYST, CallbackResearcher, CaptureIntegrityError, ConfigError, ControlIntegrityAnalyst, ConvergenceTracker, CostAccountingIncompleteError, CostCallConflictError, CostCeilingReachedError, CostLedger, CostLedgerPersistenceError, CostReceiptCaptureError, CostReservationExceededError, CostTracker, CrossFamilyError, DECISION_PAIRED_DELTA_STATISTIC, DEFAULT_AGENT_SLOS, DEFAULT_COMPLEXITY_WEIGHTS, DEFAULT_RULES as DEFAULT_FAILURE_RULES, DEFAULT_FINDERS, DEFAULT_HARNESS_OBJECTIVES, DEFAULT_MUTATION_PRIMITIVES, DEFAULT_MUTATORS, DEFAULT_PERMUTATIONS, DEFAULT_PR_REVIEW_SCORE_WEIGHTS, DEFAULT_REDACTION_RULES, DEFAULT_RED_TEAM_CORPUS, DEFAULT_RUN_SCORE_WEIGHTS, DEFAULT_SEVERITY_WEIGHTS, DEFAULT_TRACE_ANALYST_BUDGETS, DEFAULT_TRACE_ANALYST_KINDS, DEFAULT_TRACE_ANALYST_LIMITS, Dataset, DescriptionLengthGate, DockerSandboxDriver, DualAgentBench, ERROR_COUNT_PATTERNS, EquivalenceProtocolError, EvalTraceStore, ExactAnalystRunExecutionError, ExperimentTracker, FAILURE_CLASSES, FAILURE_MODE_KIND_SPEC, FileSystemFeedbackTrajectoryStore, FileSystemRawProviderSink, FileSystemTraceStore, FindingsStore, HARBOR_IMPORT_GAP, HARNESS_BRIEFS, HARNESS_NATIVE_MODEL, HeldOutGate, HoldoutAuditor, HoldoutLockedError, IMPROVEMENT_KIND_SPEC, INPUT_VALUE, INTENT_MATCH_JUDGE_VERSION, InMemoryFeedbackTrajectoryStore, InMemoryRawProviderSink, InMemoryTraceStore, InMemoryWorkspaceInspector, JudgeError, JudgeParseError, JudgeRunner, KNOWLEDGE_GAP_KIND_SPEC, KNOWLEDGE_POISONING_KIND_SPEC, LLM_CACHED_TOKENS, LLM_CACHED_TOKEN_ATTR_KEYS, LLM_CACHE_WRITE_TOKENS, LLM_CACHE_WRITE_TOKEN_ATTR_KEYS, LLM_CONTEXT_TOKENS, LLM_COST_ATTR_KEYS, LLM_COST_USD, LLM_INPUT_TOKENS, LLM_INPUT_TOKEN_ATTR_KEYS, LLM_MODEL_ATTR_KEYS, LLM_MODEL_NAME, LLM_OUTPUT_TOKENS, LLM_OUTPUT_TOKEN_ATTR_KEYS, LLM_REASONING_TOKENS, LLM_REASONING_TOKEN_ATTR_KEYS, LimitExceededError, LlmCallError, LlmClient, LlmResponseError, LlmRouteAssertionError, LockedJsonlAppender, MANN_WHITNEY_EXACT_MAX_STATES, MANN_WHITNEY_EXACT_MAX_WORK, MODEL_PRICING, MetricsCollector, ModelSubstitutionError, ModelsUnreachableError, MultiLayerVerifier, Mutex, NoopRawProviderSink, NoopResearcher, NotFoundError, OPENINFERENCE_SPAN_KIND, OTEL_AGENT_EVAL_SCOPE, OUTPUT_VALUE, OtlpFileTraceStore, PROBE_MAX_TOKENS, PairwiseSteeringOptimizer, ProductClient, PromptRegistry, REDACTION_VERSION, REFERENCE_EQUIVALENCE_INPUT_LIMITS, REFERENCE_EQUIVALENCE_JUDGE_VERSION, RESEARCH_REPORT_HARD_PAIR_FLOOR, ROLLOUT_SCHEMA, RUN_COST_ATTR_KEYS, ReplayCache, ReplayCacheMissError, ReplayError, RunCritic, RunIntegrityError, RunRecordValidationError, SEMANTIC_CONCEPT_JUDGE_VERSION, SKILL_USAGE_ANALYST, SPAN_KIND_ATTR_KEYS, SUPERVISOR_RUN_INTEGRITY_SCHEMA, SUPERVISOR_RUN_SCHEMA, SandboxHarness, ScenarioRegistry, SeatUnsetError, ServedCrossFamilyError, SingleBackendError, SkillUsageAnalyst, SpanNotFoundError, SubprocessSandboxDriver, TOOL_ARGS_CAPTURED, TOOL_LATENCY_MS, TOOL_NAME, TOOL_NAME_ATTR_KEYS, TRACE_ANALYSIS_LIMITS, TRACE_ANALYST_ACTOR_DESCRIPTION, TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION, TRACE_ANALYST_TOOL_NAMESPACE, TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, TRACE_SCHEMA_VERSION, TokenCounter, ToolTraceMissingError, TraceAnalysisLimitError, TraceAnalysisStoreContractError, TraceAnalysisValidationError, TraceContractBuilder, TraceEmitter, TraceFileMalformedError, TraceFileMissingError, TraceFileTooLargeError, TraceNotFoundError, UI_FINDING_SEVERITIES, UI_LENSES, UNIVERSAL_FINDERS, VERIFICATION_STRATEGIES, VERIFICATION_STRATEGY_SOURCES, ValidationError, VerificationError, WILCOXON_EXACT_MAX_N, WORKER_DRIVER_DOCTRINE, acquisitionPlansForKnowledgeGaps, agentProfileCellHashMaterial, agentProfileCellKey, agentProfileHash, agentProfileId, agentProfileModelId, agentVisibleFields, aggregateJudgeVerdicts, aggregateLlm, aggregatePrReviewScore, aggregateRunScore, allCriticalPassed, analystFindingDigest, analystRunDigest, analystRunToFeedbackTrajectory, analystRunToReviewRequests, analyzeAntiSlop, analyzeSeries, analyzeSupervisorRun, analyzeSupervisorRunIntegrity, analyzeSupervisorRunSources, analyzeTraces, appendScorecard, applyLlmSpanOtlpAttributes, applyToolSpanOtlpAttributes, argHash, asNumber, asString, assertCapabilityHeadroom, assertCrossFamily, assertCrossFamilyServed, assertExactRegistryRunOpts, assertLlmRoute, assertMinted, assertMintedLines, assertModelsServed, assertNoHiddenLeak, assertProductBenchmarkRun, assertRealAgentReceipts, assertRealBackend, assertReleaseConfidence, assertRolloutLine, assertRunAgentProfileCell, assertRunCaptured, assertServedModel, assertServedModels, assertSingleBackend, assignFeedbackSplit, assignHeldOutTag, attachCostToReport, attest, attributeCounterfactuals, backoffMs, deterministicSplit as benchmarkDeterministicSplit, benchmarks_exports as benchmarks, benjaminiHochberg, bisect, blendHeldout, blockingKnowledgeEval, bonferroni, bootstrapCi, buildAgentInterfaceProfileCell, buildAgentProfileCell, buildDefaultAnalystRegistry, buildDriverSystemPrompt, buildEquivalenceRecord, buildProductBenchmarkManifest, buildReflectionPrompt, buildReviewerPrompt, buildTraceAnalysisToolDescriptors, buildTraceInsightContext, buildTraceInsightPrompt, buildTrajectory, byteLengthRange, cachedJudge, calibrateJudge, calibrateJudgeContinuous, callLlm, callLlmJson, canaryLeakView, canonicalJson, canonicalize, capabilityHeadroom, captureFetchToRawSink, causalAttribution, checkBehavioralCanary, checkCanaries, checkServedModel, checkSlos, checkTraceContracts, clamp01, classifyFailure, classifyOtlpSpanRole, classifyTreatment, claudeCodeSupervisorRunReader, cliffsDelta, clusteredPairedBinary, cohensD, collectionPreserved, commentsForSource, commitBisect, comparePairedArms, compareReferenceReplay, compareToBaseline, compilerJudge, completionVerdict, composeParsers, composeValidators, computeExperimentStats, computeFindingId, computeToolUseMetrics, computeTraceMetrics, confidenceInterval, containsAll, contentHash, contextInputTokens, continuousAgreement, contractJudge, controlFailureClassFromVerification, controlRunToFeedbackTrajectory, controlRunToRunRecord, convertTraceStoresToOtlp, corpusInterRaterAgreement, corpusInterRaterAgreementFromJudgeScores, costForTokenPricing, costForUsage, costReceiptFromLlm, costReceiptFromLlmError, costReport, createAntiSlopJudge, createBoundedTraceAnalysisStore, createChatClient, createDefaultReviewer, createDspyRlmTraceEngine, createFeedbackTrajectory, createIntentMatchJudge, createLlmCorrectnessChecker, createLlmReviewer, createOtelExporter, createOtelTracingStore, createReferenceEquivalenceJudge, createReplayFetch, createSandboxPool, createSemanticConceptJudge, createTokenRecallChecker, createTraceAnalyst, crossTraceDiff, crowdingDistance, dataDescriptionBits, decideNextUserTurn, decidePairedPromotion, decideReferenceReplayPromotion, decideReferenceReplayRunPromotion, defaultBlendWeights, defaultIsMaterial, defaultProviderRedactor, defaultReferenceReplayMatcher, defaultTraceInsightPanel, defineEquivalenceCheck, defineTraceAnalyst, deployGateLayer, describeTraceInsightScope, diffFindings, diffScorecard, discoverPersonas, distillPlaybook, domainEvidencePattern, dominates, eProcess, emitControlIntegrityFindings, ensembleJudge, errorStreakDetector, estimateCost, estimateTokens, evaluateActionPolicy, evaluateContract, evaluateHypothesis, evaluateInterimReleaseConfidence, evaluateOracles, evaluateReleaseConfidence, evaluateTraceContract, executeScenario, expandProfileAxes, expectAgent, exportProductBenchmark, exportProductBenchmarkRuns, exportRewardModel, exportRunAsOtlp, extractAssetUrls, extractErrorCount, extractOtlpAttributes, extractProducedState, extractUsage, extractUsageFromResponse, extractUsageFromSse, feedbackTrajectoriesToDatasetScenarios, feedbackTrajectoriesToOptimizerRows, feedbackTrajectoryToDatasetScenario, feedbackTrajectoryToOptimizerRow, fileContains, fileExists, fileExperimentStore, fileVerdictCache, findAutoMatchNoExpectation, findConstructorCwdDropped, findFallbackToPass, findLiteralTruePass, findProductBenchmarkArtifacts, findSkipCountsAsPass, firstNumberAttr, firstStringAttr, flattenOtlpExportToNdjson, flowLayer, fnv1a32, formatBenchmarkReport, formatDriverReport, formatFindings, formatScorecardDiff, fromHarborTrajectory, gainHistogram, gateTreatmentApplied, gateTreatmentFromMetrics, gateTreatmentFromSpans, gateTreatmentFromToolSpans, ghCliClient, gitProvenanceReader, precision as goldenPrecision, gradeOnHidden, gradeSemanticStatus, groupBy, groupRunsByAgentProfileCell, harnessAxisOf, hasCapturedToolArgs, hashContent, hashJson, hashScenarios, hashToUnit, hiddenGrade, holm, htmlContainsElement, httpGithubClient, improvementVerdict, inMemoryExperimentStore, inMemoryReferenceReplayStore, inMemoryReviewStore, inMemoryRunRecordBackend, inMemoryVerdictCache, inferDomainKeywords, inferOtlpKind, interRaterReliability, interpretCliffs, iqr, isBinaryOutcomeVector, isHiddenDestination, isJudgeSpan, isLlmSpan, isModelPriced, isOtelConfigured, isOtlpModelCall, isRealnessGated, isRetrievalSpan, isRolloutLine, isRunRecord, isRuntimeSupervisorRunDir, isSandboxSpan, isToolSpan, isTrainableSplit, isTransientLlmError, isUnavailable, iterateRawCalls, jestTestParser, jsonHasKeys, jsonShape, jsonlReferenceReplayStore, jsonlReviewStore, jsonlRunRecordBackend, judgeFamily, judgeReplayGate, judgeSpans, keyPreserved, knowledgeReadinessTracePayload, leaderboard, linterJudge, llmJudge, llmSpanFromProvider, llmSpans, loadScorecard, loadScorerFromGrader, localCommandRunner, lowercaseMutator, makeEvalTools, makeFinding, makeProposalFinding, mannWhitneyU, mapConcurrent, matchGoldens, matchSpan, maximumChargeForLlmRequest, mcnemar, mcnemarPower, mcnemarRequiredN, mergeLayerResults, mergeSteeringBundle, minimumPairsForPairedDeltaTest, mintRolloutRows, modelDescriptionBits, modelHasSnapshot, modelPriceKey, mulberry32, multiToolchainLayer, noProgressDetector, normalCdf, normalizeModelId, normalizeScores, notBlocked, objectiveEval, observeAll, observedScore, observedSplitScore, otelRunCompleteHook, otlpRowsToRunRecords, otlpRowsToTraceRunRecords, otlpTextToTraceAnalysisStore, otlpToRunRecords, otlpToTraceRunRecords, pairArms, pairRunRecords, pairedBinaryScale, pairedBootstrap, pairedCohensDz, pairedDecisionShape, pairedDeltaTest, pairedDeltaTieFraction, pairedEvalueSequence, pairedMde, pairedRiskDifference, pairedRiskDifferenceExact, pairedRiskDifferenceScore, pairedSignTest, pairedTTest, paraphraseRobustness, paraphraseRobustnessScenarios, paretoChart, paretoFrontier, paretoFrontierWithCrowding, parseCorrectnessResponse, parseFeedbackTrajectoriesJsonl, parseReflectionResponse, parseRunRecordSafe, parseRuntimeTrajectoryHookEvent, partialCredit, partitionHeldOut, passAtK, passOrthogonality, pearsonR, pixelDeltaRatio, planTraceInsightQuestions, politenessPrefixMutator, positionalBias, preflightModels, printDriverSummary, probeLlm, productBenchmarkIntegrityFailures, productBenchmarkMutableSurfaces, productBenchmarkRepoIdentity, productBenchmarkSplits, profile_exports as profile, projectOtlpFlatLine, projectRuntimeTrajectoryEvidence, promptBisect, proposeSynthesisTargets, providerFromBaseUrl, pytestTestParser, ranks, readClaudeCodeSupervisorRun, readOtlpStatus, readProductBenchmarkManifest, readProductBenchmarkRecords, readRuntimeSupervisorRun, recordRuns, recordRunsToScorecard, redTeamDataset, redTeamReport, redactString, redactValue, referenceReplayRunsToSteeringRows, referenceReplayScenarioToRunScore, regexMatch, regexMatches, relabelImportedSplit, renderMarkdownReport, renderPlaybookMarkdown, renderPreferenceMemoryMarkdown, renderPriorFindings, renderReleaseReport, renderSteeringText, renderSupervisorRunHeadline, renderSupervisorRunMarkdown, renderUpstreamFindings, repeatedActionDetector, replayFeedbackTrajectories, replayFeedbackTrajectory, replayScorerOverCorpus, replayTraceThroughJudge, requireAgentProfileCell, requiredPairedSampleSize, requiredSampleSize, researchReport, resetDeprecationWarnings, resolveModelPricing, resolveSeat, resolveTraceAnalystLimits, rollupSupervisorRuns, roundTripRunRecord, routeFields, rowCount, rowWhere, runAgentControlLoop, runAssertions, runBehavioralCanaries, runCanaries, runCounterfactual, runE2EWorkflow, runEquivalenceCheck, runEvalCampaign, runExpectations, runFailureClass, runHarnessExperiment, runIntentMatchJudge, runJudgeFleet, runKeywordCoverageJudge, runKeywordCoverageJudgeUrl, runLiveProof, runProposeReview, runProposeReviewAsControlLoop, runRecordToProductBenchmarkRecord, runReferenceEquivalenceJudge, runReferenceReplay, runScore, runSelfPlay, runSemanticConceptJudge, runTaskScore, runTestGradedScenario, runTraceAnalyst, runsForScenario, runtimeSupervisorRunReader, scalarScore, scanForMuffledGates, scoreContinuity, scoreFromEvals, scoreKnowledgeReadiness, scoreOrigin, scorePrReviewComments, scorePrReviewSource, scoreRedTeamOutput, scoreReferenceReplay, scoreTraceInsightReadiness, seatPresets, securityJudge, selectHarnessVariant, selfPreference, sentenceReorderMutator, serializeFeedbackTrajectoriesJsonl, servedModelAcceptable, showMeasured, signManifest, spearmanR, statusAdvanced, stopOnNoProgress, stopOnRepeatedAction, stringField, stripFencedJson, studentTCdf, subjectiveEval, summarizeAgentReceiptIntegrity, summarizeBackendIntegrity, summarizeHarnessResults, summarizePrReviewBenchmark, summarizePreferenceMemory, summaryTable, supervisorRunRolloutLines, testJudge, textInSnapshot, throwIfRunIncomplete, toAgentProfileJson, toHarborTrajectories, toHarborTrajectory, toJsonl, toLangfuseEnvelope, toOpenAiTool, toPrometheusText, toRewardRows, toSftRows, tokenizeDomainWords, toolNamesForRun, toolSpans, toolSpansToTraceAnalysisStore, traceAnalystFunctionGroup, traceAnalystOnRunComplete, traceContract, traceJudge, traceJudgeEnsemble, traceSpanKindToOpenInferenceKind, tracedAnalyzeTraces, trainingReward, trainingScore, typoMutator, unmintableReasons, urlContains, userQuestionsForKnowledgeGaps, validateAgentProfileCell, validateProductBenchmarkManifest, validateProductBenchmarkRecord, validateProductBenchmarkRun, validateRolloutLine, validateRunRecord, verbosityBias, verifyAgentProfileCell, verifyAttestation, verifyCompletion, verifyManifest, visualDiff, viteDeployRunner, vitestTestParser, warnDeprecatedOnce, weightedComposite, weightedMean, weightedRecall, welchsTTest, whitespaceCollapseMutator, wilcoxonSignedRank, wilson, withAssignedFeedbackSplit, withHeldoutBlend, withJudgeRetry, withOtelPipeline, wranglerDeployRunner, writeSupervisorRunReport };
|
|
12404
11988
|
|
|
12405
11989
|
//# sourceMappingURL=index.js.map
|