@tangle-network/agent-eval 0.180.0 → 0.181.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +60 -0
- package/README.md +119 -159
- package/dist/adapters/http.d.ts +2 -2
- package/dist/{agent-profile-B7yErX0q.d.ts → agent-profile-CivaSsSy.d.ts} +4 -4
- package/dist/{agent-profile-B7yErX0q.d.ts.map → agent-profile-CivaSsSy.d.ts.map} +1 -1
- package/dist/{agent-profile-cell-0gSi5ffD.js → agent-profile-cell-Cv6UA-W_.js} +20 -57
- package/dist/agent-profile-cell-Cv6UA-W_.js.map +1 -0
- package/dist/{agent-profile-cell-CTOZJUuE.d.ts → agent-profile-cell-s__adRnK.d.ts} +3 -3
- package/dist/agent-profile-cell-s__adRnK.d.ts.map +1 -0
- package/dist/analyst/index.d.ts +10 -10
- package/dist/analyst/index.js +3 -3
- package/dist/ast-CP9ae9B0.js +557 -0
- package/dist/ast-CP9ae9B0.js.map +1 -0
- package/dist/ast-hI-vjW6J.d.ts +457 -0
- package/dist/ast-hI-vjW6J.d.ts.map +1 -0
- package/dist/{benchmark-command-D4vpnAdO.js → benchmark-command-B57n9vjz.js} +7 -6
- package/dist/{benchmark-command-D4vpnAdO.js.map → benchmark-command-B57n9vjz.js.map} +1 -1
- package/dist/benchmarks/index.d.ts +4 -4
- package/dist/benchmarks/index.js +3 -3
- package/dist/campaign/index.d.ts +6 -6
- package/dist/campaign/index.js +8 -8
- package/dist/{campaign-BGEurASO.js → campaign-4_ppJW5X.js} +12 -12
- package/dist/{campaign-BGEurASO.js.map → campaign-4_ppJW5X.js.map} +1 -1
- package/dist/{campaign-evidence-D8DBLqLI.js → campaign-evidence-B8oF9xQ6.js} +515 -471
- package/dist/campaign-evidence-B8oF9xQ6.js.map +1 -0
- package/dist/cli.js +4 -7
- package/dist/cli.js.map +1 -1
- package/dist/{client-BlLY6o2w.js → client-CXE-U1SA.js} +3 -1
- package/dist/client-CXE-U1SA.js.map +1 -0
- package/dist/{client-CuQgX33c.d.ts → client-kh2jOjTK.d.ts} +4 -4
- package/dist/{client-CuQgX33c.d.ts.map → client-kh2jOjTK.d.ts.map} +1 -1
- package/dist/contract/index.d.ts +13 -13
- package/dist/contract/index.js +10 -9
- package/dist/contract/index.js.map +1 -1
- package/dist/{default-registry-IGDE9XIC.d.ts → default-registry-BwDSWVzg.d.ts} +6 -6
- package/dist/{default-registry-IGDE9XIC.d.ts.map → default-registry-BwDSWVzg.d.ts.map} +1 -1
- package/dist/{define-agent-eval-Cx4Ls9ta.d.ts → define-agent-eval-CwOWWQt_.d.ts} +33 -12
- package/dist/define-agent-eval-CwOWWQt_.d.ts.map +1 -0
- package/dist/{define-agent-eval-Dzidv34q.js → define-agent-eval-Ddu33JH9.js} +134 -67
- package/dist/define-agent-eval-Ddu33JH9.js.map +1 -0
- package/dist/{dspy-rlm-engine-xKiWmj_G.js → dspy-rlm-engine-S53V0HhE.js} +2 -2
- package/dist/{dspy-rlm-engine-xKiWmj_G.js.map → dspy-rlm-engine-S53V0HhE.js.map} +1 -1
- package/dist/{engine-CX8ReXkn.d.ts → engine-DS1cysJy.d.ts} +10 -7
- package/dist/engine-DS1cysJy.d.ts.map +1 -0
- package/dist/{eval-campaign-Cs-7MiCs.js → eval-campaign-aYdtjtJR.js} +4 -4
- package/dist/{eval-campaign-Cs-7MiCs.js.map → eval-campaign-aYdtjtJR.js.map} +1 -1
- package/dist/{exact-types-B7LC1EyX.d.ts → exact-types-BZDe0W2D.d.ts} +2 -2
- package/dist/{exact-types-B7LC1EyX.d.ts.map → exact-types-BZDe0W2D.d.ts.map} +1 -1
- package/dist/experiment/index.d.ts +27 -477
- package/dist/experiment/index.d.ts.map +1 -1
- package/dist/experiment/index.js +95 -559
- package/dist/experiment/index.js.map +1 -1
- package/dist/{experiment-tracker-B3TiF5-u.d.ts → experiment-tracker-C7PfnF4b.d.ts} +2 -2
- package/dist/{experiment-tracker-B3TiF5-u.d.ts.map → experiment-tracker-C7PfnF4b.d.ts.map} +1 -1
- package/dist/{external-optimizer-process-Dlz8YxrT.js → external-optimizer-process-QDRURJAM.js} +3 -3
- package/dist/{external-optimizer-process-Dlz8YxrT.js.map → external-optimizer-process-QDRURJAM.js.map} +1 -1
- package/dist/{external-optimizer-subprocess-q3VzlGAO.js → external-optimizer-subprocess-D4dzUBZI.js} +3 -2
- package/dist/{external-optimizer-subprocess-q3VzlGAO.js.map → external-optimizer-subprocess-D4dzUBZI.js.map} +1 -1
- package/dist/{feedback-trajectory-eHWNv5Aj.d.ts → feedback-trajectory-CXmtITBo.d.ts} +3 -3
- package/dist/{feedback-trajectory-eHWNv5Aj.d.ts.map → feedback-trajectory-CXmtITBo.d.ts.map} +1 -1
- package/dist/hosted/index.d.ts +2 -2
- package/dist/hosted/index.d.ts.map +1 -1
- package/dist/hosted/index.js +1 -1
- package/dist/{index-BxWvILU8.d.ts → index-Bp_6sj3x.d.ts} +109 -56
- package/dist/index-Bp_6sj3x.d.ts.map +1 -0
- package/dist/{index-e7LXeRVa.d.ts → index-CJ3LhKIX.d.ts} +2 -2
- package/dist/{index-e7LXeRVa.d.ts.map → index-CJ3LhKIX.d.ts.map} +1 -1
- package/dist/{index-CiUjjEIa.d.ts → index-DNntP4ch.d.ts} +7 -7
- package/dist/{index-CiUjjEIa.d.ts.map → index-DNntP4ch.d.ts.map} +1 -1
- package/dist/{index-DxNYmx4a.d.ts → index-DoykkxW0.d.ts} +11 -11
- package/dist/{index-DxNYmx4a.d.ts.map → index-DoykkxW0.d.ts.map} +1 -1
- package/dist/index.d.ts +28 -28
- package/dist/index.js +24 -15
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-DETqPc_A.d.ts → insight-report-D1qa0HWs.d.ts} +9 -5
- package/dist/{insight-report-DETqPc_A.d.ts.map → insight-report-D1qa0HWs.d.ts.map} +1 -1
- package/dist/{integrity-BKTcA-HP.d.ts → integrity-rGOfSUle.d.ts} +2 -2
- package/dist/{integrity-BKTcA-HP.d.ts.map → integrity-rGOfSUle.d.ts.map} +1 -1
- package/dist/{ledger-core-Cs9f7385.js → journal-Cs9f7385.js} +1 -1
- package/dist/journal-Cs9f7385.js.map +1 -0
- package/dist/{judge-calibration-C5CbMYce.d.ts → judge-calibration-DFtEMlde.d.ts} +31 -2
- package/dist/judge-calibration-DFtEMlde.d.ts.map +1 -0
- package/dist/{judge-calibration-BnpVKtnb.js → judge-calibration-DYmaBtJr.js} +48 -2
- package/dist/{judge-calibration-BnpVKtnb.js.map → judge-calibration-DYmaBtJr.js.map} +1 -1
- package/dist/ledger-core/index.d.ts +1 -1
- package/dist/ledger-core/index.js +1 -1
- package/dist/{llm-judge-v80Kmu9g.js → llm-judge-DEFZeSiu.js} +645 -456
- package/dist/llm-judge-DEFZeSiu.js.map +1 -0
- package/dist/{matrix-DeMmnWrP.d.ts → matrix-CyhW-vgJ.d.ts} +2 -2
- package/dist/{matrix-DeMmnWrP.d.ts.map → matrix-CyhW-vgJ.d.ts.map} +1 -1
- package/dist/meta-eval/index.d.ts +138 -7
- package/dist/meta-eval/index.d.ts.map +1 -1
- package/dist/meta-eval/index.js +245 -97
- package/dist/meta-eval/index.js.map +1 -1
- package/dist/{mint-Cc1_zwRQ.js → mint-ySIIkKlV.js} +2 -2
- package/dist/{mint-Cc1_zwRQ.js.map → mint-ySIIkKlV.js.map} +1 -1
- package/dist/multishot/golden/index.d.ts +1 -1
- package/dist/multishot/index.d.ts +2 -2
- package/dist/openapi.json +1 -1
- package/dist/outcome-store-BXlkwMPR.js +131 -0
- package/dist/outcome-store-BXlkwMPR.js.map +1 -0
- package/dist/{outcome-store-BYHIuO0e.d.ts → outcome-store-CNt4iZ67.d.ts} +18 -25
- package/dist/outcome-store-CNt4iZ67.d.ts.map +1 -0
- package/dist/{paired-promotion-decision-CGzg0cI_.d.ts → paired-promotion-decision-DPsMQm-0.d.ts} +13 -7
- package/dist/{paired-promotion-decision-CGzg0cI_.d.ts.map → paired-promotion-decision-DPsMQm-0.d.ts.map} +1 -1
- package/dist/pipelines/index.js +1 -1
- package/dist/{produced-state-Cv0kJJuP.js → produced-state-BHboMaab.js} +3 -3
- package/dist/{produced-state-Cv0kJJuP.js.map → produced-state-BHboMaab.js.map} +1 -1
- package/dist/profile-cell.d.ts +1 -1
- package/dist/profile-cell.js +1 -1
- package/dist/{promotion-policy-DWOm70gx.js → promotion-policy-CDMMxzb6.js} +28 -40
- package/dist/promotion-policy-CDMMxzb6.js.map +1 -0
- package/dist/{registry-ByVld1-5.d.ts → registry-BRbB6Y0v.d.ts} +4 -4
- package/dist/{registry-ByVld1-5.d.ts.map → registry-BRbB6Y0v.d.ts.map} +1 -1
- package/dist/{release-confidence-BAcNYOf1.d.ts → release-confidence-BcqeQTHW.d.ts} +3 -3
- package/dist/{release-confidence-BAcNYOf1.d.ts.map → release-confidence-BcqeQTHW.d.ts.map} +1 -1
- package/dist/{release-confidence-BcGCclTB.js → release-confidence-DMg8n18l.js} +2 -2
- package/dist/{release-confidence-BcGCclTB.js.map → release-confidence-DMg8n18l.js.map} +1 -1
- package/dist/reporting.d.ts +4 -4
- package/dist/reporting.js +3 -3
- package/dist/{researcher-jsW1X94L.d.ts → researcher-64T49THL.d.ts} +6 -6
- package/dist/{researcher-jsW1X94L.d.ts.map → researcher-64T49THL.d.ts.map} +1 -1
- package/dist/{reward-hacking-ZXEi9VCq.d.ts → reward-hacking-uzO_ihep.d.ts} +2 -2
- package/dist/{reward-hacking-ZXEi9VCq.d.ts.map → reward-hacking-uzO_ihep.d.ts.map} +1 -1
- package/dist/rl.d.ts +53 -99
- package/dist/rl.d.ts.map +1 -1
- package/dist/rl.js +182 -169
- package/dist/rl.js.map +1 -1
- package/dist/rollout/index.d.ts +1 -1
- package/dist/rollout/index.js +2 -2
- package/dist/{rollout-DmoJVqrF.js → rollout-B-UF5R6w.js} +2 -2
- package/dist/{rollout-DmoJVqrF.js.map → rollout-B-UF5R6w.js.map} +1 -1
- package/dist/rubric-predictive-validity-Bmj2_cll.d.ts +79 -0
- package/dist/rubric-predictive-validity-Bmj2_cll.d.ts.map +1 -0
- package/dist/rubric-predictive-validity-CCK-1B7w.js +178 -0
- package/dist/rubric-predictive-validity-CCK-1B7w.js.map +1 -0
- package/dist/{run-record-DTv1MdjK.d.ts → run-record-BiTWauyO.d.ts} +2 -2
- package/dist/{run-record-DTv1MdjK.d.ts.map → run-record-BiTWauyO.d.ts.map} +1 -1
- package/dist/run-record-Br-Yzt_k.js +464 -0
- package/dist/run-record-Br-Yzt_k.js.map +1 -0
- package/dist/{run-record-DQpSf7t-.js → run-record-DualPTn2.js} +2 -2
- package/dist/{run-record-DQpSf7t-.js.map → run-record-DualPTn2.js.map} +1 -1
- package/dist/{semantic-concept-judge-Bi6_iGqg.js → semantic-concept-judge-Bm5JDEKO.js} +3 -3
- package/dist/{semantic-concept-judge-Bi6_iGqg.js.map → semantic-concept-judge-Bm5JDEKO.js.map} +1 -1
- package/dist/{sequential-B5gXgcyp.js → sequential-DAsyV2T9.js} +42 -25
- package/dist/sequential-DAsyV2T9.js.map +1 -0
- package/dist/{series-convergence-DeG33RpC.d.ts → series-convergence-BnMs_uAr.d.ts} +3 -3
- package/dist/{series-convergence-DeG33RpC.d.ts.map → series-convergence-BnMs_uAr.d.ts.map} +1 -1
- package/dist/{skillopt-optimization-method-C3oYul8v.js → skillopt-optimization-method-CL_0aArC.js} +5 -5
- package/dist/{skillopt-optimization-method-C3oYul8v.js.map → skillopt-optimization-method-CL_0aArC.js.map} +1 -1
- package/dist/{statistical-heldout-0La5ZTlv.d.ts → statistical-heldout-CpVd6FmY.d.ts} +207 -144
- package/dist/statistical-heldout-CpVd6FmY.d.ts.map +1 -0
- package/dist/{store-tool-spans-4J1EDElP.d.ts → store-tool-spans-Dt-YdAuE.d.ts} +6 -6
- package/dist/{store-tool-spans-4J1EDElP.d.ts.map → store-tool-spans-Dt-YdAuE.d.ts.map} +1 -1
- package/dist/{summary-report-gMrbYawB.d.ts → summary-report-D1h4dlrK.d.ts} +3 -3
- package/dist/{summary-report-gMrbYawB.d.ts.map → summary-report-D1h4dlrK.d.ts.map} +1 -1
- package/dist/{summary-report-B16xy9Kd.js → summary-report-e-MaOAHV.js} +2 -2
- package/dist/{summary-report-B16xy9Kd.js.map → summary-report-e-MaOAHV.js.map} +1 -1
- package/dist/{tool-groups-2QA0S7dK.d.ts → tool-groups-B2bSNaJB.d.ts} +3 -3
- package/dist/tool-groups-B2bSNaJB.d.ts.map +1 -0
- package/dist/{tool-waste-B9tdWV6g.js → tool-waste-C7MU9u1e.js} +2 -2
- package/dist/{tool-waste-B9tdWV6g.js.map → tool-waste-C7MU9u1e.js.map} +1 -1
- package/dist/trace-repair/index.d.ts +2 -2
- package/dist/traces.d.ts +6 -6
- package/dist/traces.js +1 -1
- package/dist/{types-BmlkCrg0.d.ts → types-BvZoPTGa.d.ts} +3 -3
- package/dist/{types-BmlkCrg0.d.ts.map → types-BvZoPTGa.d.ts.map} +1 -1
- package/dist/{types-gvRsyJLh.d.ts → types-CBbLtr2J.d.ts} +38 -3
- package/dist/{types-gvRsyJLh.d.ts.map → types-CBbLtr2J.d.ts.map} +1 -1
- package/dist/{types-C34V4Vto.d.ts → types-CS0qk_Yp.d.ts} +4 -4
- package/dist/{types-C34V4Vto.d.ts.map → types-CS0qk_Yp.d.ts.map} +1 -1
- package/dist/{types-DzuaM493.d.ts → types-D7gEdPoQ.d.ts} +3 -3
- package/dist/{types-DzuaM493.d.ts.map → types-D7gEdPoQ.d.ts.map} +1 -1
- package/dist/wire/index.d.ts +2 -2
- package/docs/adapters-observability.md +14 -0
- package/docs/campaign-proposers.md +86 -128
- package/docs/charter.md +108 -112
- package/docs/concepts.md +157 -69
- package/docs/design/mlbenchmarks-book-review.md +440 -0
- package/docs/design/mlbenchmarks-review/observations.json +713 -0
- package/docs/design/mlbenchmarks-review/probes.mts +476 -0
- package/docs/design/mlbenchmarks-review/sources.json +200 -0
- package/docs/design/self-improvement-evidence-audit.md +263 -0
- package/docs/design.md +2 -1
- package/docs/eval-surface-map.md +95 -42
- package/docs/evaluation-integrity.md +220 -0
- package/docs/experiment.md +111 -55
- package/docs/feature-guide.md +5 -6
- package/docs/hosted-ingest-spec.md +4 -11
- package/docs/insight-report.md +187 -455
- package/docs/outcome-validity.md +182 -0
- package/docs/product-eval-adoption.md +1 -2
- package/docs/research-report-methodology.md +7 -7
- package/docs/search-history-receipts.md +8 -0
- package/docs/statistical-evidence.md +129 -0
- package/docs/verdicts.md +76 -49
- package/package.json +1 -1
- package/dist/agent-profile-cell-0gSi5ffD.js.map +0 -1
- package/dist/agent-profile-cell-CTOZJUuE.d.ts.map +0 -1
- package/dist/campaign-evidence-D8DBLqLI.js.map +0 -1
- package/dist/client-BlLY6o2w.js.map +0 -1
- package/dist/define-agent-eval-Cx4Ls9ta.d.ts.map +0 -1
- package/dist/define-agent-eval-Dzidv34q.js.map +0 -1
- package/dist/engine-CX8ReXkn.d.ts.map +0 -1
- package/dist/index-BxWvILU8.d.ts.map +0 -1
- package/dist/judge-calibration-C5CbMYce.d.ts.map +0 -1
- package/dist/ledger-core-Cs9f7385.js.map +0 -1
- package/dist/llm-judge-v80Kmu9g.js.map +0 -1
- package/dist/outcome-store-BYHIuO0e.d.ts.map +0 -1
- package/dist/outcome-store-ChBKlTd_.js +0 -75
- package/dist/outcome-store-ChBKlTd_.js.map +0 -1
- package/dist/promotion-policy-DWOm70gx.js.map +0 -1
- package/dist/rubric-predictive-validity-2D5Gw9z9.js +0 -131
- package/dist/rubric-predictive-validity-2D5Gw9z9.js.map +0 -1
- package/dist/rubric-predictive-validity-Dl1dvKCv.d.ts +0 -75
- package/dist/rubric-predictive-validity-Dl1dvKCv.d.ts.map +0 -1
- package/dist/run-record-CR63CpHK.js +0 -216
- package/dist/run-record-CR63CpHK.js.map +0 -1
- package/dist/sequential-B5gXgcyp.js.map +0 -1
- package/dist/statistical-heldout-0La5ZTlv.d.ts.map +0 -1
- package/dist/tool-groups-2QA0S7dK.d.ts.map +0 -1
package/dist/rl.js
CHANGED
|
@@ -1,22 +1,22 @@
|
|
|
1
1
|
import { s as ValidationError } from "./errors-Dngq5h35.js";
|
|
2
2
|
import { i as compareCodeUnits } from "./canonical-DPyQ_rpt.js";
|
|
3
|
-
import {
|
|
3
|
+
import { a as medianInPlace } from "./internal-BMFSR8Ns.js";
|
|
4
4
|
import { t as mulberry32 } from "./random-Dn5fPWkt.js";
|
|
5
|
-
import { o as benjaminiHochberg } from "./power-and-mde-B8F2RdcD.js";
|
|
6
5
|
import { l as wilcoxonSignedRank } from "./paired-arms-D4aeIHUy.js";
|
|
6
|
+
import { r as pairedBootstrap } from "./paired-tests-C8iCsioC.js";
|
|
7
|
+
import { a as campaignCellToRunRecord, s as decidePairedPromotion } from "./run-record-Br-Yzt_k.js";
|
|
7
8
|
import { r as observedSplitScore, s as trainingScore } from "./reward-nw2xZGZG.js";
|
|
8
|
-
import { s as runTaskScore } from "./run-record-
|
|
9
|
-
import { a as campaignCellToRunRecord } from "./run-record-CR63CpHK.js";
|
|
9
|
+
import { s as runTaskScore } from "./run-record-DualPTn2.js";
|
|
10
10
|
import { i as filterDeterministicallyRewarded, n as extractVerifiableReward, r as extractVerifiableRewardsFromRecords, t as detectRewardHacking } from "./reward-hacking-D0XwhVWE.js";
|
|
11
11
|
import { n as InMemoryTraceStore } from "./store-DNe_Uv1Q.js";
|
|
12
|
-
import { t as runEvalCampaign } from "./eval-campaign-
|
|
12
|
+
import { t as runEvalCampaign } from "./eval-campaign-aYdtjtJR.js";
|
|
13
13
|
import { s as assertRewardGate } from "./schema-C1aaAxTf.js";
|
|
14
14
|
import { t as isSplitEligible } from "./exporters-Df7TgHFv.js";
|
|
15
|
-
import { t as mintRolloutRows } from "./mint-
|
|
15
|
+
import { t as mintRolloutRows } from "./mint-ySIIkKlV.js";
|
|
16
16
|
import { t as evaluateInterimReleaseConfidence } from "./sequential-CzK5DarL.js";
|
|
17
|
-
import { t as rubricPredictiveValidity } from "./rubric-predictive-validity-
|
|
17
|
+
import { n as assertUniqueObservationIds, s as validateOutcomeMetricSpecifications, t as rubricPredictiveValidity } from "./rubric-predictive-validity-CCK-1B7w.js";
|
|
18
18
|
import { n as thompsonCurriculum, r as varianceBasedCurriculum, t as observationsFromRunRecords } from "./active-curriculum-OjIWgrUJ.js";
|
|
19
|
-
import { n as InMemoryOutcomeStore, t as FileSystemOutcomeStore } from "./outcome-store-
|
|
19
|
+
import { n as InMemoryOutcomeStore, t as FileSystemOutcomeStore } from "./outcome-store-BXlkwMPR.js";
|
|
20
20
|
import { createHash } from "node:crypto";
|
|
21
21
|
import { appendFileSync, existsSync, mkdirSync, readFileSync } from "node:fs";
|
|
22
22
|
import { dirname, join } from "node:path";
|
|
@@ -24,29 +24,10 @@ import { dirname, join } from "node:path";
|
|
|
24
24
|
/**
|
|
25
25
|
* Sample-efficient adaptation evaluation.
|
|
26
26
|
*
|
|
27
|
-
*
|
|
28
|
-
*
|
|
29
|
-
*
|
|
30
|
-
*
|
|
31
|
-
* that needs 50. Standard meta-learning eval (Finn et al., MAML, RL² lit)
|
|
32
|
-
* reports an *adaptation curve*: score after k=0, 1, 2, 4, 8, 16, …
|
|
33
|
-
* in-context examples or fine-tune steps.
|
|
34
|
-
*
|
|
35
|
-
* This module ships:
|
|
36
|
-
*
|
|
37
|
-
* 1. `runAdaptationCurve` — given a runner that takes k demonstrations
|
|
38
|
-
* and returns a score, produce the (k, score) curve.
|
|
39
|
-
* 2. `compareAdaptationCurves` — paired comparison across two policies.
|
|
40
|
-
* Returns per-k delta with bootstrap CIs and an "area-under-curve"
|
|
41
|
-
* summary statistic.
|
|
42
|
-
* 3. `firstPassK` — for pass/fail evaluation, the minimum k at which
|
|
43
|
-
* the policy reliably passes (≥ pass-rate threshold over reps).
|
|
44
|
-
*
|
|
45
|
-
* Use cases:
|
|
46
|
-
* - Compare two prompt designs that have similar end-state performance
|
|
47
|
-
* but different in-context efficiency.
|
|
48
|
-
* - Decide between fine-tuning and prompting based on adaptation cost.
|
|
49
|
-
* - Detect when a policy "memorizes" k=0 inputs vs. genuinely adapts.
|
|
27
|
+
* An adaptation curve records scores after k demonstrations or training steps.
|
|
28
|
+
* Comparison pairs the same scenarios and resamples their whole curves.
|
|
29
|
+
* The normalized area summarizes performance over the observed k range.
|
|
30
|
+
* A first-pass k is descriptive and carries no separate reliability claim.
|
|
50
31
|
*/
|
|
51
32
|
async function runAdaptationCurve(opts) {
|
|
52
33
|
const ks = opts.ks ?? [
|
|
@@ -59,6 +40,10 @@ async function runAdaptationCurve(opts) {
|
|
|
59
40
|
];
|
|
60
41
|
const reps = opts.reps ?? 3;
|
|
61
42
|
const passThreshold = opts.passThreshold ?? .5;
|
|
43
|
+
assertKs(ks, "runAdaptationCurve");
|
|
44
|
+
assertScenarioIds(opts.scenarios, "runAdaptationCurve");
|
|
45
|
+
if (!Number.isInteger(reps) || reps < 1) throw new ValidationError("runAdaptationCurve: reps must be a positive integer");
|
|
46
|
+
if (!Number.isFinite(passThreshold) || passThreshold < 0 || passThreshold > 1) throw new ValidationError("runAdaptationCurve: passThreshold must be in [0,1]");
|
|
62
47
|
const sortedKs = [...ks].sort((a, b) => a - b);
|
|
63
48
|
const points = [];
|
|
64
49
|
for (const k of sortedKs) {
|
|
@@ -67,7 +52,7 @@ async function runAdaptationCurve(opts) {
|
|
|
67
52
|
let totalPasses = 0;
|
|
68
53
|
let totalAttempts = 0;
|
|
69
54
|
for (const scenario of opts.scenarios) {
|
|
70
|
-
const sid = scenario.scenarioId
|
|
55
|
+
const sid = scenario.scenarioId;
|
|
71
56
|
const scores = [];
|
|
72
57
|
let passes = 0;
|
|
73
58
|
for (let r = 0; r < reps; r++) {
|
|
@@ -76,6 +61,7 @@ async function runAdaptationCurve(opts) {
|
|
|
76
61
|
k,
|
|
77
62
|
rep: r
|
|
78
63
|
});
|
|
64
|
+
assertScore(score, `runAdaptationCurve: scenario '${sid}', k=${k}, rep=${r}`);
|
|
79
65
|
scores.push(score);
|
|
80
66
|
if (score >= passThreshold) passes++;
|
|
81
67
|
allScores.push(score);
|
|
@@ -118,69 +104,106 @@ async function runAdaptationCurve(opts) {
|
|
|
118
104
|
};
|
|
119
105
|
}
|
|
120
106
|
/**
|
|
121
|
-
*
|
|
122
|
-
*
|
|
123
|
-
*
|
|
107
|
+
* Compare identical scenario cohorts on identical k grids, paired by scenarioId.
|
|
108
|
+
* Missing pairs, duplicate identities, and cohort changes across k are refused.
|
|
109
|
+
* The bootstrap resamples whole scenarios, preserving dependence across k.
|
|
110
|
+
* Per-k intervals describe the curve; shared paired area decisions determine the verdict.
|
|
111
|
+
* Bootstrap eligibility is necessary but does not establish scenario independence.
|
|
124
112
|
*/
|
|
125
113
|
function compareAdaptationCurves(a, b, opts = {}) {
|
|
126
|
-
const
|
|
127
|
-
const resamples = opts.bootstrapResamples ??
|
|
128
|
-
const
|
|
129
|
-
|
|
130
|
-
|
|
131
|
-
|
|
132
|
-
|
|
133
|
-
|
|
134
|
-
|
|
135
|
-
|
|
136
|
-
|
|
137
|
-
|
|
138
|
-
|
|
139
|
-
|
|
140
|
-
|
|
141
|
-
|
|
142
|
-
|
|
143
|
-
|
|
144
|
-
|
|
145
|
-
|
|
146
|
-
|
|
147
|
-
|
|
148
|
-
|
|
114
|
+
const confidence = opts.confidence ?? .95;
|
|
115
|
+
const resamples = opts.bootstrapResamples ?? 2e3;
|
|
116
|
+
const minimumEffect = opts.minimumEffect ?? 0;
|
|
117
|
+
if (!Number.isFinite(confidence) || confidence <= 0 || confidence >= 1) throw new ValidationError("compareAdaptationCurves: confidence must be in (0,1)");
|
|
118
|
+
if (!Number.isInteger(resamples) || resamples < 1) throw new ValidationError("compareAdaptationCurves: bootstrapResamples must be a positive integer");
|
|
119
|
+
if (!Number.isFinite(minimumEffect) || minimumEffect < 0 || minimumEffect > 1) throw new ValidationError("compareAdaptationCurves: minimumEffect must be in [0,1]");
|
|
120
|
+
const aPoints = indexCurve(a, "A");
|
|
121
|
+
const bPoints = indexCurve(b, "B");
|
|
122
|
+
const ks = [...aPoints.keys()].sort((x, y) => x - y);
|
|
123
|
+
const missingInA = [...bPoints.keys()].filter((k) => !aPoints.has(k));
|
|
124
|
+
const missingInB = ks.filter((k) => !bPoints.has(k));
|
|
125
|
+
if (missingInA.length > 0 || missingInB.length > 0) throw new ValidationError(`compareAdaptationCurves: k grids differ; missing in A=[${missingInA}], missing in B=[${missingInB}]`);
|
|
126
|
+
const scenarioIds = [...aPoints.get(ks[0]).keys()].sort();
|
|
127
|
+
const expectedIds = new Set(scenarioIds);
|
|
128
|
+
for (const [arm, points] of [["A", aPoints], ["B", bPoints]]) for (const [k, cells] of points) {
|
|
129
|
+
const missing = scenarioIds.filter((id) => !cells.has(id));
|
|
130
|
+
const extra = [...cells.keys()].filter((id) => !expectedIds.has(id));
|
|
131
|
+
if (missing.length > 0 || extra.length > 0) throw new ValidationError(`compareAdaptationCurves: scenario pairs differ in ${arm} at k=${k}; missing=[${missing}], unexpected=[${extra}]`);
|
|
132
|
+
}
|
|
133
|
+
const bootstrapOptions = {
|
|
134
|
+
confidence,
|
|
135
|
+
resamples,
|
|
136
|
+
statistic: "mean",
|
|
137
|
+
seed: opts.seed
|
|
138
|
+
};
|
|
139
|
+
const perK = ks.map((k) => ({
|
|
140
|
+
k,
|
|
141
|
+
delta: pairedBootstrap(scenarioIds.map((id) => bPoints.get(k).get(id)), scenarioIds.map((id) => aPoints.get(k).get(id)), bootstrapOptions)
|
|
142
|
+
}));
|
|
143
|
+
const aAreas = scenarioIds.map((id) => scenarioArea(ks, aPoints, id));
|
|
144
|
+
const bAreas = scenarioIds.map((id) => scenarioArea(ks, bPoints, id));
|
|
145
|
+
const decisionOptions = {
|
|
146
|
+
...bootstrapOptions,
|
|
147
|
+
threshold: minimumEffect
|
|
148
|
+
};
|
|
149
|
+
const aImprovement = decidePairedPromotion(bAreas, aAreas, decisionOptions);
|
|
150
|
+
const bImprovement = decidePairedPromotion(aAreas, bAreas, decisionOptions);
|
|
151
|
+
const areaDelta = aImprovement.bootstrap ?? pairedBootstrap(bAreas, aAreas, bootstrapOptions);
|
|
149
152
|
let verdict;
|
|
150
|
-
if (
|
|
151
|
-
else if (
|
|
152
|
-
else if (
|
|
153
|
-
else verdict = "
|
|
154
|
-
const rationale = `
|
|
153
|
+
if (!aImprovement.sufficient || ks.length < 2) verdict = "insufficient_evidence";
|
|
154
|
+
else if (aImprovement.promote) verdict = "a_better";
|
|
155
|
+
else if (bImprovement.promote) verdict = "b_better";
|
|
156
|
+
else verdict = "inconclusive";
|
|
157
|
+
const rationale = `paired scenarios=${scenarioIds.length}, area delta=${areaDelta.mean.toFixed(3)}, ${confidence * 100}% ${aImprovement.statistic} interval=[${aImprovement.low.toFixed(3)}, ${aImprovement.high.toFixed(3)}], minimum effect=${minimumEffect}; ${verdict}`;
|
|
155
158
|
return {
|
|
156
159
|
perK,
|
|
157
160
|
areaDelta,
|
|
158
|
-
|
|
161
|
+
aImprovement,
|
|
162
|
+
bImprovement,
|
|
163
|
+
scenarioIds,
|
|
159
164
|
verdict,
|
|
160
165
|
rationale
|
|
161
166
|
};
|
|
162
167
|
}
|
|
163
|
-
/** First k
|
|
168
|
+
/** First observed k whose pass rate reaches the threshold; this is a descriptive summary. */
|
|
164
169
|
function firstPassK(curve, threshold = .5) {
|
|
165
170
|
return curve.points.find((p) => p.passRate >= threshold)?.k ?? null;
|
|
166
171
|
}
|
|
167
|
-
function
|
|
168
|
-
if (
|
|
169
|
-
|
|
170
|
-
|
|
171
|
-
|
|
172
|
-
|
|
173
|
-
|
|
174
|
-
|
|
175
|
-
|
|
176
|
-
|
|
172
|
+
function assertKs(ks, where) {
|
|
173
|
+
if (ks.length === 0 || ks.some((k) => !Number.isInteger(k) || k < 0)) throw new ValidationError(`${where}: ks must contain nonnegative integers`);
|
|
174
|
+
if (new Set(ks).size !== ks.length) throw new ValidationError(`${where}: duplicate k values`);
|
|
175
|
+
}
|
|
176
|
+
function assertScenarioIds(cells, where) {
|
|
177
|
+
if (cells.length === 0 || cells.some((cell) => !cell.scenarioId?.trim())) throw new ValidationError(`${where}: scenarios must have explicit nonempty scenarioId values`);
|
|
178
|
+
const seen = /* @__PURE__ */ new Set();
|
|
179
|
+
for (const { scenarioId } of cells) {
|
|
180
|
+
if (seen.has(scenarioId)) throw new ValidationError(`${where}: duplicate scenarioId '${scenarioId}'`);
|
|
181
|
+
seen.add(scenarioId);
|
|
182
|
+
}
|
|
183
|
+
}
|
|
184
|
+
function assertScore(score, where) {
|
|
185
|
+
if (!Number.isFinite(score) || score < 0 || score > 1) throw new ValidationError(`${where}: score must be finite and in [0,1], got ${score}`);
|
|
186
|
+
}
|
|
187
|
+
function indexCurve(curve, arm) {
|
|
188
|
+
const where = `compareAdaptationCurves: ${arm}`;
|
|
189
|
+
assertKs(curve.points.map((point) => point.k), where);
|
|
190
|
+
return new Map(curve.points.map((point) => {
|
|
191
|
+
assertScenarioIds(point.perScenario, `${where} at k=${point.k}`);
|
|
192
|
+
return [point.k, new Map(point.perScenario.map((cell) => {
|
|
193
|
+
assertScore(cell.meanScore, `${where}: '${cell.scenarioId}' at k=${point.k}`);
|
|
194
|
+
return [cell.scenarioId, cell.meanScore];
|
|
195
|
+
}))];
|
|
196
|
+
}));
|
|
197
|
+
}
|
|
198
|
+
function scenarioArea(ks, points, id) {
|
|
199
|
+
let area = 0;
|
|
200
|
+
for (let i = 1; i < ks.length; i++) {
|
|
201
|
+
const left = ks[i - 1];
|
|
202
|
+
const right = ks[i];
|
|
203
|
+
area += (points.get(left).get(id) + points.get(right).get(id)) * (right - left) / 2;
|
|
177
204
|
}
|
|
178
|
-
|
|
179
|
-
|
|
180
|
-
return {
|
|
181
|
-
low: samples[Math.floor(alpha / 2 * resamples)],
|
|
182
|
-
high: samples[Math.min(resamples - 1, Math.ceil((1 - alpha / 2) * resamples) - 1)]
|
|
183
|
-
};
|
|
205
|
+
const maxK = ks[ks.length - 1];
|
|
206
|
+
return maxK === 0 ? 0 : area / maxK;
|
|
184
207
|
}
|
|
185
208
|
//#endregion
|
|
186
209
|
//#region src/rl/compute-curves.ts
|
|
@@ -311,81 +334,58 @@ function fitLogSlope(points) {
|
|
|
311
334
|
/**
|
|
312
335
|
* Contamination probe — held-out perturbation tests.
|
|
313
336
|
*
|
|
314
|
-
*
|
|
315
|
-
*
|
|
316
|
-
*
|
|
317
|
-
*
|
|
318
|
-
*
|
|
319
|
-
* whether scores diverge significantly. Genuine capability transfers; rote
|
|
320
|
-
* memorization doesn't.
|
|
321
|
-
*
|
|
322
|
-
* This module ships the probe contract:
|
|
323
|
-
*
|
|
324
|
-
* 1. A `ScenarioPerturbation` strategy type — function that produces a
|
|
325
|
-
* perturbed scenario from an original.
|
|
326
|
-
* 2. `runContaminationProbe({ originals, perturbed, scoreFn })` — runs
|
|
327
|
-
* both halves and reports per-scenario score divergence + a global
|
|
328
|
-
* contamination verdict via paired Wilcoxon.
|
|
329
|
-
* 3. Several stock perturbations: `renameVariables`, `shuffleOrder`,
|
|
330
|
-
* `paraphrasePrompt`, `injectIrrelevantClause`. Each preserves the
|
|
331
|
-
* task's structural difficulty while breaking surface memorization.
|
|
332
|
-
*
|
|
333
|
-
* The verdict is conservative: if the perturbed-vs-original score
|
|
334
|
-
* difference is statistically significant (BH-adjusted p < 0.05) AND
|
|
335
|
-
* the median drop is > 5 percentage points, we flag *contamination
|
|
336
|
-
* suspected*. False positives are possible (the perturbation might
|
|
337
|
-
* actually be harder); the default is to flag for review, not to
|
|
338
|
-
* autoreject.
|
|
337
|
+
* Score each scenario and its perturbation, then test the paired differences.
|
|
338
|
+
* A significant global Wilcoxon result plus a worthwhile median drop flags
|
|
339
|
+
* contamination for review. Perturbations may change difficulty, so the result
|
|
340
|
+
* does not identify contamination as the cause. Per-item differences have no
|
|
341
|
+
* calibrated sampling null and carry no p-values or q-values.
|
|
339
342
|
*/
|
|
340
343
|
async function runContaminationProbe(input, opts = {}) {
|
|
341
|
-
const
|
|
344
|
+
const alpha = opts.alpha ?? .05;
|
|
342
345
|
const minMedianDrop = opts.minMedianDrop ?? .05;
|
|
343
346
|
const floor = opts.scoreFloor ?? 0;
|
|
347
|
+
if (!Number.isFinite(alpha) || alpha <= 0 || alpha >= 1) throw new ValidationError("runContaminationProbe: alpha must be in (0,1)");
|
|
348
|
+
for (const [name, value] of [["scoreFloor", floor], ["minMedianDrop", minMedianDrop]]) if (!Number.isFinite(value) || value < 0 || value > 1) throw new ValidationError(`runContaminationProbe: ${name} must be in [0,1]`);
|
|
349
|
+
const ids = input.originals.map(input.scenarioId);
|
|
350
|
+
if (ids.some((id) => !id?.trim()) || new Set(ids).size !== ids.length) throw new ValidationError("runContaminationProbe: original scenario IDs must be nonempty and unique");
|
|
344
351
|
if (!input.perturbed && !input.perturbation) throw new ValidationError("runContaminationProbe: must supply either `perturbed` or `perturbation`.");
|
|
345
352
|
const perturbed = input.perturbed ?? await Promise.all(input.originals.map((s) => input.perturbation.apply(s)));
|
|
346
353
|
if (perturbed.length !== input.originals.length) throw new ValidationError(`runContaminationProbe: perturbed length ${perturbed.length} ≠ originals ${input.originals.length}`);
|
|
347
354
|
const origScores = await Promise.all(input.originals.map((s) => input.scoreFn(s)));
|
|
348
355
|
const pertScores = await Promise.all(perturbed.map((s) => input.scoreFn(s)));
|
|
349
|
-
const
|
|
350
|
-
|
|
356
|
+
for (const score of [...origScores, ...pertScores]) if (!Number.isFinite(score) || score < 0 || score > 1) throw new ValidationError(`runContaminationProbe: scores must be finite and in [0,1], got ${score}`);
|
|
357
|
+
const perScenario = ids.map((scenarioId, i) => ({
|
|
358
|
+
scenarioId,
|
|
351
359
|
originalScore: origScores[i],
|
|
352
360
|
perturbedScore: pertScores[i],
|
|
353
|
-
delta: pertScores[i] - origScores[i]
|
|
354
|
-
qValue: NaN
|
|
361
|
+
delta: pertScores[i] - origScores[i]
|
|
355
362
|
}));
|
|
356
363
|
const valid = perScenario.filter((p) => p.originalScore >= floor && p.perturbedScore >= floor);
|
|
364
|
+
const excludedScenarioIds = perScenario.filter((p) => p.originalScore < floor || p.perturbedScore < floor).map((p) => p.scenarioId);
|
|
365
|
+
const deltas = valid.map((p) => p.delta);
|
|
366
|
+
const medianDelta = deltas.length === 0 ? null : medianInPlace(deltas);
|
|
367
|
+
const meanDelta = deltas.length === 0 ? null : deltas.reduce((sum, d) => sum + d, 0) / deltas.length;
|
|
357
368
|
if (valid.length < 4) return {
|
|
358
369
|
perScenario,
|
|
359
|
-
pairedTest:
|
|
360
|
-
|
|
361
|
-
|
|
362
|
-
},
|
|
363
|
-
medianDelta: 0,
|
|
364
|
-
meanDelta: 0,
|
|
370
|
+
pairedTest: null,
|
|
371
|
+
medianDelta,
|
|
372
|
+
meanDelta,
|
|
365
373
|
contaminationSuspected: false,
|
|
366
374
|
reason: `insufficient valid scenarios (n=${valid.length}, need ≥ 4)`,
|
|
367
|
-
n: valid.length
|
|
375
|
+
n: valid.length,
|
|
376
|
+
excludedScenarioIds
|
|
368
377
|
};
|
|
369
378
|
const pairedTest = wilcoxonSignedRank(valid.map((p) => p.originalScore), valid.map((p) => p.perturbedScore));
|
|
370
|
-
const
|
|
371
|
-
const sortedDeltas = [...deltas].sort((a, b) => a - b);
|
|
372
|
-
const median = sortedDeltas[Math.floor(sortedDeltas.length / 2)];
|
|
373
|
-
const mean = deltas.reduce((s, d) => s + d, 0) / deltas.length;
|
|
374
|
-
const { qValues } = benjaminiHochberg(valid.map((p) => Math.min(1, Math.max(1e-6, 1 - Math.abs(p.delta) / 1))), fdr);
|
|
375
|
-
for (let i = 0; i < valid.length; i++) {
|
|
376
|
-
const v = valid[i];
|
|
377
|
-
const idx = perScenario.findIndex((p) => p.scenarioId === v.scenarioId);
|
|
378
|
-
if (idx >= 0) perScenario[idx].qValue = qValues[i];
|
|
379
|
-
}
|
|
380
|
-
const contaminationSuspected = pairedTest.p < fdr && median <= -minMedianDrop;
|
|
379
|
+
const contaminationSuspected = pairedTest.p < alpha && medianDelta <= -minMedianDrop;
|
|
381
380
|
return {
|
|
382
381
|
perScenario,
|
|
383
382
|
pairedTest,
|
|
384
|
-
medianDelta
|
|
385
|
-
meanDelta
|
|
383
|
+
medianDelta,
|
|
384
|
+
meanDelta,
|
|
386
385
|
contaminationSuspected,
|
|
387
|
-
reason: contaminationSuspected ? `paired p=${pairedTest.p.toFixed(4)} < ${
|
|
388
|
-
n: valid.length
|
|
386
|
+
reason: contaminationSuspected ? `paired p=${pairedTest.p.toFixed(4)} < ${alpha} and median drop ${(-medianDelta).toFixed(4)} ≥ ${minMedianDrop}` : pairedTest.p >= alpha ? `no significant difference (paired p=${pairedTest.p.toFixed(4)})` : `significant but no qualifying drop (median delta ${medianDelta.toFixed(4)})`,
|
|
387
|
+
n: valid.length,
|
|
388
|
+
excludedScenarioIds
|
|
389
389
|
};
|
|
390
390
|
}
|
|
391
391
|
/**
|
|
@@ -1474,14 +1474,21 @@ function clamp(x, lo, hi) {
|
|
|
1474
1474
|
//#endregion
|
|
1475
1475
|
//#region src/rl/predictive-validity-researcher.ts
|
|
1476
1476
|
/**
|
|
1477
|
-
*
|
|
1478
|
-
*
|
|
1477
|
+
* Proposes rubric experiments against one declared outcome.
|
|
1478
|
+
* A correlation supports a hypothesis; the caller must measure any resulting change.
|
|
1479
1479
|
*/
|
|
1480
1480
|
var PredictiveValidityResearcher = class {
|
|
1481
1481
|
opts;
|
|
1482
1482
|
lastReport = null;
|
|
1483
1483
|
constructor(opts) {
|
|
1484
|
-
|
|
1484
|
+
validateOutcomeMetricSpecifications([opts.targetOutcome]);
|
|
1485
|
+
if (opts.rubrics !== void 0) assertUniqueObservationIds(opts.rubrics, "rubric");
|
|
1486
|
+
if (opts.failureThreshold !== void 0 && !Number.isFinite(opts.failureThreshold)) throw new Error("failureThreshold must be finite");
|
|
1487
|
+
this.opts = {
|
|
1488
|
+
...opts,
|
|
1489
|
+
targetOutcome: { ...opts.targetOutcome },
|
|
1490
|
+
rubrics: opts.rubrics === void 0 ? void 0 : [...opts.rubrics]
|
|
1491
|
+
};
|
|
1485
1492
|
}
|
|
1486
1493
|
async inspectFailures(runs) {
|
|
1487
1494
|
const threshold = this.opts.failureThreshold ?? .5;
|
|
@@ -1521,35 +1528,34 @@ var PredictiveValidityResearcher = class {
|
|
|
1521
1528
|
payload: { directive: "researcher.collect-more-outcomes" },
|
|
1522
1529
|
rationale: "predictive-validity researcher has no prior report; cannot recommend rubric reweighting until at least one report exists"
|
|
1523
1530
|
}];
|
|
1524
|
-
const decorativeThreshold = this.opts.decorativeThreshold ?? .4;
|
|
1525
1531
|
const changes = [];
|
|
1526
|
-
|
|
1527
|
-
|
|
1528
|
-
|
|
1529
|
-
|
|
1530
|
-
|
|
1531
|
-
|
|
1532
|
-
|
|
1533
|
-
|
|
1534
|
-
|
|
1535
|
-
|
|
1536
|
-
|
|
1537
|
-
|
|
1538
|
-
|
|
1539
|
-
|
|
1540
|
-
|
|
1541
|
-
for (const ranking of this.lastReport.ranked.slice(0, 1)) {
|
|
1542
|
-
if (ranking.verdict !== "load_bearing") continue;
|
|
1532
|
+
const target = { ...this.opts.targetOutcome };
|
|
1533
|
+
const pairs = this.lastReport.pairs.filter((pair) => pair.outcome === target.id && pair.outcomeDirection === target.direction && (this.opts.rubrics === void 0 || this.opts.rubrics.includes(pair.rubric)));
|
|
1534
|
+
if (pairs.length === 0) return [{
|
|
1535
|
+
kind: "threshold",
|
|
1536
|
+
payload: {
|
|
1537
|
+
directive: "researcher.collect-more-outcomes",
|
|
1538
|
+
targetOutcome: target
|
|
1539
|
+
},
|
|
1540
|
+
rationale: `no estimable rubric association with ${target.id}; collect independent outcome observations before proposing weight changes`
|
|
1541
|
+
}];
|
|
1542
|
+
for (const pair of pairs) {
|
|
1543
|
+
const interval = pair.alignedSpearmanCi95;
|
|
1544
|
+
const aligned = pair.alignedSpearman >= .4 && interval !== null && interval.lower > 0;
|
|
1545
|
+
const inverse = pair.alignedSpearman <= -.4 && interval !== null && interval.upper < 0;
|
|
1546
|
+
const action = aligned ? "test-up-weight" : inverse ? "test-reverse-or-replace" : "collect-calibration-evidence";
|
|
1543
1547
|
changes.push({
|
|
1544
1548
|
kind: "reviewer_prompt",
|
|
1545
1549
|
payload: {
|
|
1546
|
-
rubric:
|
|
1547
|
-
action
|
|
1548
|
-
|
|
1549
|
-
|
|
1550
|
+
rubric: pair.rubric,
|
|
1551
|
+
action,
|
|
1552
|
+
targetOutcome: target,
|
|
1553
|
+
spearman: pair.spearman,
|
|
1554
|
+
alignedSpearman: pair.alignedSpearman,
|
|
1555
|
+
alignedSpearmanCi95: interval === null ? null : { ...interval },
|
|
1556
|
+
samples: pair.n
|
|
1550
1557
|
},
|
|
1551
|
-
rationale: `
|
|
1552
|
-
expectedDelta: Math.max(0, Math.abs(ranking.spearman) - .5) * .1
|
|
1558
|
+
rationale: aligned ? `higher ${pair.rubric} scores associate with better ${target.id}; test increased weight on fresh evidence before adopting it` : inverse ? `higher ${pair.rubric} scores associate with worse ${target.id}; test reversal or replacement on fresh evidence` : `the association of ${pair.rubric} with desired ${target.id} does not support a direction of change; collect calibration evidence`
|
|
1553
1559
|
});
|
|
1554
1560
|
}
|
|
1555
1561
|
return changes;
|
|
@@ -1604,11 +1610,11 @@ var PredictiveValidityResearcher = class {
|
|
|
1604
1610
|
const report = await rubricPredictiveValidity({
|
|
1605
1611
|
runs,
|
|
1606
1612
|
outcomes: this.opts.outcomes,
|
|
1607
|
-
outcomeMetrics: this.opts.
|
|
1613
|
+
outcomeMetrics: [this.opts.targetOutcome],
|
|
1608
1614
|
rubrics: this.opts.rubrics
|
|
1609
1615
|
});
|
|
1610
|
-
if (this.opts.onReport) await this.opts.onReport(report);
|
|
1611
|
-
this.
|
|
1616
|
+
if (this.opts.onReport) await this.opts.onReport(structuredClone(report));
|
|
1617
|
+
this.setReport(report);
|
|
1612
1618
|
return report;
|
|
1613
1619
|
}
|
|
1614
1620
|
/**
|
|
@@ -1617,10 +1623,11 @@ var PredictiveValidityResearcher = class {
|
|
|
1617
1623
|
* researcher's later proposals informed by it.
|
|
1618
1624
|
*/
|
|
1619
1625
|
setReport(report) {
|
|
1620
|
-
this.
|
|
1626
|
+
if (report.outcomeMetrics.find((metric) => metric.id === this.opts.targetOutcome.id)?.direction !== this.opts.targetOutcome.direction) throw new Error("predictive validity report does not match the declared target outcome and direction");
|
|
1627
|
+
this.lastReport = structuredClone(report);
|
|
1621
1628
|
}
|
|
1622
1629
|
getLastReport() {
|
|
1623
|
-
return this.lastReport;
|
|
1630
|
+
return this.lastReport === null ? null : structuredClone(this.lastReport);
|
|
1624
1631
|
}
|
|
1625
1632
|
};
|
|
1626
1633
|
/** Coverage of a split that was never dealt any work. */
|
|
@@ -2044,6 +2051,12 @@ function prmTrainingPairs(stepRewardsByRun, opts = {}) {
|
|
|
2044
2051
|
* `result.rewardSignals` to a custom RL loop.
|
|
2045
2052
|
*/
|
|
2046
2053
|
async function runRLCampaign(opts) {
|
|
2054
|
+
const outcomeStore = opts.outcomeStore;
|
|
2055
|
+
const outcomeMetrics = opts.outcomeMetrics?.map((metric) => ({ ...metric }));
|
|
2056
|
+
if (outcomeStore !== void 0 || outcomeMetrics !== void 0) {
|
|
2057
|
+
if (outcomeStore === void 0 || outcomeMetrics === void 0) throw new Error("runRLCampaign requires outcomeStore and outcomeMetrics together");
|
|
2058
|
+
validateOutcomeMetricSpecifications(outcomeMetrics);
|
|
2059
|
+
}
|
|
2047
2060
|
const splitTag = opts.splitTag ?? "search";
|
|
2048
2061
|
const campaign = await runEvalCampaign({
|
|
2049
2062
|
...opts,
|
|
@@ -2080,10 +2093,10 @@ async function runRLCampaign(opts) {
|
|
|
2080
2093
|
verifiableRewardOptions: opts.verifiableReward
|
|
2081
2094
|
});
|
|
2082
2095
|
let predictiveValidity = null;
|
|
2083
|
-
if (
|
|
2096
|
+
if (outcomeStore && outcomeMetrics) predictiveValidity = await rubricPredictiveValidity({
|
|
2084
2097
|
runs: campaign.runs,
|
|
2085
|
-
outcomes:
|
|
2086
|
-
outcomeMetrics
|
|
2098
|
+
outcomes: outcomeStore,
|
|
2099
|
+
outcomeMetrics
|
|
2087
2100
|
});
|
|
2088
2101
|
const trainerRows = {};
|
|
2089
2102
|
if (opts.trainerExport?.dpo) trainerRows.dpo = await toDpoRows(preferences.pairs, opts.trainerExport.dpo, { lines: rolloutLines });
|
|
@@ -2215,7 +2228,7 @@ function buildSummary(args) {
|
|
|
2215
2228
|
lines.push(`reward-hacking: ${args.rewardHacking.verdict} (${args.rewardHacking.findings.length} signals checked)`);
|
|
2216
2229
|
if (args.predictiveValidity) {
|
|
2217
2230
|
const top = args.predictiveValidity.ranked[0];
|
|
2218
|
-
lines.push(`top-rubric: ${top
|
|
2231
|
+
lines.push(top ? `top-rubric: ${top.rubric} aligned ρ=${top.alignedSpearman.toFixed(2)} vs ${top.bestOutcome} (${top.outcomeDirection}; ${top.verdict})` : "top-rubric: none (no estimable outcome associations)");
|
|
2219
2232
|
}
|
|
2220
2233
|
return lines.join(" | ");
|
|
2221
2234
|
}
|