@tangle-network/agent-eval 0.132.0 → 0.133.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +15 -0
- package/README.md +1 -1
- package/dist/agent-profile-cell-OhuTee9n.js +335 -0
- package/dist/agent-profile-cell-OhuTee9n.js.map +1 -0
- package/dist/analyst/index.js +3 -3
- package/dist/{analyze-runs-BjPn_fOS.js → analyze-runs-B-afTpCv.js} +6 -19
- package/dist/analyze-runs-B-afTpCv.js.map +1 -0
- package/dist/{analyze-runs-AFDI5RI0.d.ts → analyze-runs-DZr7JW-m.d.ts} +2 -2
- package/dist/{analyze-runs-AFDI5RI0.d.ts.map → analyze-runs-DZr7JW-m.d.ts.map} +1 -1
- package/dist/{baseline-HsBvw_dk.js → baseline-DcX5hQDv.js} +2 -70
- package/dist/{baseline-HsBvw_dk.js.map → baseline-DcX5hQDv.js.map} +1 -1
- package/dist/baseline-hG3K85h4.d.ts.map +1 -1
- package/dist/benchmarks/index.d.ts +1 -1
- package/dist/benchmarks/index.js +1 -1
- package/dist/{benchmarks-DTrT3UH-.js → benchmarks-BU7P6PCW.js} +3 -3
- package/dist/{benchmarks-DTrT3UH-.js.map → benchmarks-BU7P6PCW.js.map} +1 -1
- package/dist/builder-eval/index.js +1 -1
- package/dist/campaign/index.d.ts +3 -3
- package/dist/campaign/index.js +2 -2
- package/dist/{campaign-Cx6CfMR4.js → campaign-CnzHQndg.js} +21 -15
- package/dist/{campaign-Cx6CfMR4.js.map → campaign-CnzHQndg.js.map} +1 -1
- package/dist/cli.js +1 -1
- package/dist/{client-aZDHJiKO.d.ts → client-D4F9hdzR.d.ts} +2 -2
- package/dist/{client-aZDHJiKO.d.ts.map → client-D4F9hdzR.d.ts.map} +1 -1
- package/dist/contract/index.d.ts +145 -6
- package/dist/contract/index.d.ts.map +1 -1
- package/dist/contract/index.js +1395 -11
- package/dist/contract/index.js.map +1 -1
- package/dist/control.js +1 -1
- package/dist/{cost-ledger-ZAa_P4r0.js → cost-ledger-BrJxbrMy.js} +238 -3
- package/dist/cost-ledger-BrJxbrMy.js.map +1 -0
- package/dist/{default-registry-B1JcpnRv.js → default-registry-D3T9XbuY.js} +3 -3
- package/dist/{default-registry-B1JcpnRv.js.map → default-registry-D3T9XbuY.js.map} +1 -1
- package/dist/{eval-campaign-mDKhkdUq.js → eval-campaign-DXhpZghy.js} +5 -6
- package/dist/{eval-campaign-mDKhkdUq.js.map → eval-campaign-DXhpZghy.js.map} +1 -1
- package/dist/{task-failure-attributes-CQZlB3et.js → extract-usage-2j25whHw.js} +154 -2
- package/dist/extract-usage-2j25whHw.js.map +1 -0
- package/dist/fuzz.js +1 -1
- package/dist/hosted/index.d.ts +1 -1
- package/dist/{index-3cdlURSk2.d.ts → index-3cdlURSk.d.ts} +1 -1
- package/dist/index-3cdlURSk.d.ts.map +1 -0
- package/dist/{index-FpfWFsKm.d.ts → index-Ba636PKl.d.ts} +29 -6
- package/dist/{index-FpfWFsKm.d.ts.map → index-Ba636PKl.d.ts.map} +1 -1
- package/dist/{index-p2TR_iWJ.d.ts → index-Wek5mU0y.d.ts} +3 -3
- package/dist/{index-p2TR_iWJ.d.ts.map → index-Wek5mU0y.d.ts.map} +1 -1
- package/dist/{index-BAvgST_9.d.ts → index-nhIYz9hn.d.ts} +67 -9
- package/dist/index-nhIYz9hn.d.ts.map +1 -0
- package/dist/index.d.ts +55 -9
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +111 -24
- package/dist/index.js.map +1 -1
- package/dist/ledger-core/index.d.ts +2 -2
- package/dist/ledger-core/index.js +2 -2
- package/dist/ledger-core-CPZfcrC2.js +620 -0
- package/dist/ledger-core-CPZfcrC2.js.map +1 -0
- package/dist/{llm-client-BNcP4v08.js → llm-client-ClPW-dWB.js} +2 -2
- package/dist/{llm-client-BNcP4v08.js.map → llm-client-ClPW-dWB.js.map} +1 -1
- package/dist/meta-eval/index.d.ts +215 -2
- package/dist/{index-CXs7QlR5.d.ts.map → meta-eval/index.d.ts.map} +1 -1
- package/dist/meta-eval/index.js +93 -3
- package/dist/meta-eval/index.js.map +1 -1
- package/dist/{mint-D5_87M5L.js → mint-BvkwcYZU.js} +2 -2
- package/dist/{mint-D5_87M5L.js.map → mint-BvkwcYZU.js.map} +1 -1
- package/dist/openapi.json +1 -1
- package/dist/{paired-arms-D9D0wXj2.js → paired-arms-6XItKzd1.js} +2 -2
- package/dist/{paired-arms-D9D0wXj2.js.map → paired-arms-6XItKzd1.js.map} +1 -1
- package/dist/pipelines/index.js +2 -2
- package/dist/profile-cell.js +1 -242
- package/dist/{propose-review-control-Bqb7daEJ.js → propose-review-control-SQ-n9-We.js} +2 -2
- package/dist/{propose-review-control-Bqb7daEJ.js.map → propose-review-control-SQ-n9-We.js.map} +1 -1
- package/dist/{release-report-oWt9f2k-.js → release-report-wuilQkvK.js} +3 -3
- package/dist/{release-report-oWt9f2k-.js.map → release-report-wuilQkvK.js.map} +1 -1
- package/dist/{replay-D18-pBAA.js → replay-CJfGLdx4.js} +4 -5
- package/dist/{replay-D18-pBAA.js.map → replay-CJfGLdx4.js.map} +1 -1
- package/dist/reporting.d.ts +1 -1
- package/dist/reporting.js +4 -4
- package/dist/{reward-hacking-BEvjdUtD.js → reward-hacking-Dl2UBzej.js} +3 -3
- package/dist/{reward-hacking-BEvjdUtD.js.map → reward-hacking-Dl2UBzej.js.map} +1 -1
- package/dist/rl.d.ts +153 -3
- package/dist/rl.d.ts.map +1 -1
- package/dist/rl.js +222 -7
- package/dist/rl.js.map +1 -1
- package/dist/rollout/index.d.ts +1 -1
- package/dist/rollout/index.js +2 -2
- package/dist/{rollout-VEo41J0N.js → rollout-CeTlDrf6.js} +2 -2
- package/dist/{rollout-VEo41J0N.js.map → rollout-CeTlDrf6.js.map} +1 -1
- package/dist/{rubric-predictive-validity-B3xmbmS1.js → rubric-predictive-validity-QG7ydk0s.js} +2 -2
- package/dist/{rubric-predictive-validity-B3xmbmS1.js.map → rubric-predictive-validity-QG7ydk0s.js.map} +1 -1
- package/dist/{run-record-CN8Zd21B.js → run-record-BIwU2wdV.js} +2 -2
- package/dist/{run-record-CN8Zd21B.js.map → run-record-BIwU2wdV.js.map} +1 -1
- package/dist/{semantic-concept-judge-DKCtoOz8.js → semantic-concept-judge-BypLt6Fw.js} +4 -4
- package/dist/{semantic-concept-judge-DKCtoOz8.js.map → semantic-concept-judge-BypLt6Fw.js.map} +1 -1
- package/dist/{server-Dc_lsOYd.js → server-BPqlDBWK.js} +3 -3
- package/dist/{server-Dc_lsOYd.js.map → server-BPqlDBWK.js.map} +1 -1
- package/dist/{skillopt-optimization-method-CQlz8GQM.js → skillopt-optimization-method-BoIzh7Dl.js} +6 -6
- package/dist/{skillopt-optimization-method-CQlz8GQM.js.map → skillopt-optimization-method-BoIzh7Dl.js.map} +1 -1
- package/dist/{skillopt-optimization-method-C9M_lxdo.d.ts → skillopt-optimization-method-CAASpcS3.d.ts} +3 -3
- package/dist/{skillopt-optimization-method-C9M_lxdo.d.ts.map → skillopt-optimization-method-CAASpcS3.d.ts.map} +1 -1
- package/dist/{statistics-CnnxdpOg.js → statistics-DWM_AyLe.js} +90 -76
- package/dist/statistics-DWM_AyLe.js.map +1 -0
- package/dist/statistics-DbvkkDPa.d.ts.map +1 -1
- package/dist/{summary-report-BNs5nmXI.js → summary-report-Ci17nIdU.js} +4 -4
- package/dist/{summary-report-BNs5nmXI.js.map → summary-report-Ci17nIdU.js.map} +1 -1
- package/dist/traces.js +2 -2
- package/dist/wire/index.js +1 -1
- package/package.json +2 -7
- package/dist/analyze-runs-BjPn_fOS.js.map +0 -1
- package/dist/belief-state/index.d.ts +0 -622
- package/dist/belief-state/index.d.ts.map +0 -1
- package/dist/belief-state/index.js +0 -1819
- package/dist/belief-state/index.js.map +0 -1
- package/dist/calibration-CNWWA6K8.js +0 -94
- package/dist/calibration-CNWWA6K8.js.map +0 -1
- package/dist/code-agent-session-BjkMTQ7H.js +0 -1390
- package/dist/code-agent-session-BjkMTQ7H.js.map +0 -1
- package/dist/code-agent-session-D5URqc3_.d.ts +0 -143
- package/dist/code-agent-session-D5URqc3_.d.ts.map +0 -1
- package/dist/cost-ledger-ZAa_P4r0.js.map +0 -1
- package/dist/extract-usage-BrQ8mCLX.js +0 -155
- package/dist/extract-usage-BrQ8mCLX.js.map +0 -1
- package/dist/index-3cdlURSk2.d.ts.map +0 -1
- package/dist/index-BAvgST_9.d.ts.map +0 -1
- package/dist/index-CXs7QlR5.d.ts +0 -217
- package/dist/ledger-core-eqaI3PCD.js +0 -388
- package/dist/ledger-core-eqaI3PCD.js.map +0 -1
- package/dist/metrics-C9YY1OcL.js +0 -239
- package/dist/metrics-C9YY1OcL.js.map +0 -1
- package/dist/off-policy-DvgzvtIx.js +0 -220
- package/dist/off-policy-DvgzvtIx.js.map +0 -1
- package/dist/off-policy-mskQw8Mb.d.ts +0 -153
- package/dist/off-policy-mskQw8Mb.d.ts.map +0 -1
- package/dist/pre-registration-DakwTRXk.js +0 -96
- package/dist/pre-registration-DakwTRXk.js.map +0 -1
- package/dist/profile-cell.js.map +0 -1
- package/dist/runtime-trajectory-1gyaTOoC.js +0 -93
- package/dist/runtime-trajectory-1gyaTOoC.js.map +0 -1
- package/dist/runtime-trajectory-BW9Wszb-.d.ts +0 -50
- package/dist/runtime-trajectory-BW9Wszb-.d.ts.map +0 -1
- package/dist/statistics-CnnxdpOg.js.map +0 -1
- package/dist/task-failure-attributes-CQZlB3et.js.map +0 -1
package/dist/rl.js
CHANGED
|
@@ -1,17 +1,16 @@
|
|
|
1
1
|
import { s as ValidationError } from "./errors-8YnH8WlF.js";
|
|
2
|
-
import { N as wilcoxonSignedRank, t as benjaminiHochberg } from "./statistics-
|
|
2
|
+
import { N as wilcoxonSignedRank, t as benjaminiHochberg } from "./statistics-DWM_AyLe.js";
|
|
3
3
|
import { r as observedSplitScore, s as trainingScore } from "./reward-nw2xZGZG.js";
|
|
4
|
-
import { o as runTaskScore } from "./run-record-
|
|
4
|
+
import { o as runTaskScore } from "./run-record-BIwU2wdV.js";
|
|
5
5
|
import { l as assertRewardGate } from "./schema-C6DW4ZHR.js";
|
|
6
6
|
import { t as isSplitEligible } from "./exporters-q9iL-2Jf.js";
|
|
7
|
-
import { t as mintRolloutRows } from "./mint-
|
|
7
|
+
import { t as mintRolloutRows } from "./mint-BvkwcYZU.js";
|
|
8
8
|
import { a as InMemoryTraceStore } from "./integrity-BzRbCHzi.js";
|
|
9
|
-
import { c as campaignCellToRunRecord, i as filterDeterministicallyRewarded, n as extractVerifiableReward, r as extractVerifiableRewardsFromRecords, t as detectRewardHacking } from "./reward-hacking-
|
|
10
|
-
import { t as runEvalCampaign } from "./eval-campaign-
|
|
9
|
+
import { c as campaignCellToRunRecord, i as filterDeterministicallyRewarded, n as extractVerifiableReward, r as extractVerifiableRewardsFromRecords, t as detectRewardHacking } from "./reward-hacking-Dl2UBzej.js";
|
|
10
|
+
import { t as runEvalCampaign } from "./eval-campaign-DXhpZghy.js";
|
|
11
11
|
import { t as evaluateInterimReleaseConfidence } from "./sequential-Br0mAPHA.js";
|
|
12
|
-
import { t as rubricPredictiveValidity } from "./rubric-predictive-validity-
|
|
12
|
+
import { t as rubricPredictiveValidity } from "./rubric-predictive-validity-QG7ydk0s.js";
|
|
13
13
|
import { n as thompsonCurriculum, r as varianceBasedCurriculum, t as observationsFromRunRecords } from "./active-curriculum-C4mk67HP.js";
|
|
14
|
-
import { i as selfNormalizedImportanceWeighting, n as inverseProbabilityWeighting, r as offPolicyEstimateAll, t as doublyRobust } from "./off-policy-DvgzvtIx.js";
|
|
15
14
|
import { n as InMemoryOutcomeStore, t as FileSystemOutcomeStore } from "./outcome-store-ChBKlTd_.js";
|
|
16
15
|
import { appendFileSync, existsSync, mkdirSync, readFileSync } from "node:fs";
|
|
17
16
|
import { dirname } from "node:path";
|
|
@@ -1242,6 +1241,222 @@ async function buildDatasetFromCorpus(corpusPath, config, opts = {}) {
|
|
|
1242
1241
|
return buildRlDataset(rows, lookups, config);
|
|
1243
1242
|
}
|
|
1244
1243
|
//#endregion
|
|
1244
|
+
//#region src/rl/off-policy.ts
|
|
1245
|
+
/**
|
|
1246
|
+
* Off-policy evaluation primitives.
|
|
1247
|
+
*
|
|
1248
|
+
* Standard inverse-probability-weighted (IPS), self-normalized
|
|
1249
|
+
* importance-weighted (SNIPS), and doubly-robust (DR) estimators for the
|
|
1250
|
+
* value of a *target* policy given trajectories collected under a
|
|
1251
|
+
* *behavior* policy. This is the canonical RL eval task: "we have last
|
|
1252
|
+
* week's runs, we changed the policy — how would the new one do without
|
|
1253
|
+
* re-running?"
|
|
1254
|
+
*
|
|
1255
|
+
* The math here is textbook (Dudík, Langford, Li 2011 for DR; Swaminathan
|
|
1256
|
+
* & Joachims 2015 for SNIPS) but the *application* to LLM-agent
|
|
1257
|
+
* evaluation needs care:
|
|
1258
|
+
*
|
|
1259
|
+
* - The "policy" is the (prompt, tool config, model snapshot) triple.
|
|
1260
|
+
* Two policies have the same probability over an action *iff* their
|
|
1261
|
+
* LLM call would emit the same token with the same probability —
|
|
1262
|
+
* which is generally unknowable without the model log-probs.
|
|
1263
|
+
* - For LLM agents, propensity scores must be supplied by the caller
|
|
1264
|
+
* (logged in the trace, recovered from token log-probs, or estimated
|
|
1265
|
+
* via a learned propensity model). We do NOT estimate propensity here.
|
|
1266
|
+
* - Doubly-robust requires two outputs from a Q-function: its prediction
|
|
1267
|
+
* for the logged action and its expectation under the target policy.
|
|
1268
|
+
* Consumers compute these with a tabular estimate, regression fit, or
|
|
1269
|
+
* learned reward model before constructing the trajectories.
|
|
1270
|
+
*
|
|
1271
|
+
* Bias / variance tradeoffs:
|
|
1272
|
+
* - IPS: unbiased; high variance for small overlap, infinite variance
|
|
1273
|
+
* when target has support outside behavior.
|
|
1274
|
+
* - SNIPS: lower variance, slight bias; usually preferred in practice.
|
|
1275
|
+
* - DR: doubly-robust — unbiased if either propensity OR Q-function is
|
|
1276
|
+
* correct. Lowest practical variance when Q is decent. Use this.
|
|
1277
|
+
*
|
|
1278
|
+
* Caveat the panel will land: on the LLM-agent setting, propensity scores
|
|
1279
|
+
* recovered from token log-probs are noisy, the action space is enormous,
|
|
1280
|
+
* and overlap is often poor. These estimators are useful but not magic;
|
|
1281
|
+
* complement with `replayCampaign` (exact replay where the request hashes
|
|
1282
|
+
* match) for high-confidence answers and OPE for the gap.
|
|
1283
|
+
*/
|
|
1284
|
+
/**
|
|
1285
|
+
* Inverse Probability Weighting (Horvitz-Thompson). Unbiased estimator
|
|
1286
|
+
* of E[reward under target policy]. Variance scales with the spread of
|
|
1287
|
+
* target/behavior ratios.
|
|
1288
|
+
*/
|
|
1289
|
+
function inverseProbabilityWeighting(trajectories, opts = {}) {
|
|
1290
|
+
const cap = opts.weightCap ?? Infinity;
|
|
1291
|
+
const clip = opts.rewardClip ?? {
|
|
1292
|
+
low: 0,
|
|
1293
|
+
high: 1
|
|
1294
|
+
};
|
|
1295
|
+
if (trajectories.length === 0) return zeroEstimate();
|
|
1296
|
+
const weights = [];
|
|
1297
|
+
const weightedRewards = [];
|
|
1298
|
+
let maxW = 0;
|
|
1299
|
+
for (const t of trajectories) {
|
|
1300
|
+
if (t.behaviorProb <= 0) throw new ValidationError(`inverseProbabilityWeighting: behaviorProb must be > 0 (runId=${t.runId})`);
|
|
1301
|
+
const w = Math.min(cap, t.targetProb / t.behaviorProb);
|
|
1302
|
+
const r = clamp(t.reward, clip.low, clip.high);
|
|
1303
|
+
weights.push(w);
|
|
1304
|
+
weightedRewards.push(w * r);
|
|
1305
|
+
if (w > maxW) maxW = w;
|
|
1306
|
+
}
|
|
1307
|
+
const n = weights.length;
|
|
1308
|
+
const value = weightedRewards.reduce((s, x) => s + x, 0) / n;
|
|
1309
|
+
const variance = weightedRewards.reduce((s, x) => s + (x - value) ** 2, 0) / Math.max(1, n - 1);
|
|
1310
|
+
const sumW = weights.reduce((s, w) => s + w, 0);
|
|
1311
|
+
const sumW2 = weights.reduce((s, w) => s + w * w, 0);
|
|
1312
|
+
const effN = sumW === 0 ? 0 : sumW * sumW / sumW2;
|
|
1313
|
+
return {
|
|
1314
|
+
value,
|
|
1315
|
+
standardError: Math.sqrt(variance / n),
|
|
1316
|
+
effectiveSampleSize: effN,
|
|
1317
|
+
n,
|
|
1318
|
+
maxImportanceWeight: maxW
|
|
1319
|
+
};
|
|
1320
|
+
}
|
|
1321
|
+
/**
|
|
1322
|
+
* Self-Normalized Importance Sampling. Lower variance than vanilla IPS at
|
|
1323
|
+
* the cost of small bias (vanishing as N grows). The right default for
|
|
1324
|
+
* LLM-agent evaluation where overlap is often poor.
|
|
1325
|
+
*/
|
|
1326
|
+
function selfNormalizedImportanceWeighting(trajectories, opts = {}) {
|
|
1327
|
+
const cap = opts.weightCap ?? Infinity;
|
|
1328
|
+
const clip = opts.rewardClip ?? {
|
|
1329
|
+
low: 0,
|
|
1330
|
+
high: 1
|
|
1331
|
+
};
|
|
1332
|
+
if (trajectories.length === 0) return zeroEstimate();
|
|
1333
|
+
const weights = [];
|
|
1334
|
+
const rewards = [];
|
|
1335
|
+
let maxW = 0;
|
|
1336
|
+
for (const t of trajectories) {
|
|
1337
|
+
if (t.behaviorProb <= 0) throw new ValidationError(`selfNormalizedImportanceWeighting: behaviorProb must be > 0 (runId=${t.runId})`);
|
|
1338
|
+
const w = Math.min(cap, t.targetProb / t.behaviorProb);
|
|
1339
|
+
weights.push(w);
|
|
1340
|
+
rewards.push(clamp(t.reward, clip.low, clip.high));
|
|
1341
|
+
if (w > maxW) maxW = w;
|
|
1342
|
+
}
|
|
1343
|
+
const sumW = weights.reduce((s, w) => s + w, 0);
|
|
1344
|
+
const sumWR = weights.reduce((s, w, i) => s + w * rewards[i], 0);
|
|
1345
|
+
const value = sumW === 0 ? 0 : sumWR / sumW;
|
|
1346
|
+
const sumW2 = weights.reduce((s, w) => s + w * w, 0);
|
|
1347
|
+
const effN = sumW === 0 ? 0 : sumW * sumW / sumW2;
|
|
1348
|
+
const variance = weights.map((w, i) => w * (rewards[i] - value)).reduce((s, x) => s + x * x, 0) / Math.max(1, sumW * sumW);
|
|
1349
|
+
return {
|
|
1350
|
+
value,
|
|
1351
|
+
standardError: Math.sqrt(variance),
|
|
1352
|
+
effectiveSampleSize: effN,
|
|
1353
|
+
n: trajectories.length,
|
|
1354
|
+
maxImportanceWeight: maxW
|
|
1355
|
+
};
|
|
1356
|
+
}
|
|
1357
|
+
/**
|
|
1358
|
+
* Doubly-robust off-policy estimator (Dudík, Langford, Li 2011).
|
|
1359
|
+
*
|
|
1360
|
+
* V_DR = (1/N) * sum_i [ v_hat_target_i
|
|
1361
|
+
* + (target_prob_i / behavior_prob_i) * (r_i - q_hat_chosen_i) ]
|
|
1362
|
+
*
|
|
1363
|
+
* Unbiased if EITHER:
|
|
1364
|
+
* - the importance ratios are correct (IPS-style validity), OR
|
|
1365
|
+
* - the Q-hat function is correct (model-based validity).
|
|
1366
|
+
*
|
|
1367
|
+
* In practice both are imperfect, but the residual bias is the *product*
|
|
1368
|
+
* of both errors — much smaller than either alone. This is why DR is the
|
|
1369
|
+
* default in production OPE pipelines.
|
|
1370
|
+
*
|
|
1371
|
+
* `qHatChosen` and `vHatTarget` must be supplied together. Rows with neither
|
|
1372
|
+
* use the exact IPS contribution. `contributionCounts` makes the mix explicit
|
|
1373
|
+
* in the result.
|
|
1374
|
+
* Callers must cross-fit the Q-function or train it on independent rows;
|
|
1375
|
+
* fitting and evaluating Q on the same outcomes leaks the answer.
|
|
1376
|
+
*/
|
|
1377
|
+
function doublyRobust(trajectories, opts = {}) {
|
|
1378
|
+
const cap = opts.weightCap ?? Infinity;
|
|
1379
|
+
const clip = opts.rewardClip ?? {
|
|
1380
|
+
low: 0,
|
|
1381
|
+
high: 1
|
|
1382
|
+
};
|
|
1383
|
+
if (trajectories.length === 0) return {
|
|
1384
|
+
...zeroEstimate(),
|
|
1385
|
+
contributionCounts: {
|
|
1386
|
+
dr: 0,
|
|
1387
|
+
ipsFallback: 0
|
|
1388
|
+
}
|
|
1389
|
+
};
|
|
1390
|
+
const contributions = [];
|
|
1391
|
+
const contributionCounts = {
|
|
1392
|
+
dr: 0,
|
|
1393
|
+
ipsFallback: 0
|
|
1394
|
+
};
|
|
1395
|
+
let maxW = 0;
|
|
1396
|
+
let sumW = 0;
|
|
1397
|
+
let sumW2 = 0;
|
|
1398
|
+
for (const t of trajectories) {
|
|
1399
|
+
if (t.behaviorProb <= 0) throw new ValidationError(`doublyRobust: behaviorProb must be > 0 (runId=${t.runId})`);
|
|
1400
|
+
const w = Math.min(cap, t.targetProb / t.behaviorProb);
|
|
1401
|
+
const r = clamp(t.reward, clip.low, clip.high);
|
|
1402
|
+
const rawQHatChosen = t.qHatChosen;
|
|
1403
|
+
const rawVHatTarget = t.vHatTarget;
|
|
1404
|
+
const hasQHatChosen = rawQHatChosen !== null && rawQHatChosen !== void 0;
|
|
1405
|
+
const hasVHatTarget = rawVHatTarget !== null && rawVHatTarget !== void 0;
|
|
1406
|
+
if (hasQHatChosen !== hasVHatTarget) throw new ValidationError(`doublyRobust: qHatChosen and vHatTarget must be supplied together (runId=${t.runId})`);
|
|
1407
|
+
if (hasQHatChosen && hasVHatTarget) {
|
|
1408
|
+
if (!Number.isFinite(rawQHatChosen) || !Number.isFinite(rawVHatTarget)) throw new ValidationError(`doublyRobust: qHatChosen and vHatTarget must be finite (runId=${t.runId})`);
|
|
1409
|
+
const qHatChosen = clamp(rawQHatChosen, clip.low, clip.high);
|
|
1410
|
+
const vHatTarget = clamp(rawVHatTarget, clip.low, clip.high);
|
|
1411
|
+
contributions.push(vHatTarget + w * (r - qHatChosen));
|
|
1412
|
+
contributionCounts.dr += 1;
|
|
1413
|
+
} else {
|
|
1414
|
+
contributions.push(w * r);
|
|
1415
|
+
contributionCounts.ipsFallback += 1;
|
|
1416
|
+
}
|
|
1417
|
+
if (w > maxW) maxW = w;
|
|
1418
|
+
sumW += w;
|
|
1419
|
+
sumW2 += w * w;
|
|
1420
|
+
}
|
|
1421
|
+
const n = contributions.length;
|
|
1422
|
+
const value = contributions.reduce((s, x) => s + x, 0) / n;
|
|
1423
|
+
const variance = contributions.reduce((s, x) => s + (x - value) ** 2, 0) / Math.max(1, n - 1);
|
|
1424
|
+
const effN = sumW === 0 ? 0 : sumW * sumW / sumW2;
|
|
1425
|
+
return {
|
|
1426
|
+
value,
|
|
1427
|
+
standardError: Math.sqrt(variance / n),
|
|
1428
|
+
effectiveSampleSize: effN,
|
|
1429
|
+
n,
|
|
1430
|
+
maxImportanceWeight: maxW,
|
|
1431
|
+
contributionCounts
|
|
1432
|
+
};
|
|
1433
|
+
}
|
|
1434
|
+
/**
|
|
1435
|
+
* Convenience: run all three estimators and return them side-by-side.
|
|
1436
|
+
* The recommended diagnostic — agreement across estimators is a much
|
|
1437
|
+
* stronger signal than any single one.
|
|
1438
|
+
*/
|
|
1439
|
+
function offPolicyEstimateAll(trajectories, opts = {}) {
|
|
1440
|
+
return {
|
|
1441
|
+
ips: inverseProbabilityWeighting(trajectories, opts),
|
|
1442
|
+
snips: selfNormalizedImportanceWeighting(trajectories, opts),
|
|
1443
|
+
dr: doublyRobust(trajectories, opts)
|
|
1444
|
+
};
|
|
1445
|
+
}
|
|
1446
|
+
function zeroEstimate() {
|
|
1447
|
+
return {
|
|
1448
|
+
value: 0,
|
|
1449
|
+
standardError: 0,
|
|
1450
|
+
effectiveSampleSize: 0,
|
|
1451
|
+
n: 0,
|
|
1452
|
+
maxImportanceWeight: 0
|
|
1453
|
+
};
|
|
1454
|
+
}
|
|
1455
|
+
function clamp(x, lo, hi) {
|
|
1456
|
+
if (!Number.isFinite(x)) return lo;
|
|
1457
|
+
return Math.max(lo, Math.min(hi, x));
|
|
1458
|
+
}
|
|
1459
|
+
//#endregion
|
|
1245
1460
|
//#region src/rl/predictive-validity-researcher.ts
|
|
1246
1461
|
/**
|
|
1247
1462
|
* Concrete `Researcher` driven by `rubricPredictiveValidity`. The brain:
|