@tangle-network/agent-eval 0.180.0 → 0.181.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +60 -0
- package/README.md +119 -159
- package/dist/adapters/http.d.ts +2 -2
- package/dist/{agent-profile-B7yErX0q.d.ts → agent-profile-CivaSsSy.d.ts} +4 -4
- package/dist/{agent-profile-B7yErX0q.d.ts.map → agent-profile-CivaSsSy.d.ts.map} +1 -1
- package/dist/{agent-profile-cell-0gSi5ffD.js → agent-profile-cell-Cv6UA-W_.js} +20 -57
- package/dist/agent-profile-cell-Cv6UA-W_.js.map +1 -0
- package/dist/{agent-profile-cell-CTOZJUuE.d.ts → agent-profile-cell-s__adRnK.d.ts} +3 -3
- package/dist/agent-profile-cell-s__adRnK.d.ts.map +1 -0
- package/dist/analyst/index.d.ts +10 -10
- package/dist/analyst/index.js +3 -3
- package/dist/ast-CP9ae9B0.js +557 -0
- package/dist/ast-CP9ae9B0.js.map +1 -0
- package/dist/ast-hI-vjW6J.d.ts +457 -0
- package/dist/ast-hI-vjW6J.d.ts.map +1 -0
- package/dist/{benchmark-command-D4vpnAdO.js → benchmark-command-B57n9vjz.js} +7 -6
- package/dist/{benchmark-command-D4vpnAdO.js.map → benchmark-command-B57n9vjz.js.map} +1 -1
- package/dist/benchmarks/index.d.ts +4 -4
- package/dist/benchmarks/index.js +3 -3
- package/dist/campaign/index.d.ts +6 -6
- package/dist/campaign/index.js +8 -8
- package/dist/{campaign-BGEurASO.js → campaign-4_ppJW5X.js} +12 -12
- package/dist/{campaign-BGEurASO.js.map → campaign-4_ppJW5X.js.map} +1 -1
- package/dist/{campaign-evidence-D8DBLqLI.js → campaign-evidence-B8oF9xQ6.js} +515 -471
- package/dist/campaign-evidence-B8oF9xQ6.js.map +1 -0
- package/dist/cli.js +4 -7
- package/dist/cli.js.map +1 -1
- package/dist/{client-BlLY6o2w.js → client-CXE-U1SA.js} +3 -1
- package/dist/client-CXE-U1SA.js.map +1 -0
- package/dist/{client-CuQgX33c.d.ts → client-kh2jOjTK.d.ts} +4 -4
- package/dist/{client-CuQgX33c.d.ts.map → client-kh2jOjTK.d.ts.map} +1 -1
- package/dist/contract/index.d.ts +13 -13
- package/dist/contract/index.js +10 -9
- package/dist/contract/index.js.map +1 -1
- package/dist/{default-registry-IGDE9XIC.d.ts → default-registry-BwDSWVzg.d.ts} +6 -6
- package/dist/{default-registry-IGDE9XIC.d.ts.map → default-registry-BwDSWVzg.d.ts.map} +1 -1
- package/dist/{define-agent-eval-Cx4Ls9ta.d.ts → define-agent-eval-CwOWWQt_.d.ts} +33 -12
- package/dist/define-agent-eval-CwOWWQt_.d.ts.map +1 -0
- package/dist/{define-agent-eval-Dzidv34q.js → define-agent-eval-Ddu33JH9.js} +134 -67
- package/dist/define-agent-eval-Ddu33JH9.js.map +1 -0
- package/dist/{dspy-rlm-engine-xKiWmj_G.js → dspy-rlm-engine-S53V0HhE.js} +2 -2
- package/dist/{dspy-rlm-engine-xKiWmj_G.js.map → dspy-rlm-engine-S53V0HhE.js.map} +1 -1
- package/dist/{engine-CX8ReXkn.d.ts → engine-DS1cysJy.d.ts} +10 -7
- package/dist/engine-DS1cysJy.d.ts.map +1 -0
- package/dist/{eval-campaign-Cs-7MiCs.js → eval-campaign-aYdtjtJR.js} +4 -4
- package/dist/{eval-campaign-Cs-7MiCs.js.map → eval-campaign-aYdtjtJR.js.map} +1 -1
- package/dist/{exact-types-B7LC1EyX.d.ts → exact-types-BZDe0W2D.d.ts} +2 -2
- package/dist/{exact-types-B7LC1EyX.d.ts.map → exact-types-BZDe0W2D.d.ts.map} +1 -1
- package/dist/experiment/index.d.ts +27 -477
- package/dist/experiment/index.d.ts.map +1 -1
- package/dist/experiment/index.js +95 -559
- package/dist/experiment/index.js.map +1 -1
- package/dist/{experiment-tracker-B3TiF5-u.d.ts → experiment-tracker-C7PfnF4b.d.ts} +2 -2
- package/dist/{experiment-tracker-B3TiF5-u.d.ts.map → experiment-tracker-C7PfnF4b.d.ts.map} +1 -1
- package/dist/{external-optimizer-process-Dlz8YxrT.js → external-optimizer-process-QDRURJAM.js} +3 -3
- package/dist/{external-optimizer-process-Dlz8YxrT.js.map → external-optimizer-process-QDRURJAM.js.map} +1 -1
- package/dist/{external-optimizer-subprocess-q3VzlGAO.js → external-optimizer-subprocess-D4dzUBZI.js} +3 -2
- package/dist/{external-optimizer-subprocess-q3VzlGAO.js.map → external-optimizer-subprocess-D4dzUBZI.js.map} +1 -1
- package/dist/{feedback-trajectory-eHWNv5Aj.d.ts → feedback-trajectory-CXmtITBo.d.ts} +3 -3
- package/dist/{feedback-trajectory-eHWNv5Aj.d.ts.map → feedback-trajectory-CXmtITBo.d.ts.map} +1 -1
- package/dist/hosted/index.d.ts +2 -2
- package/dist/hosted/index.d.ts.map +1 -1
- package/dist/hosted/index.js +1 -1
- package/dist/{index-BxWvILU8.d.ts → index-Bp_6sj3x.d.ts} +109 -56
- package/dist/index-Bp_6sj3x.d.ts.map +1 -0
- package/dist/{index-e7LXeRVa.d.ts → index-CJ3LhKIX.d.ts} +2 -2
- package/dist/{index-e7LXeRVa.d.ts.map → index-CJ3LhKIX.d.ts.map} +1 -1
- package/dist/{index-CiUjjEIa.d.ts → index-DNntP4ch.d.ts} +7 -7
- package/dist/{index-CiUjjEIa.d.ts.map → index-DNntP4ch.d.ts.map} +1 -1
- package/dist/{index-DxNYmx4a.d.ts → index-DoykkxW0.d.ts} +11 -11
- package/dist/{index-DxNYmx4a.d.ts.map → index-DoykkxW0.d.ts.map} +1 -1
- package/dist/index.d.ts +28 -28
- package/dist/index.js +24 -15
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-DETqPc_A.d.ts → insight-report-D1qa0HWs.d.ts} +9 -5
- package/dist/{insight-report-DETqPc_A.d.ts.map → insight-report-D1qa0HWs.d.ts.map} +1 -1
- package/dist/{integrity-BKTcA-HP.d.ts → integrity-rGOfSUle.d.ts} +2 -2
- package/dist/{integrity-BKTcA-HP.d.ts.map → integrity-rGOfSUle.d.ts.map} +1 -1
- package/dist/{ledger-core-Cs9f7385.js → journal-Cs9f7385.js} +1 -1
- package/dist/journal-Cs9f7385.js.map +1 -0
- package/dist/{judge-calibration-C5CbMYce.d.ts → judge-calibration-DFtEMlde.d.ts} +31 -2
- package/dist/judge-calibration-DFtEMlde.d.ts.map +1 -0
- package/dist/{judge-calibration-BnpVKtnb.js → judge-calibration-DYmaBtJr.js} +48 -2
- package/dist/{judge-calibration-BnpVKtnb.js.map → judge-calibration-DYmaBtJr.js.map} +1 -1
- package/dist/ledger-core/index.d.ts +1 -1
- package/dist/ledger-core/index.js +1 -1
- package/dist/{llm-judge-v80Kmu9g.js → llm-judge-DEFZeSiu.js} +645 -456
- package/dist/llm-judge-DEFZeSiu.js.map +1 -0
- package/dist/{matrix-DeMmnWrP.d.ts → matrix-CyhW-vgJ.d.ts} +2 -2
- package/dist/{matrix-DeMmnWrP.d.ts.map → matrix-CyhW-vgJ.d.ts.map} +1 -1
- package/dist/meta-eval/index.d.ts +138 -7
- package/dist/meta-eval/index.d.ts.map +1 -1
- package/dist/meta-eval/index.js +245 -97
- package/dist/meta-eval/index.js.map +1 -1
- package/dist/{mint-Cc1_zwRQ.js → mint-ySIIkKlV.js} +2 -2
- package/dist/{mint-Cc1_zwRQ.js.map → mint-ySIIkKlV.js.map} +1 -1
- package/dist/multishot/golden/index.d.ts +1 -1
- package/dist/multishot/index.d.ts +2 -2
- package/dist/openapi.json +1 -1
- package/dist/outcome-store-BXlkwMPR.js +131 -0
- package/dist/outcome-store-BXlkwMPR.js.map +1 -0
- package/dist/{outcome-store-BYHIuO0e.d.ts → outcome-store-CNt4iZ67.d.ts} +18 -25
- package/dist/outcome-store-CNt4iZ67.d.ts.map +1 -0
- package/dist/{paired-promotion-decision-CGzg0cI_.d.ts → paired-promotion-decision-DPsMQm-0.d.ts} +13 -7
- package/dist/{paired-promotion-decision-CGzg0cI_.d.ts.map → paired-promotion-decision-DPsMQm-0.d.ts.map} +1 -1
- package/dist/pipelines/index.js +1 -1
- package/dist/{produced-state-Cv0kJJuP.js → produced-state-BHboMaab.js} +3 -3
- package/dist/{produced-state-Cv0kJJuP.js.map → produced-state-BHboMaab.js.map} +1 -1
- package/dist/profile-cell.d.ts +1 -1
- package/dist/profile-cell.js +1 -1
- package/dist/{promotion-policy-DWOm70gx.js → promotion-policy-CDMMxzb6.js} +28 -40
- package/dist/promotion-policy-CDMMxzb6.js.map +1 -0
- package/dist/{registry-ByVld1-5.d.ts → registry-BRbB6Y0v.d.ts} +4 -4
- package/dist/{registry-ByVld1-5.d.ts.map → registry-BRbB6Y0v.d.ts.map} +1 -1
- package/dist/{release-confidence-BAcNYOf1.d.ts → release-confidence-BcqeQTHW.d.ts} +3 -3
- package/dist/{release-confidence-BAcNYOf1.d.ts.map → release-confidence-BcqeQTHW.d.ts.map} +1 -1
- package/dist/{release-confidence-BcGCclTB.js → release-confidence-DMg8n18l.js} +2 -2
- package/dist/{release-confidence-BcGCclTB.js.map → release-confidence-DMg8n18l.js.map} +1 -1
- package/dist/reporting.d.ts +4 -4
- package/dist/reporting.js +3 -3
- package/dist/{researcher-jsW1X94L.d.ts → researcher-64T49THL.d.ts} +6 -6
- package/dist/{researcher-jsW1X94L.d.ts.map → researcher-64T49THL.d.ts.map} +1 -1
- package/dist/{reward-hacking-ZXEi9VCq.d.ts → reward-hacking-uzO_ihep.d.ts} +2 -2
- package/dist/{reward-hacking-ZXEi9VCq.d.ts.map → reward-hacking-uzO_ihep.d.ts.map} +1 -1
- package/dist/rl.d.ts +53 -99
- package/dist/rl.d.ts.map +1 -1
- package/dist/rl.js +182 -169
- package/dist/rl.js.map +1 -1
- package/dist/rollout/index.d.ts +1 -1
- package/dist/rollout/index.js +2 -2
- package/dist/{rollout-DmoJVqrF.js → rollout-B-UF5R6w.js} +2 -2
- package/dist/{rollout-DmoJVqrF.js.map → rollout-B-UF5R6w.js.map} +1 -1
- package/dist/rubric-predictive-validity-Bmj2_cll.d.ts +79 -0
- package/dist/rubric-predictive-validity-Bmj2_cll.d.ts.map +1 -0
- package/dist/rubric-predictive-validity-CCK-1B7w.js +178 -0
- package/dist/rubric-predictive-validity-CCK-1B7w.js.map +1 -0
- package/dist/{run-record-DTv1MdjK.d.ts → run-record-BiTWauyO.d.ts} +2 -2
- package/dist/{run-record-DTv1MdjK.d.ts.map → run-record-BiTWauyO.d.ts.map} +1 -1
- package/dist/run-record-Br-Yzt_k.js +464 -0
- package/dist/run-record-Br-Yzt_k.js.map +1 -0
- package/dist/{run-record-DQpSf7t-.js → run-record-DualPTn2.js} +2 -2
- package/dist/{run-record-DQpSf7t-.js.map → run-record-DualPTn2.js.map} +1 -1
- package/dist/{semantic-concept-judge-Bi6_iGqg.js → semantic-concept-judge-Bm5JDEKO.js} +3 -3
- package/dist/{semantic-concept-judge-Bi6_iGqg.js.map → semantic-concept-judge-Bm5JDEKO.js.map} +1 -1
- package/dist/{sequential-B5gXgcyp.js → sequential-DAsyV2T9.js} +42 -25
- package/dist/sequential-DAsyV2T9.js.map +1 -0
- package/dist/{series-convergence-DeG33RpC.d.ts → series-convergence-BnMs_uAr.d.ts} +3 -3
- package/dist/{series-convergence-DeG33RpC.d.ts.map → series-convergence-BnMs_uAr.d.ts.map} +1 -1
- package/dist/{skillopt-optimization-method-C3oYul8v.js → skillopt-optimization-method-CL_0aArC.js} +5 -5
- package/dist/{skillopt-optimization-method-C3oYul8v.js.map → skillopt-optimization-method-CL_0aArC.js.map} +1 -1
- package/dist/{statistical-heldout-0La5ZTlv.d.ts → statistical-heldout-CpVd6FmY.d.ts} +207 -144
- package/dist/statistical-heldout-CpVd6FmY.d.ts.map +1 -0
- package/dist/{store-tool-spans-4J1EDElP.d.ts → store-tool-spans-Dt-YdAuE.d.ts} +6 -6
- package/dist/{store-tool-spans-4J1EDElP.d.ts.map → store-tool-spans-Dt-YdAuE.d.ts.map} +1 -1
- package/dist/{summary-report-gMrbYawB.d.ts → summary-report-D1h4dlrK.d.ts} +3 -3
- package/dist/{summary-report-gMrbYawB.d.ts.map → summary-report-D1h4dlrK.d.ts.map} +1 -1
- package/dist/{summary-report-B16xy9Kd.js → summary-report-e-MaOAHV.js} +2 -2
- package/dist/{summary-report-B16xy9Kd.js.map → summary-report-e-MaOAHV.js.map} +1 -1
- package/dist/{tool-groups-2QA0S7dK.d.ts → tool-groups-B2bSNaJB.d.ts} +3 -3
- package/dist/tool-groups-B2bSNaJB.d.ts.map +1 -0
- package/dist/{tool-waste-B9tdWV6g.js → tool-waste-C7MU9u1e.js} +2 -2
- package/dist/{tool-waste-B9tdWV6g.js.map → tool-waste-C7MU9u1e.js.map} +1 -1
- package/dist/trace-repair/index.d.ts +2 -2
- package/dist/traces.d.ts +6 -6
- package/dist/traces.js +1 -1
- package/dist/{types-BmlkCrg0.d.ts → types-BvZoPTGa.d.ts} +3 -3
- package/dist/{types-BmlkCrg0.d.ts.map → types-BvZoPTGa.d.ts.map} +1 -1
- package/dist/{types-gvRsyJLh.d.ts → types-CBbLtr2J.d.ts} +38 -3
- package/dist/{types-gvRsyJLh.d.ts.map → types-CBbLtr2J.d.ts.map} +1 -1
- package/dist/{types-C34V4Vto.d.ts → types-CS0qk_Yp.d.ts} +4 -4
- package/dist/{types-C34V4Vto.d.ts.map → types-CS0qk_Yp.d.ts.map} +1 -1
- package/dist/{types-DzuaM493.d.ts → types-D7gEdPoQ.d.ts} +3 -3
- package/dist/{types-DzuaM493.d.ts.map → types-D7gEdPoQ.d.ts.map} +1 -1
- package/dist/wire/index.d.ts +2 -2
- package/docs/adapters-observability.md +14 -0
- package/docs/campaign-proposers.md +86 -128
- package/docs/charter.md +108 -112
- package/docs/concepts.md +157 -69
- package/docs/design/mlbenchmarks-book-review.md +440 -0
- package/docs/design/mlbenchmarks-review/observations.json +713 -0
- package/docs/design/mlbenchmarks-review/probes.mts +476 -0
- package/docs/design/mlbenchmarks-review/sources.json +200 -0
- package/docs/design/self-improvement-evidence-audit.md +263 -0
- package/docs/design.md +2 -1
- package/docs/eval-surface-map.md +95 -42
- package/docs/evaluation-integrity.md +220 -0
- package/docs/experiment.md +111 -55
- package/docs/feature-guide.md +5 -6
- package/docs/hosted-ingest-spec.md +4 -11
- package/docs/insight-report.md +187 -455
- package/docs/outcome-validity.md +182 -0
- package/docs/product-eval-adoption.md +1 -2
- package/docs/research-report-methodology.md +7 -7
- package/docs/search-history-receipts.md +8 -0
- package/docs/statistical-evidence.md +129 -0
- package/docs/verdicts.md +76 -49
- package/package.json +1 -1
- package/dist/agent-profile-cell-0gSi5ffD.js.map +0 -1
- package/dist/agent-profile-cell-CTOZJUuE.d.ts.map +0 -1
- package/dist/campaign-evidence-D8DBLqLI.js.map +0 -1
- package/dist/client-BlLY6o2w.js.map +0 -1
- package/dist/define-agent-eval-Cx4Ls9ta.d.ts.map +0 -1
- package/dist/define-agent-eval-Dzidv34q.js.map +0 -1
- package/dist/engine-CX8ReXkn.d.ts.map +0 -1
- package/dist/index-BxWvILU8.d.ts.map +0 -1
- package/dist/judge-calibration-C5CbMYce.d.ts.map +0 -1
- package/dist/ledger-core-Cs9f7385.js.map +0 -1
- package/dist/llm-judge-v80Kmu9g.js.map +0 -1
- package/dist/outcome-store-BYHIuO0e.d.ts.map +0 -1
- package/dist/outcome-store-ChBKlTd_.js +0 -75
- package/dist/outcome-store-ChBKlTd_.js.map +0 -1
- package/dist/promotion-policy-DWOm70gx.js.map +0 -1
- package/dist/rubric-predictive-validity-2D5Gw9z9.js +0 -131
- package/dist/rubric-predictive-validity-2D5Gw9z9.js.map +0 -1
- package/dist/rubric-predictive-validity-Dl1dvKCv.d.ts +0 -75
- package/dist/rubric-predictive-validity-Dl1dvKCv.d.ts.map +0 -1
- package/dist/run-record-CR63CpHK.js +0 -216
- package/dist/run-record-CR63CpHK.js.map +0 -1
- package/dist/sequential-B5gXgcyp.js.map +0 -1
- package/dist/statistical-heldout-0La5ZTlv.d.ts.map +0 -1
- package/dist/tool-groups-2QA0S7dK.d.ts.map +0 -1
|
@@ -0,0 +1,557 @@
|
|
|
1
|
+
import { s as ValidationError } from "./errors-Dngq5h35.js";
|
|
2
|
+
import { t as mulberry32 } from "./random-Dn5fPWkt.js";
|
|
3
|
+
//#region src/experiment/ast.ts
|
|
4
|
+
/**
|
|
5
|
+
* The registered-rule AST: every rule an experiment registers is DATA.
|
|
6
|
+
*
|
|
7
|
+
* A closure cannot be canonicalized or hashed; a node tree can. `sealExperiment`
|
|
8
|
+
* hashes the whole tree, and every interpreter in this file takes only a node
|
|
9
|
+
* plus evidence records — no parameter for alpha, threshold, metric, or
|
|
10
|
+
* stopping rule exists on any executable surface. The registered object and
|
|
11
|
+
* the executed object are therefore the same object, and registered-vs-ran
|
|
12
|
+
* drift is unrepresentable rather than checked.
|
|
13
|
+
*
|
|
14
|
+
* Node families:
|
|
15
|
+
* Predicate closed-key comparisons — the only leaf
|
|
16
|
+
* AdmissionRule monotone funnel stages with registered waivers
|
|
17
|
+
* SelectionRule deterministic subsets over a closed field set
|
|
18
|
+
* Estimand what the experiment measures
|
|
19
|
+
* IntervalSpec how uncertainty is computed, seed included
|
|
20
|
+
* Condition decision guards over named derived quantities
|
|
21
|
+
* DecisionRule ordered verdict table, or a registered absence of one
|
|
22
|
+
* Obligation a control that must exist before a verdict class is read
|
|
23
|
+
* ValidityGate pre-spend design checks
|
|
24
|
+
* HaltRule gates as prerequisites — failure refuses the spend
|
|
25
|
+
* BudgetRule spend schedules with a named ledger
|
|
26
|
+
* MatchedBudgetRule arm budget matching as a refusal
|
|
27
|
+
* ReissuePolicy carrier faults are reissued; model outcomes stand
|
|
28
|
+
*/
|
|
29
|
+
/** A decision rule's branches did not cover the evidence. */
|
|
30
|
+
var DecisionTableNotTotalError = class extends ValidationError {};
|
|
31
|
+
/** Read a dot-separated field path. Missing segments yield `undefined`. */
|
|
32
|
+
function readField(record, path) {
|
|
33
|
+
let current = record;
|
|
34
|
+
for (const key of path.split(".")) {
|
|
35
|
+
if (current === null || typeof current !== "object") return void 0;
|
|
36
|
+
current = current[key];
|
|
37
|
+
}
|
|
38
|
+
return current;
|
|
39
|
+
}
|
|
40
|
+
/** Evaluate a predicate against one evidence record. */
|
|
41
|
+
function evaluatePredicate(predicate, record) {
|
|
42
|
+
switch (predicate.kind) {
|
|
43
|
+
case "compare": return compareValues(readField(record, predicate.field), predicate.op, predicate.value);
|
|
44
|
+
case "in": return predicate.values.includes(readField(record, predicate.field));
|
|
45
|
+
case "all": return predicate.of.every((p) => evaluatePredicate(p, record));
|
|
46
|
+
case "any": return predicate.of.some((p) => evaluatePredicate(p, record));
|
|
47
|
+
case "not": return !evaluatePredicate(predicate.of, record);
|
|
48
|
+
}
|
|
49
|
+
}
|
|
50
|
+
/** Numbers compare numerically; everything else compares as strings. */
|
|
51
|
+
function compareValues(value, op, target) {
|
|
52
|
+
if (op === "eq") return value === target;
|
|
53
|
+
if (op === "ne") return value !== target;
|
|
54
|
+
const numeric = typeof value === "number" && typeof target === "number";
|
|
55
|
+
const left = numeric ? value : String(value);
|
|
56
|
+
const right = numeric ? target : String(target);
|
|
57
|
+
switch (op) {
|
|
58
|
+
case "lt": return left < right;
|
|
59
|
+
case "lte": return left <= right;
|
|
60
|
+
case "gt": return left > right;
|
|
61
|
+
case "gte": return left >= right;
|
|
62
|
+
}
|
|
63
|
+
}
|
|
64
|
+
/**
|
|
65
|
+
* Execute a selection rule.
|
|
66
|
+
*
|
|
67
|
+
* Round-robin walks groups in lexicographic order and takes ids in
|
|
68
|
+
* within-group order until `take` ids are chosen. Filter-of keeps the ids of
|
|
69
|
+
* `bases[rule.base]` whose record satisfies the predicate, in the registered
|
|
70
|
+
* order. Ids absent from `records` are evaluated on their id alone (fields
|
|
71
|
+
* derived from the id via `idFields`), so a sealed base outlives its source
|
|
72
|
+
* records.
|
|
73
|
+
*/
|
|
74
|
+
function runSelectionRule(rule, records, options) {
|
|
75
|
+
if (rule.kind === "round-robin") {
|
|
76
|
+
const allowed = new Set(rule.reads);
|
|
77
|
+
for (const field of [rule.groupBy, rule.withinOrder.field]) if (!allowed.has(field)) throw new ValidationError(`runSelectionRule: round-robin reads '${field}' but its closed read set is [${rule.reads.join(", ")}]`);
|
|
78
|
+
const byGroup = /* @__PURE__ */ new Map();
|
|
79
|
+
for (const record of records) {
|
|
80
|
+
const group = String(readField(record, rule.groupBy));
|
|
81
|
+
const id = String(readField(record, rule.withinOrder.field));
|
|
82
|
+
const bucket = byGroup.get(group);
|
|
83
|
+
if (bucket) bucket.push(id);
|
|
84
|
+
else byGroup.set(group, [id]);
|
|
85
|
+
}
|
|
86
|
+
for (const ids of byGroup.values()) {
|
|
87
|
+
ids.sort();
|
|
88
|
+
if (rule.withinOrder.dir === "desc") ids.reverse();
|
|
89
|
+
}
|
|
90
|
+
const groups = [...byGroup.keys()].sort();
|
|
91
|
+
const chosen = [];
|
|
92
|
+
let cursor = 0;
|
|
93
|
+
while (chosen.length < rule.take && groups.some((g) => byGroup.get(g).length > 0)) {
|
|
94
|
+
const group = groups[cursor % groups.length];
|
|
95
|
+
const ids = byGroup.get(group);
|
|
96
|
+
if (ids.length > 0) chosen.push(ids.shift());
|
|
97
|
+
cursor += 1;
|
|
98
|
+
}
|
|
99
|
+
return chosen;
|
|
100
|
+
}
|
|
101
|
+
const base = options.bases?.[rule.base];
|
|
102
|
+
if (!base) throw new ValidationError(`runSelectionRule: filter-of base '${rule.base}' was not provided`);
|
|
103
|
+
const index = new Map(records.map((r) => [String(readField(r, options.idField)), r]));
|
|
104
|
+
const sorted = [...base.filter((id) => {
|
|
105
|
+
const record = index.get(id) ?? options.idFields?.(id);
|
|
106
|
+
if (!record) throw new ValidationError(`runSelectionRule: base id '${id}' has no record and no idFields derivation`);
|
|
107
|
+
return evaluatePredicate(rule.keep, record);
|
|
108
|
+
})].sort();
|
|
109
|
+
if (rule.order.dir === "desc") sorted.reverse();
|
|
110
|
+
return sorted;
|
|
111
|
+
}
|
|
112
|
+
function evaluateSetExpr(expr, rows, armField, idField) {
|
|
113
|
+
if (expr.kind === "rows-where") {
|
|
114
|
+
const ids = /* @__PURE__ */ new Set();
|
|
115
|
+
for (const row of rows) {
|
|
116
|
+
if (String(readField(row, armField)) !== expr.arm) continue;
|
|
117
|
+
if (evaluatePredicate(expr.event, row)) ids.add(String(readField(row, idField)));
|
|
118
|
+
}
|
|
119
|
+
return ids;
|
|
120
|
+
}
|
|
121
|
+
if (expr.of.length === 0) throw new ValidationError("computeEstimand: empty intersect");
|
|
122
|
+
const [first, ...rest] = expr.of.map((e) => evaluateSetExpr(e, rows, armField, idField));
|
|
123
|
+
const out = /* @__PURE__ */ new Set();
|
|
124
|
+
for (const id of first) if (rest.every((s) => s.has(id))) out.add(id);
|
|
125
|
+
return out;
|
|
126
|
+
}
|
|
127
|
+
/**
|
|
128
|
+
* Read one registered outcome field as a number.
|
|
129
|
+
*
|
|
130
|
+
* A binary outcome reaches evidence as `true` or `false`. Its mean is the pass
|
|
131
|
+
* rate and the mean of its paired differences is the risk difference, so
|
|
132
|
+
* `true` reads as 1 and `false` as 0 — the same quantity a caller would
|
|
133
|
+
* otherwise encode by hand, and the same reading in every interpreter here.
|
|
134
|
+
* Every other type, and a non-finite number, is a measurement defect: it
|
|
135
|
+
* rejects instead of poisoning the mean with `NaN` or a coerced zero.
|
|
136
|
+
*/
|
|
137
|
+
function readNumericOutcome(raw, context, field, where) {
|
|
138
|
+
if (typeof raw === "boolean") return raw ? 1 : 0;
|
|
139
|
+
if (typeof raw === "number" && Number.isFinite(raw)) return raw;
|
|
140
|
+
throw new ValidationError(`${context}: value field '${field}' is not a finite number or a boolean on ${where}`);
|
|
141
|
+
}
|
|
142
|
+
/** Compute an estimand over evidence rows. Pure; reads only registered fields. */
|
|
143
|
+
function computeEstimand(estimand, rows) {
|
|
144
|
+
switch (estimand.kind) {
|
|
145
|
+
case "rate": {
|
|
146
|
+
const numerator = rows.filter((r) => evaluatePredicate(estimand.event, r)).length;
|
|
147
|
+
if (rows.length === 0) throw new ValidationError("computeEstimand: rate over zero rows");
|
|
148
|
+
return {
|
|
149
|
+
value: numerator / rows.length,
|
|
150
|
+
numerator,
|
|
151
|
+
denominator: rows.length
|
|
152
|
+
};
|
|
153
|
+
}
|
|
154
|
+
case "rate-at-least-once": {
|
|
155
|
+
const byGroup = /* @__PURE__ */ new Map();
|
|
156
|
+
for (const row of rows) {
|
|
157
|
+
const group = String(readField(row, estimand.groupBy));
|
|
158
|
+
const hit = evaluatePredicate(estimand.event, row);
|
|
159
|
+
byGroup.set(group, (byGroup.get(group) ?? false) || hit);
|
|
160
|
+
}
|
|
161
|
+
if (byGroup.size === 0) throw new ValidationError("computeEstimand: rate-at-least-once over zero groups");
|
|
162
|
+
const numerator = [...byGroup.values()].filter(Boolean).length;
|
|
163
|
+
return {
|
|
164
|
+
value: numerator / byGroup.size,
|
|
165
|
+
numerator,
|
|
166
|
+
denominator: byGroup.size
|
|
167
|
+
};
|
|
168
|
+
}
|
|
169
|
+
case "paired-mean-diff": {
|
|
170
|
+
const byPair = /* @__PURE__ */ new Map();
|
|
171
|
+
for (const row of rows) {
|
|
172
|
+
const arm = String(readField(row, estimand.armField));
|
|
173
|
+
if (arm !== estimand.treatment && arm !== estimand.control) continue;
|
|
174
|
+
const pair = String(readField(row, estimand.pairBy));
|
|
175
|
+
const value = readNumericOutcome(readField(row, estimand.value), "computeEstimand paired-mean-diff", estimand.value, `pair '${pair}'`);
|
|
176
|
+
const slot = byPair.get(pair) ?? {};
|
|
177
|
+
if (arm === estimand.treatment) slot.treatment = value;
|
|
178
|
+
else slot.control = value;
|
|
179
|
+
byPair.set(pair, slot);
|
|
180
|
+
}
|
|
181
|
+
if (byPair.size === 0) throw new ValidationError("computeEstimand: paired-mean-diff over zero pairs");
|
|
182
|
+
let sum = 0;
|
|
183
|
+
for (const slot of byPair.values()) sum += (slot.treatment ?? 0) - (slot.control ?? 0);
|
|
184
|
+
return {
|
|
185
|
+
value: sum / byPair.size,
|
|
186
|
+
numerator: sum,
|
|
187
|
+
denominator: byPair.size
|
|
188
|
+
};
|
|
189
|
+
}
|
|
190
|
+
case "set-ratio": {
|
|
191
|
+
const numeratorSet = evaluateSetExpr(estimand.numerator, rows, estimand.armField, estimand.idField);
|
|
192
|
+
const denominatorSet = evaluateSetExpr(estimand.denominator, rows, estimand.armField, estimand.idField);
|
|
193
|
+
if (denominatorSet.size === 0) throw new ValidationError("computeEstimand: set-ratio denominator set is empty");
|
|
194
|
+
return {
|
|
195
|
+
value: numeratorSet.size / denominatorSet.size,
|
|
196
|
+
numerator: numeratorSet.size,
|
|
197
|
+
denominator: denominatorSet.size
|
|
198
|
+
};
|
|
199
|
+
}
|
|
200
|
+
}
|
|
201
|
+
}
|
|
202
|
+
/** Shared by registration and direct execution so malformed intervals cannot produce evidence. */
|
|
203
|
+
function intervalSpecProblems(spec) {
|
|
204
|
+
if (spec === null || typeof spec !== "object" || Array.isArray(spec)) return ["interval must be an object"];
|
|
205
|
+
if (spec.kind !== "cluster-bootstrap" && spec.kind !== "clopper-pearson") return ["interval kind must be cluster-bootstrap or clopper-pearson"];
|
|
206
|
+
const problems = [];
|
|
207
|
+
if (!Number.isFinite(spec.level) || spec.level <= 0 || spec.level >= 1) problems.push("level must be a finite number between zero and one");
|
|
208
|
+
if (spec.kind === "cluster-bootstrap") {
|
|
209
|
+
for (const field of ["clusterBy", "value"]) {
|
|
210
|
+
const path = spec[field];
|
|
211
|
+
if (typeof path !== "string" || path.split(".").some((segment) => segment.length === 0 || segment.trim() !== segment)) problems.push(`${field} must be a nonempty dot-separated field path`);
|
|
212
|
+
}
|
|
213
|
+
if (!Number.isSafeInteger(spec.resamples) || spec.resamples < 1 || spec.resamples > 4294967295) problems.push("resamples must be a positive integer within the supported array length");
|
|
214
|
+
if (!Number.isSafeInteger(spec.seed)) problems.push("seed must be a safe integer");
|
|
215
|
+
if (spec.method !== "percentile") problems.push("method must be percentile");
|
|
216
|
+
}
|
|
217
|
+
return problems;
|
|
218
|
+
}
|
|
219
|
+
/**
|
|
220
|
+
* Execute an interval spec.
|
|
221
|
+
*
|
|
222
|
+
* Cluster-bootstrap resamples whole clusters of its registered `value` field and
|
|
223
|
+
* takes percentile bounds of the pooled mean. Clopper-Pearson computes the
|
|
224
|
+
* exact binomial interval and requires `successes`/`trials` evidence instead
|
|
225
|
+
* of rows.
|
|
226
|
+
*/
|
|
227
|
+
function computeInterval(spec, evidence) {
|
|
228
|
+
const problems = intervalSpecProblems(spec);
|
|
229
|
+
if (problems.length > 0) throw new ValidationError(`computeInterval: ${problems.join("; ")}`);
|
|
230
|
+
if (evidence === null || typeof evidence !== "object" || Array.isArray(evidence)) throw new ValidationError("computeInterval: evidence must be an object");
|
|
231
|
+
if (spec.kind === "cluster-bootstrap") {
|
|
232
|
+
if (evidence.kind !== "rows") throw new ValidationError("computeInterval: cluster-bootstrap requires row evidence");
|
|
233
|
+
if ("value" in evidence) throw new ValidationError("computeInterval: register value in the interval spec, not row evidence");
|
|
234
|
+
if (!Array.isArray(evidence.rows)) throw new ValidationError("computeInterval: rows must be an array");
|
|
235
|
+
const clusters = /* @__PURE__ */ new Map();
|
|
236
|
+
for (const [index, row] of evidence.rows.entries()) {
|
|
237
|
+
if (row === null || typeof row !== "object" || Array.isArray(row)) throw new ValidationError(`computeInterval: row ${index} must be an object`);
|
|
238
|
+
const cluster = readField(row, spec.clusterBy);
|
|
239
|
+
if (!(typeof cluster === "string" && cluster.length > 0 && cluster.trim() === cluster) && !(typeof cluster === "number" && Number.isFinite(cluster))) throw new ValidationError(`computeInterval: cluster field '${spec.clusterBy}' must be a nonempty string or finite number on row ${index}`);
|
|
240
|
+
const clusterKey = `${typeof cluster}:${cluster}`;
|
|
241
|
+
const value = readNumericOutcome(readField(row, spec.value), "computeInterval cluster-bootstrap", spec.value, `cluster '${cluster}'`);
|
|
242
|
+
const bucket = clusters.get(clusterKey);
|
|
243
|
+
if (bucket) bucket.push(value);
|
|
244
|
+
else clusters.set(clusterKey, [value]);
|
|
245
|
+
}
|
|
246
|
+
const clusterValues = [...clusters.entries()].sort(([a], [b]) => a < b ? -1 : a > b ? 1 : 0).map(([, values]) => values);
|
|
247
|
+
if (clusterValues.length < 2) throw new ValidationError(`computeInterval: cluster-bootstrap needs >= 2 clusters, got ${clusterValues.length}`);
|
|
248
|
+
const rng = mulberry32(spec.seed);
|
|
249
|
+
const means = new Array(spec.resamples);
|
|
250
|
+
for (let draw = 0; draw < spec.resamples; draw++) {
|
|
251
|
+
let sum = 0;
|
|
252
|
+
let count = 0;
|
|
253
|
+
for (let pick = 0; pick < clusterValues.length; pick++) {
|
|
254
|
+
const cluster = clusterValues[Math.floor(rng() * clusterValues.length)];
|
|
255
|
+
for (const value of cluster) sum += value;
|
|
256
|
+
count += cluster.length;
|
|
257
|
+
}
|
|
258
|
+
means[draw] = sum / count;
|
|
259
|
+
if (!Number.isFinite(means[draw])) throw new ValidationError("computeInterval: cluster mean overflowed the finite numeric range");
|
|
260
|
+
}
|
|
261
|
+
means.sort((a, b) => a - b);
|
|
262
|
+
const alpha = 1 - spec.level;
|
|
263
|
+
const lowerIndex = Math.floor(alpha / 2 * spec.resamples);
|
|
264
|
+
const upperIndex = Math.min(spec.resamples - 1, Math.ceil((1 - alpha / 2) * spec.resamples) - 1);
|
|
265
|
+
return {
|
|
266
|
+
lower: means[lowerIndex],
|
|
267
|
+
upper: means[Math.max(lowerIndex, upperIndex)],
|
|
268
|
+
level: spec.level
|
|
269
|
+
};
|
|
270
|
+
}
|
|
271
|
+
if (evidence.kind !== "binomial") throw new ValidationError("computeInterval: clopper-pearson requires binomial evidence");
|
|
272
|
+
const { successes, trials } = evidence;
|
|
273
|
+
if (!Number.isSafeInteger(successes) || !Number.isSafeInteger(trials) || trials <= 0 || successes < 0) throw new ValidationError(`computeInterval: clopper-pearson needs 0 <= successes <= trials, got ${successes}/${trials}`);
|
|
274
|
+
if (successes > trials) throw new ValidationError(`computeInterval: clopper-pearson successes ${successes} exceed trials ${trials}`);
|
|
275
|
+
const alpha = 1 - spec.level;
|
|
276
|
+
return {
|
|
277
|
+
lower: successes === 0 ? 0 : binomialQuantile(successes, trials, alpha / 2, "lower"),
|
|
278
|
+
upper: successes === trials ? 1 : binomialQuantile(successes, trials, alpha / 2, "upper"),
|
|
279
|
+
level: spec.level
|
|
280
|
+
};
|
|
281
|
+
}
|
|
282
|
+
/**
|
|
283
|
+
* Clopper-Pearson bound by bisection on the binomial tail. The lower bound is
|
|
284
|
+
* the p with P(X >= successes | p) = alpha; the upper is the p with
|
|
285
|
+
* P(X <= successes | p) = alpha. Deterministic, no special functions.
|
|
286
|
+
*/
|
|
287
|
+
function binomialQuantile(successes, trials, alpha, side) {
|
|
288
|
+
const tail = (p) => {
|
|
289
|
+
let sum = 0;
|
|
290
|
+
for (let k = 0; k <= trials; k++) {
|
|
291
|
+
if (!(side === "lower" ? k >= successes : k <= successes)) continue;
|
|
292
|
+
sum += Math.exp(logBinomialPmf(k, trials, p));
|
|
293
|
+
}
|
|
294
|
+
return sum;
|
|
295
|
+
};
|
|
296
|
+
let lo = 0;
|
|
297
|
+
let hi = 1;
|
|
298
|
+
for (let iter = 0; iter < 100; iter++) {
|
|
299
|
+
const mid = (lo + hi) / 2;
|
|
300
|
+
if (tail(mid) < alpha) if (side === "lower") lo = mid;
|
|
301
|
+
else hi = mid;
|
|
302
|
+
else if (side === "lower") hi = mid;
|
|
303
|
+
else lo = mid;
|
|
304
|
+
}
|
|
305
|
+
return (lo + hi) / 2;
|
|
306
|
+
}
|
|
307
|
+
function logBinomialPmf(k, n, p) {
|
|
308
|
+
if (p <= 0) return k === 0 ? 0 : Number.NEGATIVE_INFINITY;
|
|
309
|
+
if (p >= 1) return k === n ? 0 : Number.NEGATIVE_INFINITY;
|
|
310
|
+
return logChoose(n, k) + k * Math.log(p) + (n - k) * Math.log(1 - p);
|
|
311
|
+
}
|
|
312
|
+
function logChoose(n, k) {
|
|
313
|
+
return logFactorial(n) - logFactorial(k) - logFactorial(n - k);
|
|
314
|
+
}
|
|
315
|
+
const LOG_FACTORIAL_CACHE = [0];
|
|
316
|
+
function logFactorial(n) {
|
|
317
|
+
for (let i = LOG_FACTORIAL_CACHE.length; i <= n; i++) LOG_FACTORIAL_CACHE[i] = LOG_FACTORIAL_CACHE[i - 1] + Math.log(i);
|
|
318
|
+
return LOG_FACTORIAL_CACHE[n];
|
|
319
|
+
}
|
|
320
|
+
function evaluateCondition(condition, evidence) {
|
|
321
|
+
switch (condition.kind) {
|
|
322
|
+
case "interval-excludes-zero": {
|
|
323
|
+
const interval = evidence.intervals[condition.interval];
|
|
324
|
+
if (!interval) throw new ValidationError(`evaluateCondition: interval '${condition.interval}' is not in the evidence`);
|
|
325
|
+
return (interval.lower > 0 || interval.upper < 0) && (condition.sign === "positive" ? interval.lower > 0 : interval.upper < 0);
|
|
326
|
+
}
|
|
327
|
+
case "interval-includes-zero": {
|
|
328
|
+
const interval = evidence.intervals[condition.interval];
|
|
329
|
+
if (!interval) throw new ValidationError(`evaluateCondition: interval '${condition.interval}' is not in the evidence`);
|
|
330
|
+
return interval.lower <= 0 && interval.upper >= 0;
|
|
331
|
+
}
|
|
332
|
+
case "quantity-threshold": {
|
|
333
|
+
const value = evidence.quantities[condition.quantity];
|
|
334
|
+
if (value === void 0) throw new ValidationError(`evaluateCondition: quantity '${condition.quantity}' is not in the evidence`);
|
|
335
|
+
return compareValues(value, condition.op, condition.value);
|
|
336
|
+
}
|
|
337
|
+
case "obligation-met": return evidence.obligationsMet[condition.obligation] === true;
|
|
338
|
+
case "all": return condition.of.every((c) => evaluateCondition(c, evidence));
|
|
339
|
+
case "any": return condition.of.some((c) => evaluateCondition(c, evidence));
|
|
340
|
+
case "not": return !evaluateCondition(condition.of, evidence);
|
|
341
|
+
}
|
|
342
|
+
}
|
|
343
|
+
function executeDecisionRule(rule, evidence) {
|
|
344
|
+
if (rule.kind === "report-only") return {
|
|
345
|
+
verdict: "report-only",
|
|
346
|
+
report: [...rule.estimands, ...rule.intervals]
|
|
347
|
+
};
|
|
348
|
+
for (const branch of rule.branches) if (evaluateCondition(branch.when, evidence)) return {
|
|
349
|
+
verdict: branch.verdict,
|
|
350
|
+
report: branch.report
|
|
351
|
+
};
|
|
352
|
+
throw new DecisionTableNotTotalError("executeDecisionRule: decision table is not total — no branch matched the evidence");
|
|
353
|
+
}
|
|
354
|
+
/**
|
|
355
|
+
* Replicate-flip counting over graded states. A state whose replicates split
|
|
356
|
+
* between pass and fail is flipping; its flip rate is the minority share.
|
|
357
|
+
*/
|
|
358
|
+
function evaluateOracleDeterminismGate(id, gate, repsByState) {
|
|
359
|
+
const evidence = {};
|
|
360
|
+
let passed = true;
|
|
361
|
+
for (const [state, reps] of Object.entries(repsByState)) {
|
|
362
|
+
const passes = reps.filter(Boolean).length;
|
|
363
|
+
const flipRate = reps.length === 0 ? 0 : Math.min(passes, reps.length - passes) / reps.length;
|
|
364
|
+
evidence[state] = {
|
|
365
|
+
passes,
|
|
366
|
+
replicates: reps.length,
|
|
367
|
+
flipRate
|
|
368
|
+
};
|
|
369
|
+
if (flipRate > gate.maxFlipRate) passed = false;
|
|
370
|
+
}
|
|
371
|
+
return {
|
|
372
|
+
id,
|
|
373
|
+
passed,
|
|
374
|
+
evidence
|
|
375
|
+
};
|
|
376
|
+
}
|
|
377
|
+
/**
|
|
378
|
+
* Join two population snapshots on `joinOn` and compare the registered fields.
|
|
379
|
+
* Only rows present in both snapshots are compared; a presence change is a
|
|
380
|
+
* different failure and needs its own gate.
|
|
381
|
+
*/
|
|
382
|
+
function evaluatePopulationReproducibilityGate(id, gate, populations) {
|
|
383
|
+
const rightByKey = new Map(populations.right.map((r) => [String(readField(r, gate.joinOn)), r]));
|
|
384
|
+
const changed = [];
|
|
385
|
+
for (const left of populations.left) {
|
|
386
|
+
const key = String(readField(left, gate.joinOn));
|
|
387
|
+
const right = rightByKey.get(key);
|
|
388
|
+
if (!right) continue;
|
|
389
|
+
const moved = gate.compare.filter((f) => readField(left, f) !== readField(right, f));
|
|
390
|
+
if (moved.length > 0) changed.push(`${key} ${moved.map((f) => `${f}:${String(readField(left, f))}->${String(readField(right, f))}`).join(" ")}`);
|
|
391
|
+
}
|
|
392
|
+
return {
|
|
393
|
+
id,
|
|
394
|
+
passed: changed.length <= gate.maxChangedRows,
|
|
395
|
+
evidence: changed
|
|
396
|
+
};
|
|
397
|
+
}
|
|
398
|
+
/** The registered claim about provenance must hold on the provenance record. */
|
|
399
|
+
function evaluateProvenanceGate(id, gate, provenance) {
|
|
400
|
+
const passed = evaluatePredicate(gate.claim, provenance);
|
|
401
|
+
return {
|
|
402
|
+
id,
|
|
403
|
+
passed,
|
|
404
|
+
evidence: { claimHolds: passed }
|
|
405
|
+
};
|
|
406
|
+
}
|
|
407
|
+
/** Final path segment equality between the pinned and the served identity. */
|
|
408
|
+
function evaluateIdentityGate(id, _gate, identities) {
|
|
409
|
+
const basename = (s) => s.split("/").pop() ?? s;
|
|
410
|
+
const passed = basename(identities.pinned) === basename(identities.served);
|
|
411
|
+
return {
|
|
412
|
+
id,
|
|
413
|
+
passed,
|
|
414
|
+
evidence: {
|
|
415
|
+
pinned: identities.pinned,
|
|
416
|
+
served: identities.served,
|
|
417
|
+
matched: passed
|
|
418
|
+
}
|
|
419
|
+
};
|
|
420
|
+
}
|
|
421
|
+
/**
|
|
422
|
+
* The design's power at minimumEffect must reach the registered target.
|
|
423
|
+
* The curve must cover the registered effect grid exactly — a curve
|
|
424
|
+
* computed on a different grid is different evidence and is refused.
|
|
425
|
+
*/
|
|
426
|
+
function evaluatePowerFloorGate(id, gate, curve) {
|
|
427
|
+
const problems = powerFloorProblems(gate);
|
|
428
|
+
if (problems.length > 0) throw new ValidationError(`evaluatePowerFloorGate: ${problems.join("; ")}`);
|
|
429
|
+
if (curve.some((point) => !Number.isFinite(point.power) || point.power < 0 || point.power > 1)) throw new ValidationError("evaluatePowerFloorGate: powers must be finite and in [0,1]");
|
|
430
|
+
const byEffect = new Map(curve.map((point) => [point.effect, point.power]));
|
|
431
|
+
if (byEffect.size !== curve.length) throw new ValidationError("evaluatePowerFloorGate: curve contains duplicate effects");
|
|
432
|
+
const missing = gate.effectGrid.filter((effect) => !byEffect.has(effect));
|
|
433
|
+
if (missing.length > 0) throw new ValidationError(`evaluatePowerFloorGate: curve does not cover registered effects [${missing.join(", ")}]`);
|
|
434
|
+
const extra = curve.filter((point) => !gate.effectGrid.includes(point.effect));
|
|
435
|
+
if (extra.length > 0) throw new ValidationError(`evaluatePowerFloorGate: curve contains unregistered effects [${extra.map((point) => point.effect).join(", ")}]`);
|
|
436
|
+
const powers = gate.effectGrid.map((effect) => byEffect.get(effect));
|
|
437
|
+
const maxPower = Math.max(...powers);
|
|
438
|
+
const powerAtMinimumEffect = byEffect.get(gate.minimumEffect);
|
|
439
|
+
return {
|
|
440
|
+
id,
|
|
441
|
+
passed: powerAtMinimumEffect >= gate.target,
|
|
442
|
+
evidence: {
|
|
443
|
+
target: gate.target,
|
|
444
|
+
minimumEffect: gate.minimumEffect,
|
|
445
|
+
powerAtMinimumEffect,
|
|
446
|
+
maxPower,
|
|
447
|
+
curve: gate.effectGrid.map((effect) => ({
|
|
448
|
+
effect,
|
|
449
|
+
power: byEffect.get(effect)
|
|
450
|
+
}))
|
|
451
|
+
}
|
|
452
|
+
};
|
|
453
|
+
}
|
|
454
|
+
/** Shared by seal validation and direct execution; malformed designs never pass. */
|
|
455
|
+
function powerFloorProblems(gate) {
|
|
456
|
+
const problems = [];
|
|
457
|
+
if (!Number.isFinite(gate.target) || gate.target <= 0 || gate.target > 1) problems.push("power target must be in (0,1]");
|
|
458
|
+
if (!Number.isFinite(gate.minimumEffect) || gate.minimumEffect <= 0) problems.push("minimumEffect must be positive and finite");
|
|
459
|
+
if (gate.effectGrid.length === 0 || gate.effectGrid.some((effect) => !Number.isFinite(effect))) problems.push("effectGrid must contain finite effects");
|
|
460
|
+
if (new Set(gate.effectGrid).size !== gate.effectGrid.length) problems.push("effectGrid contains duplicate effects");
|
|
461
|
+
if (!gate.effectGrid.includes(gate.minimumEffect)) problems.push("effectGrid must contain minimumEffect exactly; no interpolation is assumed");
|
|
462
|
+
for (const field of ["trials", "resamples"]) if (!Number.isInteger(gate.sim[field]) || gate.sim[field] <= 0) problems.push(`power simulation ${field} must be a positive integer`);
|
|
463
|
+
if (!Number.isInteger(gate.sim.seed)) problems.push("power simulation seed must be an integer");
|
|
464
|
+
return problems;
|
|
465
|
+
}
|
|
466
|
+
function evaluateHaltRule(halt, gates) {
|
|
467
|
+
const seen = new Map(gates.map((g) => [g.id, g]));
|
|
468
|
+
const missing = halt.when.gates.filter((id) => !seen.has(id));
|
|
469
|
+
if (missing.length > 0) throw new ValidationError(`evaluateHaltRule: halt references gates that were not evaluated: [${missing.join(", ")}]`);
|
|
470
|
+
const failed = halt.when.gates.filter((id) => !seen.get(id).passed);
|
|
471
|
+
return failed.length > 0 ? {
|
|
472
|
+
fired: true,
|
|
473
|
+
action: halt.action,
|
|
474
|
+
failedGates: failed
|
|
475
|
+
} : {
|
|
476
|
+
fired: false,
|
|
477
|
+
action: null,
|
|
478
|
+
failedGates: []
|
|
479
|
+
};
|
|
480
|
+
}
|
|
481
|
+
/**
|
|
482
|
+
* Execute the uniform-pass schedule against measured pass costs. Pass 1 always
|
|
483
|
+
* runs; each later pass runs only when the cumulative spend plus the last
|
|
484
|
+
* measured pass cost stays at or under the registered ceiling. The registered
|
|
485
|
+
* ledger is the pre-spend the ceiling counts.
|
|
486
|
+
*/
|
|
487
|
+
function runUniformPassBudget(rule, measuredPassCosts) {
|
|
488
|
+
let cumulative = rule.ledger.reduce((sum, entry) => sum + entry.usd, 0);
|
|
489
|
+
const decisions = [];
|
|
490
|
+
let uniformN = 0;
|
|
491
|
+
for (let pass = 1; pass <= rule.maxPasses; pass++) {
|
|
492
|
+
if (pass === 1) {
|
|
493
|
+
if (measuredPassCosts[0] === void 0) break;
|
|
494
|
+
cumulative += measuredPassCosts[0];
|
|
495
|
+
uniformN = 1;
|
|
496
|
+
continue;
|
|
497
|
+
}
|
|
498
|
+
const projected = measuredPassCosts[pass - 2];
|
|
499
|
+
if (projected === void 0) break;
|
|
500
|
+
const go = cumulative + projected <= rule.ceilingUsd;
|
|
501
|
+
decisions.push({
|
|
502
|
+
pass,
|
|
503
|
+
cumulativeBefore: cumulative,
|
|
504
|
+
projected,
|
|
505
|
+
go
|
|
506
|
+
});
|
|
507
|
+
if (!go || measuredPassCosts[pass - 1] === void 0) break;
|
|
508
|
+
cumulative += measuredPassCosts[pass - 1];
|
|
509
|
+
uniformN = pass;
|
|
510
|
+
}
|
|
511
|
+
return {
|
|
512
|
+
decisions,
|
|
513
|
+
uniformN
|
|
514
|
+
};
|
|
515
|
+
}
|
|
516
|
+
/**
|
|
517
|
+
* Walk the registered n-ladder and pick the first affordable step. When no
|
|
518
|
+
* step fits the ceiling, the rule refuses and reports the projection instead
|
|
519
|
+
* of shrinking the row set — "never subset rows" is the registered invariant.
|
|
520
|
+
*/
|
|
521
|
+
function projectNLadderBudget(rule, measured) {
|
|
522
|
+
const projections = rule.steps.map((n) => {
|
|
523
|
+
const projectedUsd = measured.unitCostUsd * measured.rows * n;
|
|
524
|
+
return {
|
|
525
|
+
n,
|
|
526
|
+
projectedUsd,
|
|
527
|
+
affordable: projectedUsd <= rule.ceilingUsd
|
|
528
|
+
};
|
|
529
|
+
});
|
|
530
|
+
const first = projections.find((p) => p.affordable);
|
|
531
|
+
if (first) return {
|
|
532
|
+
chosenN: first.n,
|
|
533
|
+
projections,
|
|
534
|
+
refusal: null
|
|
535
|
+
};
|
|
536
|
+
return {
|
|
537
|
+
chosenN: null,
|
|
538
|
+
projections,
|
|
539
|
+
refusal: {
|
|
540
|
+
onExhaust: rule.onExhaust,
|
|
541
|
+
reason: `no ladder step fits the ${rule.ceilingUsd} USD ceiling at ${measured.rows} rows x ${measured.unitCostUsd} USD per unit`
|
|
542
|
+
}
|
|
543
|
+
};
|
|
544
|
+
}
|
|
545
|
+
/**
|
|
546
|
+
* Classify one rollout event under the registered reissue policy. A carrier
|
|
547
|
+
* event within the issue budget is reissued; a model outcome always stands;
|
|
548
|
+
* a carrier event past `maxIssues` is exhausted and reported, never retried.
|
|
549
|
+
*/
|
|
550
|
+
function classifyReissue(policy, event, issuesSoFar) {
|
|
551
|
+
if (!policy.carrierEvents.includes(event)) return "stands";
|
|
552
|
+
return issuesSoFar < policy.maxIssues ? "reissue" : "exhausted";
|
|
553
|
+
}
|
|
554
|
+
//#endregion
|
|
555
|
+
export { runSelectionRule as _, evaluateHaltRule as a, evaluatePopulationReproducibilityGate as c, evaluateProvenanceGate as d, executeDecisionRule as f, readField as g, projectNLadderBudget as h, evaluateCondition as i, evaluatePowerFloorGate as l, powerFloorProblems as m, computeEstimand as n, evaluateIdentityGate as o, intervalSpecProblems as p, computeInterval as r, evaluateOracleDeterminismGate as s, classifyReissue as t, evaluatePredicate as u, runUniformPassBudget as v };
|
|
556
|
+
|
|
557
|
+
//# sourceMappingURL=ast-CP9ae9B0.js.map
|