@tangle-network/agent-eval 0.146.0 → 0.147.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +14 -0
- package/dist/analyst/index.d.ts +13 -13
- package/dist/analyst/index.js +6 -6
- package/dist/{backend-integrity-HVmWyOaX.d.ts → backend-integrity-Bz8nSrRE.d.ts} +2 -2
- package/dist/{backend-integrity-HVmWyOaX.d.ts.map → backend-integrity-Bz8nSrRE.d.ts.map} +1 -1
- package/dist/{benchmark-DTQ1RApl.d.ts → benchmark-Cn7ZMFNW.d.ts} +3 -3
- package/dist/{benchmark-DTQ1RApl.d.ts.map → benchmark-Cn7ZMFNW.d.ts.map} +1 -1
- package/dist/{benchmark-command-CxLJljS1.js → benchmark-command-BTiysiYU.js} +11 -11
- package/dist/{benchmark-command-CxLJljS1.js.map → benchmark-command-BTiysiYU.js.map} +1 -1
- package/dist/benchmarks/index.d.ts +4 -4
- package/dist/benchmarks/index.js +3 -3
- package/dist/campaign/index.d.ts +8 -8
- package/dist/campaign/index.js +5 -5
- package/dist/{campaign-CbQXcARJ.js → campaign-DCDdhuv2.js} +7 -7
- package/dist/{campaign-CbQXcARJ.js.map → campaign-DCDdhuv2.js.map} +1 -1
- package/dist/{capture-fetch-BZiO2cEH.d.ts → capture-fetch-BAx3Gntt.d.ts} +2 -2
- package/dist/{capture-fetch-BZiO2cEH.d.ts.map → capture-fetch-BAx3Gntt.d.ts.map} +1 -1
- package/dist/{chat-client-BJmwfnjN.js → chat-client-BX8Wh3fn.js} +3 -3
- package/dist/{chat-client-BJmwfnjN.js.map → chat-client-BX8Wh3fn.js.map} +1 -1
- package/dist/cli.js +2 -2
- package/dist/{client-YDkU5QST.d.ts → client-BV7icIfy.d.ts} +2 -2
- package/dist/{client-YDkU5QST.d.ts.map → client-BV7icIfy.d.ts.map} +1 -1
- package/dist/contract/index.d.ts +10 -10
- package/dist/contract/index.js +9 -9
- package/dist/{cost-ledger-BSe92yAV.js → cost-ledger-B1qx30B4.js} +2 -2
- package/dist/{cost-ledger-BSe92yAV.js.map → cost-ledger-B1qx30B4.js.map} +1 -1
- package/dist/{default-registry-DONui3mA.d.ts → default-registry-BUtRMRKk.d.ts} +6 -6
- package/dist/{default-registry-DONui3mA.d.ts.map → default-registry-BUtRMRKk.d.ts.map} +1 -1
- package/dist/{define-agent-eval-BGxmy4W6.js → define-agent-eval-BqWFz3sK.js} +3 -3
- package/dist/{define-agent-eval-BGxmy4W6.js.map → define-agent-eval-BqWFz3sK.js.map} +1 -1
- package/dist/{define-agent-eval-DrpgTVPp.d.ts → define-agent-eval-D52ClbX2.d.ts} +5 -5
- package/dist/{define-agent-eval-DrpgTVPp.d.ts.map → define-agent-eval-D52ClbX2.d.ts.map} +1 -1
- package/dist/{dspy-rlm-engine-B2nst4NR.js → dspy-rlm-engine-BmtOl_kP.js} +4 -4
- package/dist/{dspy-rlm-engine-B2nst4NR.js.map → dspy-rlm-engine-BmtOl_kP.js.map} +1 -1
- package/dist/{engine-Bc3seRPe.d.ts → engine-CIy18RTX.d.ts} +4 -4
- package/dist/{engine-Bc3seRPe.d.ts.map → engine-CIy18RTX.d.ts.map} +1 -1
- package/dist/{eval-campaign-DeGLwACc.js → eval-campaign-bmZ6NIIP.js} +2 -2
- package/dist/{eval-campaign-DeGLwACc.js.map → eval-campaign-bmZ6NIIP.js.map} +1 -1
- package/dist/{exact-types-CYFbvA5U.d.ts → exact-types-rdKFzEnK.d.ts} +2 -2
- package/dist/{exact-types-CYFbvA5U.d.ts.map → exact-types-rdKFzEnK.d.ts.map} +1 -1
- package/dist/experiment/index.d.ts +5 -5
- package/dist/experiment/index.js +3 -3
- package/dist/{experiment-tracker-DWHZBAYL.d.ts → experiment-tracker-D9VHwRm-.d.ts} +6 -24
- package/dist/experiment-tracker-D9VHwRm-.d.ts.map +1 -0
- package/dist/{experiment-tracker-C29gXM4B.js → experiment-tracker-Ym6rEQT1.js} +15 -4
- package/dist/experiment-tracker-Ym6rEQT1.js.map +1 -0
- package/dist/{external-optimizer-contracts-CKEY98bK.d.ts → external-optimizer-contracts-CSjDLLmr.d.ts} +2 -2
- package/dist/{external-optimizer-contracts-CKEY98bK.d.ts.map → external-optimizer-contracts-CSjDLLmr.d.ts.map} +1 -1
- package/dist/{external-optimizer-process-DYNizz8W.js → external-optimizer-process-C4AyCcQd.js} +2 -2
- package/dist/{external-optimizer-process-DYNizz8W.js.map → external-optimizer-process-C4AyCcQd.js.map} +1 -1
- package/dist/{external-optimizer-subprocess-r6dP9DAj.js → external-optimizer-subprocess-DU1KM8yV.js} +3 -3
- package/dist/{external-optimizer-subprocess-r6dP9DAj.js.map → external-optimizer-subprocess-DU1KM8yV.js.map} +1 -1
- package/dist/{feedback-trajectory-Cg9EDxKB.d.ts → feedback-trajectory-niGeR6uJ.d.ts} +3 -3
- package/dist/{feedback-trajectory-Cg9EDxKB.d.ts.map → feedback-trajectory-niGeR6uJ.d.ts.map} +1 -1
- package/dist/fuzz.js +1 -1
- package/dist/hosted/index.d.ts +2 -2
- package/dist/{index-DGrcBMbC.d.ts → index-BjBjxiVv.d.ts} +9 -9
- package/dist/{index-DGrcBMbC.d.ts.map → index-BjBjxiVv.d.ts.map} +1 -1
- package/dist/index-CKblBxtr.d.ts +1 -0
- package/dist/{index-CevPFhDT.d.ts → index-gtjtfLEJ.d.ts} +7 -7
- package/dist/{index-CevPFhDT.d.ts.map → index-gtjtfLEJ.d.ts.map} +1 -1
- package/dist/index.d.ts +21 -51
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +16 -125
- package/dist/index.js.map +1 -1
- package/dist/{integrity-DDoeWibF.d.ts → integrity-CGfpTE5-.d.ts} +2 -2
- package/dist/{integrity-DDoeWibF.d.ts.map → integrity-CGfpTE5-.d.ts.map} +1 -1
- package/dist/{kind-factory-CPmSd58s.js → kind-factory-DmAa0h3K.js} +7 -7
- package/dist/kind-factory-DmAa0h3K.js.map +1 -0
- package/dist/{llm-client-_UE4fU7Z.js → llm-client-BMuxYoZy.js} +2 -2
- package/dist/{llm-client-_UE4fU7Z.js.map → llm-client-BMuxYoZy.js.map} +1 -1
- package/dist/{llm-judge-B7ffRF8w.js → llm-judge-DrzsVS5k.js} +4 -4
- package/dist/{llm-judge-B7ffRF8w.js.map → llm-judge-DrzsVS5k.js.map} +1 -1
- package/dist/{matrix-su7mIfbB.d.ts → matrix-C-2Qx1Zr.d.ts} +2 -2
- package/dist/{matrix-su7mIfbB.d.ts.map → matrix-C-2Qx1Zr.d.ts.map} +1 -1
- package/dist/meta-eval/index.d.ts +1 -1
- package/dist/{metrics-Cl0L1KUy.js → metrics-Qv-cpptD.js} +15 -8
- package/dist/metrics-Qv-cpptD.js.map +1 -0
- package/dist/multishot/golden/index.d.ts +1 -1
- package/dist/multishot/index.d.ts +2 -2
- package/dist/multishot/index.js +1 -1
- package/dist/openapi.json +1 -1
- package/dist/{produced-state-D3k45S9a.js → produced-state-Dtx60bUQ.js} +4 -4
- package/dist/{produced-state-D3k45S9a.js.map → produced-state-Dtx60bUQ.js.map} +1 -1
- package/dist/{promotion-policy-BKi5OK5l.d.ts → promotion-policy-CewdjfCJ.d.ts} +2 -2
- package/dist/{promotion-policy-BKi5OK5l.d.ts.map → promotion-policy-CewdjfCJ.d.ts.map} +1 -1
- package/dist/{provenance-Bb7Zmyj2.d.ts → provenance-LrOEOHQb.d.ts} +6 -6
- package/dist/{provenance-Bb7Zmyj2.d.ts.map → provenance-LrOEOHQb.d.ts.map} +1 -1
- package/dist/{registry-CZgEmVXb.d.ts → registry-CaqJIDdN.d.ts} +4 -4
- package/dist/{registry-CZgEmVXb.d.ts.map → registry-CaqJIDdN.d.ts.map} +1 -1
- package/dist/reporting.d.ts +1 -1
- package/dist/reporting.js +1 -1
- package/dist/{researcher-CC305Ed-.d.ts → researcher-BySlpla_.d.ts} +3 -3
- package/dist/{researcher-CC305Ed-.d.ts.map → researcher-BySlpla_.d.ts.map} +1 -1
- package/dist/rl.d.ts +3 -3
- package/dist/rl.js +2 -2
- package/dist/{semantic-concept-judge-BsY2Q0Oq.js → semantic-concept-judge-BI7Rrl5-.js} +3 -3
- package/dist/{semantic-concept-judge-BsY2Q0Oq.js.map → semantic-concept-judge-BI7Rrl5-.js.map} +1 -1
- package/dist/{sequential-CYwq6Ff_.d.ts → sequential-BhsrMupG.d.ts} +31 -2
- package/dist/sequential-BhsrMupG.d.ts.map +1 -0
- package/dist/{sequential-Br0mAPHA.js → sequential-CzK5DarL.js} +48 -6
- package/dist/sequential-CzK5DarL.js.map +1 -0
- package/dist/series-convergence-CjO2QdRW.js.map +1 -1
- package/dist/{series-convergence-C9G-GNYK.d.ts → series-convergence-D2fsoJ2w.d.ts} +4 -7
- package/dist/{series-convergence-C9G-GNYK.d.ts.map → series-convergence-D2fsoJ2w.d.ts.map} +1 -1
- package/dist/{server-cQXve8i4.js → server-D9wQclzG.js} +3 -3
- package/dist/{server-cQXve8i4.js.map → server-D9wQclzG.js.map} +1 -1
- package/dist/{skillopt-optimization-method-C9MrMgW9.d.ts → skillopt-optimization-method-BoC1Qccx.d.ts} +5 -5
- package/dist/{skillopt-optimization-method-C9MrMgW9.d.ts.map → skillopt-optimization-method-BoC1Qccx.d.ts.map} +1 -1
- package/dist/{skillopt-optimization-method-Cv3K4DaS.js → skillopt-optimization-method-C_UrqZs2.js} +4 -4
- package/dist/{skillopt-optimization-method-Cv3K4DaS.js.map → skillopt-optimization-method-C_UrqZs2.js.map} +1 -1
- package/dist/{statistical-heldout-fvxtHnKQ.d.ts → statistical-heldout-uxSpFEjm.d.ts} +2 -2
- package/dist/{statistical-heldout-fvxtHnKQ.d.ts.map → statistical-heldout-uxSpFEjm.d.ts.map} +1 -1
- package/dist/{store-otlp-CsptLYpN.js → store-otlp-CDYWW_8N.js} +2 -2
- package/dist/{store-otlp-CsptLYpN.js.map → store-otlp-CDYWW_8N.js.map} +1 -1
- package/dist/{store-tool-spans-Cq9mFd-q.js → store-tool-spans-CykkbOlv.js} +3 -3
- package/dist/{store-tool-spans-Cq9mFd-q.js.map → store-tool-spans-CykkbOlv.js.map} +1 -1
- package/dist/{store-tool-spans-CV0hRsUp.d.ts → store-tool-spans-D5FhM0_A.d.ts} +4 -4
- package/dist/{store-tool-spans-CV0hRsUp.d.ts.map → store-tool-spans-D5FhM0_A.d.ts.map} +1 -1
- package/dist/{task-failure-attributes-CpQ4y5RD.js → task-failure-attributes--ZTP3tYO.js} +2 -2
- package/dist/{task-failure-attributes-CpQ4y5RD.js.map → task-failure-attributes--ZTP3tYO.js.map} +1 -1
- package/dist/{tool-groups-CDDXNKhd.d.ts → tool-groups-zpufabP8.d.ts} +3 -3
- package/dist/tool-groups-zpufabP8.d.ts.map +1 -0
- package/dist/trace-repair/index.d.ts +1 -1
- package/dist/traces.d.ts +7 -7
- package/dist/traces.js +4 -4
- package/dist/{types-oAsa-fs4.d.ts → types-BohHewKK.d.ts} +2 -2
- package/dist/{types-oAsa-fs4.d.ts.map → types-BohHewKK.d.ts.map} +1 -1
- package/dist/{types-DABZgDGV.d.ts → types-C1Bmeb8X.d.ts} +3 -3
- package/dist/{types-DABZgDGV.d.ts.map → types-C1Bmeb8X.d.ts.map} +1 -1
- package/dist/{types-CWcPguwd.d.ts → types-ColZlZtc.d.ts} +2 -2
- package/dist/{types-CWcPguwd.d.ts.map → types-ColZlZtc.d.ts.map} +1 -1
- package/dist/{types-XBRdmbj8.d.ts → types-DOhGq4S1.d.ts} +1 -7
- package/dist/{types-XBRdmbj8.d.ts.map → types-DOhGq4S1.d.ts.map} +1 -1
- package/dist/wire/index.d.ts +2 -2
- package/dist/wire/index.js +1 -1
- package/package.json +2 -2
- package/dist/experiment-tracker-C29gXM4B.js.map +0 -1
- package/dist/experiment-tracker-DWHZBAYL.d.ts.map +0 -1
- package/dist/index-BgHYx1QG.d.ts +0 -1
- package/dist/kind-factory-CPmSd58s.js.map +0 -1
- package/dist/metrics-Cl0L1KUy.js.map +0 -1
- package/dist/sequential-Br0mAPHA.js.map +0 -1
- package/dist/sequential-CYwq6Ff_.d.ts.map +0 -1
- package/dist/tool-groups-CDDXNKhd.d.ts.map +0 -1
|
@@ -1,9 +1,9 @@
|
|
|
1
1
|
import { i as AgentProfileCellInput, r as AgentProfileCell } from "./agent-profile-cell-BkcRDikH.js";
|
|
2
2
|
import { a as RunRecord, c as RunTaskFailure, n as RunCostProvenance, r as RunJudgeMetadata, s as RunSplitTag, t as JudgeScoresRecord, u as RunTokenUsage } from "./run-record-DVV82Gwh.js";
|
|
3
|
-
import { A as LlmClientOptions, N as LlmRouteRequirements, mt as RawProviderSink } from "./types-
|
|
3
|
+
import { A as LlmClientOptions, N as LlmRouteRequirements, mt as RawProviderSink } from "./types-DOhGq4S1.js";
|
|
4
4
|
import { s as TraceStore } from "./store-CT9YIIve.js";
|
|
5
5
|
import { i as TraceEmitter, t as RunCompleteHook } from "./emitter-DGQGoLyj.js";
|
|
6
|
-
import { a as RunIntegrityReport, n as RunIntegrityExpectations } from "./integrity-
|
|
6
|
+
import { a as RunIntegrityReport, n as RunIntegrityExpectations } from "./integrity-CGfpTE5-.js";
|
|
7
7
|
import { b as GateDecision, d as ResearchReportOptions, s as ResearchReport } from "./summary-report-B__Y5ub3.js";
|
|
8
8
|
//#region src/eval-campaign.d.ts
|
|
9
9
|
interface CampaignVariant<V> {
|
|
@@ -278,4 +278,4 @@ interface Researcher {
|
|
|
278
278
|
}
|
|
279
279
|
//#endregion
|
|
280
280
|
export { SteeringChange as a, CampaignRunContext as c, CampaignVariant as d, EvalCampaignOptions as f, runEvalCampaign as h, Researcher as i, CampaignRunOutcome as l, FailedRun as m, ExperimentResult as n, CampaignFactoryParams as o, EvalCampaignResult as p, FailureMode as r, CampaignIntegrityPolicy as s, ExperimentPlan as t, CampaignScenario as u };
|
|
281
|
-
//# sourceMappingURL=researcher-
|
|
281
|
+
//# sourceMappingURL=researcher-BySlpla_.d.ts.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"researcher-
|
|
1
|
+
{"version":3,"file":"researcher-BySlpla_.d.ts","names":[],"sources":["../src/eval-campaign.ts","../src/researcher.ts"],"mappings":";;;;;;;;UAyEiB,gBAAgB;EAC/B;EACA,SAAS;;UAGM;EACf;;EAEA,OAAO;;UAGQ,mBAAmB;;EAElC;;EAEA;EACA,SAAS;EACT;EACA;EACA,cAAc;EACd;EACA,UAAU;;;;;;;EAOV,SAAS;EACT,OAAO;EACP,SAAS;;;;;;EAMT,SAAS;;UAGD;;EAER;;EAEA;;EAEA;;EAEA,gBAAgB;EAChB,YAAY;;EAEZ;;EAEA;;EAEA;;EAEA,MAAM;;EAEN,gBAAgB;;;;;;EAMhB,cAAc;;;;;;EAMd,eAAe,mBAAmB;;;KAIxB,qBAAqB,2BAA2B;KAEhD,eAAe,MAAM,KAAK,mBAAmB,OAAO,QAAQ;KAE5D;UAEK,oBAAoB;;;;;EAKnC;EACA,UAAU,gBAAgB;EAC1B,WAAW;;EAEX;;EAEA,WAAW;;EAEX;;;;;;EAMA,SAAS;;;;;;EAMT,oBAAoB;;;;;;EAMpB,eAAe,QAAQ,0BAA0B;;;;;;;EAOjD,kBAAkB,QAAQ,0BAA0B;;;;;EAKpD;;;;;EAKA,gBAAgB;;;;;;EAMhB,YAAY;;EAEZ,qBAAqB;;;;;EAKrB,QAAQ,eAAe;;;;;EAKvB;IAAW;MAAwB,KACjC;;;;;EAOF;;EAEA;;;;EAIA;;EAEA,SAAS,QAAQ;;;;;;;EAOjB,eACI,mBACA,0BAEE,QAAQ;IACN,SAAS;IACT,cAAc;QAGd,mBACA,wBACA,QAAQ,mBAAmB;;UAGpB;EACf;EACA;EACA;EACA;EACA;;UAGe;EACf;EACA;EACA;EACA;EACA;EACA;;UAGe;EACf;;EAEA;EACA;;EAEA,MAAM;;EAEN,kBAAkB;EAClB,YAAY;;EAEZ,SAAS;EACT;EACA;;iBAgBoB,gBAAgB,GACpC,MAAM,oBAAoB,KACzB,QAAQ;;;;UC5QM;;;EAGf;;EAEA;EACA;;;IAGE;;IAEA;;;;UAKa;EACf;;;;EAIA;;;EAGA;;EAEA;;;UAIe;EACf;EACA;EACA,SAAS;;;EAGT;;EAEA;IAAU;IAAkB;;;;UAIb;EACf,MAAM;EACN,MAAM;EACN,cAAc;;;;;;;;;;;;;;;;;;;UAoBC;EACf,gBAAgB,MAAM,cAAc,QAAQ;EAC5C,cAAc,UAAU,gBAAgB,QAAQ;EAChD,YAAY,SAAS,kBAAkB,UAAU,iBAAiB,QAAQ;EAC1E,eAAe,MAAM,iBAAiB,QAAQ"}
|
package/dist/rl.d.ts
CHANGED
|
@@ -2,13 +2,13 @@ import { s as VerificationReport } from "./multi-layer-verifier-DIguZc8Z.js";
|
|
|
2
2
|
import { _ as Span } from "./schema-BtVldJ3T.js";
|
|
3
3
|
import { a as RunRecord, s as RunSplitTag } from "./run-record-DVV82Gwh.js";
|
|
4
4
|
import { s as TraceStore } from "./store-CT9YIIve.js";
|
|
5
|
-
import { a as CampaignResult } from "./types-
|
|
5
|
+
import { a as CampaignResult } from "./types-C1Bmeb8X.js";
|
|
6
6
|
import { b as RolloutSplit, o as MintedRolloutLine } from "./schema-Cef2cFmb.js";
|
|
7
7
|
import { a as detectRewardHacking, c as VerifiableRewardSource, d as filterDeterministicallyRewarded, i as RewardHackingSignal, l as extractVerifiableReward, n as RewardHackingFinding, o as VerifiableReward, r as RewardHackingReport, s as VerifiableRewardExtractionOptions, t as DetectRewardHackingInput, u as extractVerifiableRewardsFromRecords } from "./reward-hacking-Bu8ev6PR.js";
|
|
8
8
|
import { i as InMemoryOutcomeStore, n as FileSystemOutcomeStore, o as OutcomeStore, r as FileSystemOutcomeStoreOptions, t as DeploymentOutcome } from "./outcome-store-BYHIuO0e.js";
|
|
9
|
-
import { t as InterimReleaseConfidence } from "./sequential-
|
|
9
|
+
import { t as InterimReleaseConfidence } from "./sequential-BhsrMupG.js";
|
|
10
10
|
import { t as AdversarialMutation } from "./adversarial-smnADNFS.js";
|
|
11
|
-
import { a as SteeringChange, f as EvalCampaignOptions, h as runEvalCampaign, i as Researcher, n as ExperimentResult, p as EvalCampaignResult, r as FailureMode, t as ExperimentPlan } from "./researcher-
|
|
11
|
+
import { a as SteeringChange, f as EvalCampaignOptions, h as runEvalCampaign, i as Researcher, n as ExperimentResult, p as EvalCampaignResult, r as FailureMode, t as ExperimentPlan } from "./researcher-BySlpla_.js";
|
|
12
12
|
import { r as RubricPredictiveValidityReport } from "./rubric-predictive-validity-C7LnNvF2.js";
|
|
13
13
|
//#region src/rl/active-curriculum.d.ts
|
|
14
14
|
interface CellObservation {
|
package/dist/rl.js
CHANGED
|
@@ -5,11 +5,11 @@ import { r as observedSplitScore, s as trainingScore } from "./reward-nw2xZGZG.j
|
|
|
5
5
|
import { s as runTaskScore } from "./run-record-D2lDdSAz.js";
|
|
6
6
|
import { i as filterDeterministicallyRewarded, l as campaignCellToRunRecord, n as extractVerifiableReward, r as extractVerifiableRewardsFromRecords, t as detectRewardHacking } from "./reward-hacking-DFo2FU5J.js";
|
|
7
7
|
import { n as InMemoryTraceStore } from "./store-DNe_Uv1Q.js";
|
|
8
|
-
import { t as runEvalCampaign } from "./eval-campaign-
|
|
8
|
+
import { t as runEvalCampaign } from "./eval-campaign-bmZ6NIIP.js";
|
|
9
9
|
import { l as assertRewardGate } from "./schema-C6DW4ZHR.js";
|
|
10
10
|
import { t as isSplitEligible } from "./exporters-q9iL-2Jf.js";
|
|
11
11
|
import { t as mintRolloutRows } from "./mint-BV6tLVWl.js";
|
|
12
|
-
import { t as evaluateInterimReleaseConfidence } from "./sequential-
|
|
12
|
+
import { t as evaluateInterimReleaseConfidence } from "./sequential-CzK5DarL.js";
|
|
13
13
|
import { t as rubricPredictiveValidity } from "./rubric-predictive-validity-Cwwyd7ah.js";
|
|
14
14
|
import { n as thompsonCurriculum, r as varianceBasedCurriculum, t as observationsFromRunRecords } from "./active-curriculum-C4mk67HP.js";
|
|
15
15
|
import { n as InMemoryOutcomeStore, t as FileSystemOutcomeStore } from "./outcome-store-ChBKlTd_.js";
|
|
@@ -1,6 +1,6 @@
|
|
|
1
|
-
import { i as CostLedger } from "./cost-ledger-
|
|
1
|
+
import { i as CostLedger } from "./cost-ledger-B1qx30B4.js";
|
|
2
2
|
import { l as Mutex } from "./ledger-core-DTae9rv_.js";
|
|
3
|
-
import { c as callLlmJson, f as maximumChargeForLlmRequest, l as costReceiptFromLlm, u as costReceiptFromLlmError } from "./llm-client-
|
|
3
|
+
import { c as callLlmJson, f as maximumChargeForLlmRequest, l as costReceiptFromLlm, u as costReceiptFromLlmError } from "./llm-client-BMuxYoZy.js";
|
|
4
4
|
import { appendFileSync, existsSync, mkdirSync, readFileSync } from "node:fs";
|
|
5
5
|
import { dirname } from "node:path";
|
|
6
6
|
//#region src/locked-jsonl-appender.ts
|
|
@@ -403,4 +403,4 @@ async function runSemanticConceptJudge(input, options = {}) {
|
|
|
403
403
|
//#endregion
|
|
404
404
|
export { diffFindings as a, defaultIsMaterial as i, runSemanticConceptJudge as n, FindingsStore as r, SEMANTIC_CONCEPT_JUDGE_VERSION as t };
|
|
405
405
|
|
|
406
|
-
//# sourceMappingURL=semantic-concept-judge-
|
|
406
|
+
//# sourceMappingURL=semantic-concept-judge-BI7Rrl5-.js.map
|
package/dist/{semantic-concept-judge-BsY2Q0Oq.js.map → semantic-concept-judge-BI7Rrl5-.js.map}
RENAMED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"semantic-concept-judge-BsY2Q0Oq.js","names":[],"sources":["../src/locked-jsonl-appender.ts","../src/analyst/findings-store.ts","../src/semantic-concept-judge.ts"],"sourcesContent":["/**\n * LockedJsonlAppender — mutex-serialized JSONL append helper for arbitrary\n * payloads. The reference-replay store does the same thing for typed\n * `ReferenceReplayRun` rows; this is the generic version used by\n * `MutationTelemetry`, `TrialTelemetry`, and any other consumer that wants\n * append-only durable telemetry without rolling its own lock.\n *\n * Locks are per absolute file path (process-local). Cross-process\n * concurrency is NOT addressed — that's an fcntl/flock problem.\n */\n\nimport { appendFileSync, existsSync, mkdirSync } from 'node:fs'\nimport { dirname } from 'node:path'\nimport { Mutex } from './concurrency'\n\nconst mutexes = new Map<string, Mutex>()\n\nfunction getMutex(path: string): Mutex {\n let m = mutexes.get(path)\n if (!m) {\n m = new Mutex()\n mutexes.set(path, m)\n }\n return m\n}\n\nexport class LockedJsonlAppender {\n private readonly mutex: Mutex\n constructor(public readonly path: string) {\n this.mutex = getMutex(path)\n if (!existsSync(dirname(path))) {\n mkdirSync(dirname(path), { recursive: true })\n }\n }\n\n async append(entry: unknown): Promise<void> {\n const line = `${JSON.stringify(entry)}\\n`\n await this.mutex.runExclusive(() => {\n appendFileSync(this.path, line)\n })\n }\n}\n\n/** Reset all internal mutex state — tests only. */\nexport function resetLockedAppendersForTesting(): void {\n mutexes.clear()\n}\n","/**\n * FindingsStore — durable persistence for AnalystFinding rows + a diff\n * helper so we can answer \"what changed since the last run?\" without\n * recomputing analysts.\n *\n * On-disk shape is JSONL: one finding per line, append-only, locked via\n * LockedJsonlAppender. Operators get crash-safety (no partial JSON),\n * cheap reads (sequential parse), and trivial backup (rsync the file).\n *\n * Reads are non-locking: a reader sees a consistent snapshot of all\n * fully-written lines and skips an incomplete trailing line if the\n * writer is mid-append. Cross-process locking is intentionally out of\n * scope (see locked-jsonl-appender.ts).\n *\n * The store is run-scoped: callers pass `runId` on append and on load,\n * which keeps multi-run files cleanly partitioned. The `diffFindings`\n * helper compares two run-id sets using stable `finding_id` semantics —\n * the diff is the cross-run signal the regression dashboard renders.\n */\n\nimport { existsSync, readFileSync } from 'node:fs'\n\nimport { LockedJsonlAppender } from '../locked-jsonl-appender'\nimport type { AnalystFinding } from './types'\n\n/**\n * One persisted row. We attach `run_id` on disk so a single file can\n * hold multiple runs and the diff helper can query without re-walking\n * separate files.\n */\nexport interface PersistedFinding extends AnalystFinding {\n run_id: string\n}\n\nexport class FindingsStore {\n private readonly appender: LockedJsonlAppender\n\n constructor(public readonly path: string) {\n this.appender = new LockedJsonlAppender(path)\n }\n\n async append(runId: string, findings: AnalystFinding[]): Promise<void> {\n for (const f of findings) {\n const row: PersistedFinding = { ...f, run_id: runId }\n await this.appender.append(row)\n }\n }\n\n /** Load every persisted finding. Discards malformed trailing lines silently. */\n loadAll(): PersistedFinding[] {\n if (!existsSync(this.path)) return []\n const raw = readFileSync(this.path, 'utf8')\n if (!raw) return []\n const out: PersistedFinding[] = []\n for (const line of raw.split('\\n')) {\n if (!line) continue\n try {\n out.push(JSON.parse(line) as PersistedFinding)\n } catch {\n // Skip torn trailing line — the lock guarantees no torn lines\n // mid-file, only at EOF when a writer is in-flight.\n }\n }\n return out\n }\n\n /** Filter to a single run. */\n loadRun(runId: string): PersistedFinding[] {\n return this.loadAll().filter((r) => r.run_id === runId)\n }\n}\n\n// ── Cross-run diff ──────────────────────────────────────────────────\n\nexport interface FindingsDiff {\n /** New finding ids in `current` that weren't in `previous`. */\n appeared: PersistedFinding[]\n /** Finding ids in `previous` that aren't in `current`. */\n disappeared: PersistedFinding[]\n /** Same finding id present in both runs and unchanged per the materiality test. */\n persisted: PersistedFinding[]\n /**\n * Same finding id in both runs but at least one non-identity field\n * shifted per `DiffPolicy.isMaterial`. Reported as [previous, current].\n */\n changed: Array<{ previous: PersistedFinding; current: PersistedFinding }>\n}\n\nexport interface DiffPolicy {\n /**\n * Predicate that decides whether two findings (same finding_id) count\n * as a material change. Defaults to {@link defaultIsMaterial}: severity\n * shift, confidence Δ > 0.05, or evidence count change. Compliance /\n * perf consumers MAY supply a stricter predicate (e.g. rationale text\n * diff, metric Δ thresholds).\n */\n isMaterial?: (previous: AnalystFinding, current: AnalystFinding) => boolean\n}\n\n/**\n * Default materiality test. Deliberately narrow so LLM-reword churn\n * doesn't flood the diff. Stricter tests are opt-in via DiffPolicy.\n */\nexport function defaultIsMaterial(a: AnalystFinding, b: AnalystFinding): boolean {\n if (a.severity !== b.severity) return true\n if (Math.abs((a.confidence ?? 0) - (b.confidence ?? 0)) > 0.05) return true\n if (a.evidence_refs.length !== b.evidence_refs.length) return true\n return false\n}\n\n/**\n * Diff two findings sets by stable finding_id. Callers typically load\n * the two run-id slices from the same store and pass them in.\n */\nexport function diffFindings(\n previous: PersistedFinding[],\n current: PersistedFinding[],\n policy: DiffPolicy = {},\n): FindingsDiff {\n const isMaterial = policy.isMaterial ?? defaultIsMaterial\n const prevById = new Map(previous.map((f) => [f.finding_id, f]))\n const curById = new Map(current.map((f) => [f.finding_id, f]))\n\n const appeared: PersistedFinding[] = []\n const disappeared: PersistedFinding[] = []\n const persisted: PersistedFinding[] = []\n const changed: FindingsDiff['changed'] = []\n\n for (const [id, cur] of curById) {\n const prev = prevById.get(id)\n if (!prev) {\n appeared.push(cur)\n continue\n }\n if (isMaterial(prev, cur)) {\n changed.push({ previous: prev, current: cur })\n } else {\n persisted.push(cur)\n }\n }\n for (const [id, prev] of prevById) {\n if (!curById.has(id)) disappeared.push(prev)\n }\n return { appeared, disappeared, persisted, changed }\n}\n","/**\n * Semantic concept judge — \"does the built artifact actually implement\n * the features the user asked for?\"\n *\n * Distinct from the domain/code/coherence judges in `judges.ts`:\n * - those judges score free-form conversational agent outputs along\n * quality dimensions (accuracy, depth, etc.)\n * - this judge scores a *built artifact* (served HTML + source files)\n * against an explicit list of expected concepts, returning per-concept\n * {present, score 0-10, evidence, severity}.\n *\n * The judge is strict about distinguishing (a) a working implementation\n * from (b) a keyword-present stub. \"// TODO: mint button\" is NOT present.\n * Only real, functional, wired-up code counts.\n *\n * Use via {@link createSemanticConceptJudge} or directly via\n * {@link runSemanticConceptJudge}. Soft-fails (available=false) on LLM\n * or JSON-parse errors so the caller can treat that as \"layer skipped\"\n * rather than \"layer failed\" in a multi-layer pipeline.\n */\n\nimport { CostLedger, type CostLedgerHandle, type CostReceipt } from './cost-ledger'\nimport {\n callLlmJson,\n costReceiptFromLlm,\n costReceiptFromLlmError,\n type LlmCallRequest,\n type LlmClientOptions,\n maximumChargeForLlmRequest,\n} from './llm-client'\nimport type { Severity } from './multi-layer-verifier'\n\n// ─── Types ──────────────────────────────────────────────────────────────\n\n/**\n * Implementation complexity class for weighted scoring.\n *\n * - `render` (default): the concept is a UI surface that displays static\n * data — render a list, show a counter, lay out a button. Single-file\n * work, no external integration.\n * - `integrate`: the concept requires wiring a real external system —\n * wallet connect (wagmi + RainbowKit + chain config), payment provider\n * (Stripe Elements + intent + webhook), an API client with auth.\n * Multi-file, library-knowledge, runtime correctness matters.\n * - `compute`: the concept requires algorithmic work — solver, simulator,\n * constraint propagation, ML inference. Correctness > UI polish.\n *\n * Default weights (when applied via `weightConcepts: 'complexity'`):\n * render=1.0, integrate=2.0, compute=2.5\n *\n * Cross-vertical scoring without complexity weighting silently inflates\n * the rate of UI-heavy verticals (healthcare, fintech dashboards) vs\n * integration-heavy verticals (DeFi, wallets) — all concepts treated\n * equally even though the agent does 2-3x the work for `integrate`.\n */\nexport type ConceptComplexity = 'render' | 'integrate' | 'compute'\n\nexport interface ConceptSpec {\n name: string\n /** Short hints that help the judge; not used for matching. */\n keywords?: string[]\n /** Optional explicit weight; default 1.0. Overrides complexity-derived weight. */\n weight?: number\n /** Implementation complexity class. Default `render`. */\n complexity?: ConceptComplexity\n}\n\nexport interface ConceptFinding {\n concept: string\n present: boolean\n /** 0..10. 10 = production-ready; 7 = functional thin; 4 = partial; 0 = absent. */\n score: number\n evidence: string\n severity: Severity\n}\n\nexport interface SemanticConceptJudgeInput {\n /** Full natural-language prompt the agent was handed. */\n userRequest: string\n /** Rendered HTML the preview returns (UI artifacts). Optional. */\n servedHtml?: string\n /** Top-level source files from the agent's workdir. */\n sourceFiles: Array<{ path: string; content: string }>\n /** The expected concept list. */\n expectedConcepts: ConceptSpec[]\n /** Free-form metadata (id, difficulty) to inject into the prompt. */\n artifactLabel?: string\n artifactDescription?: string\n}\n\nexport interface SemanticConceptJudgeResult {\n kind: 'semantic-concept'\n version: string\n /** Normalized 0..1 score — mean of per-concept scores / 10. */\n score: number\n presentCount: number\n totalCount: number\n findings: ConceptFinding[]\n summary: string\n durationMs: number\n costUsd: number | null\n /** False on LLM/JSON error — treat as \"skipped / unable to judge\" in pipelines. */\n available: boolean\n error?: string\n}\n\n/**\n * Score-aggregation strategy. `mean` averages 0-10 scores uniformly.\n * `complexity` applies the default weight table (render=1, integrate=2,\n * compute=2.5) unless a concept has an explicit `weight`. `explicit`\n * honors only `weight` (defaulting to 1 for unspecified).\n */\nexport type ConceptWeightStrategy = 'mean' | 'complexity' | 'explicit'\n\nexport const DEFAULT_COMPLEXITY_WEIGHTS: Record<ConceptComplexity, number> = {\n render: 1.0,\n integrate: 2.0,\n compute: 2.5,\n}\n\nexport interface SemanticConceptJudgeOptions {\n /** Model id to call. Default 'claude-sonnet-4-6' via agent-eval defaults. */\n model?: string\n /** Per-call timeout. Default 300s. */\n timeoutMs?: number\n /** Provider-enforced output limit. Default 16000. */\n maxTokens?: number\n /** Pipeline budget for the prompt (source blob truncation). Default 45000. */\n maxSourceChars?: number\n /** Per-file cap before inclusion. Default 20000. */\n maxPerFileChars?: number\n /** HTML cap. Default 30000. */\n maxHtmlChars?: number\n /** LlmClient config (baseUrl, apiKey, authHeader, …). */\n llm?: LlmClientOptions\n costLedger?: CostLedgerHandle\n costPhase?: string\n costTags?: Record<string, string>\n signal?: AbortSignal\n /**\n * Score aggregation strategy. Default `mean` — uniform average across\n * concepts. Cross-vertical comparisons should use `complexity` to\n * neutralize the integrate-vs-render asymmetry.\n */\n weightConcepts?: ConceptWeightStrategy\n /** Override the default complexity → weight table. */\n complexityWeights?: Partial<Record<ConceptComplexity, number>>\n}\n\n// ─── Prompt assembly ────────────────────────────────────────────────────\n\nexport const SEMANTIC_CONCEPT_JUDGE_VERSION = 'semantic-concept-judge-v1-2026-04-24'\n\nconst DEFAULT_MAX_SOURCE = 45_000\nconst DEFAULT_MAX_HTML = 30_000\nconst DEFAULT_MAX_PER_FILE = 20_000\nconst DEFAULT_TIMEOUT = 300_000\nconst DEFAULT_MAX_TOKENS = 16_000\nconst DEFAULT_MODEL = 'claude-sonnet-4-6'\n\nconst SEMANTIC_SCHEMA = {\n type: 'object',\n additionalProperties: false,\n required: ['summary', 'concepts'],\n properties: {\n summary: { type: 'string', minLength: 20, maxLength: 600 },\n concepts: {\n type: 'array',\n minItems: 1,\n items: {\n type: 'object',\n additionalProperties: false,\n required: ['concept', 'present', 'score', 'evidence', 'severity'],\n properties: {\n concept: { type: 'string', minLength: 1, maxLength: 120 },\n present: { type: 'boolean' },\n score: { type: 'number', minimum: 0, maximum: 10 },\n evidence: { type: 'string', minLength: 5, maxLength: 400 },\n severity: { type: 'string', enum: ['critical', 'major', 'minor', 'info'] },\n },\n },\n },\n },\n}\n\nfunction truncate(body: string, cap: number, label: string): string {\n if (body.length <= cap) return body\n return `${body.slice(0, cap)}\\n… [truncated ${body.length - cap} chars of ${label}]`\n}\n\nfunction buildPrompt(\n input: SemanticConceptJudgeInput,\n opts: Required<SemanticConceptJudgeOptions>,\n): string {\n const sourceBlob = input.sourceFiles\n .filter((f) => f.content.length <= opts.maxPerFileChars)\n .map((f) => `--- FILE: ${f.path} ---\\n${f.content}`)\n .join('\\n\\n')\n\n const html = input.servedHtml ?? ''\n\n return `You are a strict code-review judge evaluating whether an agent's 0-to-1 build actually implements the features the user asked for.\n\nYou MUST distinguish:\n (a) WORKING code that implements the concept (rendered UI, wired handler, real API call),\n (b) KEYWORD-PRESENT stub (comments mentioning the concept, variable names, TODOs),\n (c) ABSENT (concept nowhere).\n\nA comment like \"// TODO: add mint button\" is NOT present — score 2-3. Only count a concept as present if there is real functional code: a rendered component, a call handler wired to state or a network call, a computed value actually used.\n\nUSER REQUEST (what the agent was asked to build):\n${input.userRequest}\n\n${input.artifactLabel ? `ARTIFACT METADATA:\\n name: ${input.artifactLabel}\\n description: ${input.artifactDescription ?? ''}\\n\\n` : ''}EXPECTED CONCEPTS (each must be graded independently):\n${input.expectedConcepts\n .map(\n (c, i) =>\n ` ${i + 1}. \"${c.name}\"${c.keywords?.length ? ` — hints: [${c.keywords.slice(0, 6).join(' | ')}]` : ''}`,\n )\n .join('\\n')}\n\n${html ? `SERVED HTML (what the preview returns when hit):\\n${truncate(html, opts.maxHtmlChars, 'HTML')}\\n\\n` : ''}SOURCE FILES (the agent's workdir):\n${truncate(sourceBlob, opts.maxSourceChars, 'source')}\n\nFor EACH concept, return:\n - concept: the concept name as given (match exactly)\n - present: boolean — does a working implementation exist?\n - score: 0-10 — 10 = production-ready; 7 = functional but thin; 4 = partial/stubbed; 2 = keyword-only comment; 0 = absent\n - evidence: cite \"<file>:<line>\" or \"served-html:<selector>\" pointing at the strongest supporting code. If the concept is absent or stubbed, explain what's missing.\n - severity:\n \"info\" when present: true AND score >= 7\n \"minor\" when present: true AND 4 <= score < 7\n \"major\" when present: false OR score < 4\n \"critical\" when the concept is not only absent but a core user flow depends on it\n\nAlso produce a \"summary\" (one sentence, 20-600 chars): overall verdict on whether this is a shippable implementation of the user request vs a keyword-dense placeholder.\n\nBE SKEPTICAL. Keyword matching already passed — your job is to catch what keyword matching misses. If the agent shipped a working build, say so. If it shipped a stub, say so. Don't grade on effort.\n\nReturn STRICT JSON. No prose outside the JSON.`\n}\n\n// ─── Runner ─────────────────────────────────────────────────────────────\n\n/**\n * Run the semantic concept judge. Soft-fails to available=false on\n * LLM/JSON errors — callers in a MultiLayerVerifier pipeline can treat\n * that as \"skip\" rather than \"fail.\"\n */\nexport async function runSemanticConceptJudge(\n input: SemanticConceptJudgeInput,\n options: SemanticConceptJudgeOptions = {},\n): Promise<SemanticConceptJudgeResult> {\n const start = Date.now()\n const totalCount = input.expectedConcepts.length\n\n if (totalCount === 0) {\n return {\n kind: 'semantic-concept',\n version: SEMANTIC_CONCEPT_JUDGE_VERSION,\n score: 0,\n presentCount: 0,\n totalCount: 0,\n findings: [],\n summary: 'no expected concepts declared',\n durationMs: 0,\n costUsd: null,\n available: false,\n error: 'no expected concepts declared',\n }\n }\n\n const opts: Required<SemanticConceptJudgeOptions> = {\n model: options.model ?? DEFAULT_MODEL,\n timeoutMs: options.timeoutMs ?? DEFAULT_TIMEOUT,\n maxTokens: options.maxTokens ?? DEFAULT_MAX_TOKENS,\n maxSourceChars: options.maxSourceChars ?? DEFAULT_MAX_SOURCE,\n maxPerFileChars: options.maxPerFileChars ?? DEFAULT_MAX_PER_FILE,\n maxHtmlChars: options.maxHtmlChars ?? DEFAULT_MAX_HTML,\n llm: options.llm ?? {},\n costLedger: options.costLedger ?? new CostLedger(),\n costPhase: options.costPhase ?? 'judge.semantic-concept',\n costTags: options.costTags ?? {},\n signal: options.signal ?? new AbortController().signal,\n weightConcepts: options.weightConcepts ?? 'mean',\n complexityWeights: { ...DEFAULT_COMPLEXITY_WEIGHTS, ...(options.complexityWeights ?? {}) },\n }\n\n // Build a name → weight map for aggregation. Mean strategy keeps every\n // weight at 1 (uniform average). Complexity strategy reads the table\n // and lets an explicit `weight` override. Explicit strategy uses ONLY\n // the spec's `weight` (defaulting to 1).\n const weightForConcept = (spec: ConceptSpec): number => {\n if (opts.weightConcepts === 'mean') return 1\n if (spec.weight != null) return spec.weight\n if (opts.weightConcepts === 'complexity') {\n return opts.complexityWeights[spec.complexity ?? 'render'] ?? 1\n }\n return 1\n }\n const weightByName = new Map<string, number>(\n input.expectedConcepts.map((c) => [c.name, weightForConcept(c)]),\n )\n\n let receipt: CostReceipt | undefined\n try {\n const request = {\n model: opts.model,\n messages: [\n {\n role: 'system' as const,\n content:\n 'You are a strict code-review judge. Return strict JSON only. No prose outside the JSON. A keyword in a comment is NOT a working implementation.',\n },\n { role: 'user' as const, content: buildPrompt(input, opts) },\n ],\n jsonSchema: { name: 'semantic_concept_judge', schema: SEMANTIC_SCHEMA },\n temperature: 0,\n maxTokens: opts.maxTokens,\n timeoutMs: opts.timeoutMs,\n } satisfies LlmCallRequest\n const paid = await opts.costLedger.runPaidCall({\n channel: 'judge',\n phase: opts.costPhase,\n actor: 'semantic-concept',\n model: opts.model,\n ...(Object.keys(opts.costTags).length > 0 ? { tags: opts.costTags } : {}),\n maximumCharge: maximumChargeForLlmRequest(request, opts.llm),\n signal: opts.signal,\n execute: (signal, callId) =>\n callLlmJson<{ summary: string; concepts: ConceptFinding[] }>(request, {\n ...opts.llm,\n signal,\n idempotencyKey: callId,\n }),\n receipt: ({ result }) => costReceiptFromLlm(result),\n receiptFromError: costReceiptFromLlmError,\n })\n receipt = paid.receipt\n if (!paid.succeeded) throw paid.error\n const { value } = paid.value\n\n if (!value?.concepts || !Array.isArray(value.concepts)) {\n throw new Error('judge returned malformed response — expected array under \"concepts\"')\n }\n\n const findings: ConceptFinding[] = value.concepts.map((c) => ({\n concept: String(c.concept),\n present: Boolean(c.present),\n score: Math.max(0, Math.min(10, Number(c.score ?? 0))),\n evidence: String(c.evidence ?? ''),\n severity: (['critical', 'major', 'minor', 'info'] as const).includes(c.severity)\n ? c.severity\n : 'info',\n }))\n\n const presentCount = findings.filter((f) => f.present && f.score >= 7).length\n let weightSum = 0\n let weightedScoreSum = 0\n for (const f of findings) {\n const w = weightByName.get(f.concept) ?? 1\n weightSum += w\n weightedScoreSum += w * f.score\n }\n const scoreAvg =\n weightSum > 0\n ? weightedScoreSum / weightSum\n : findings.reduce((a, f) => a + f.score, 0) / Math.max(1, findings.length)\n\n return {\n kind: 'semantic-concept',\n version: SEMANTIC_CONCEPT_JUDGE_VERSION,\n score: Number((scoreAvg / 10).toFixed(3)),\n presentCount,\n totalCount,\n findings,\n summary: String(value.summary ?? ''),\n durationMs: Date.now() - start,\n costUsd: paid.receipt.costUnknown ? null : paid.receipt.costUsd,\n available: true,\n }\n } catch (err) {\n return {\n kind: 'semantic-concept',\n version: SEMANTIC_CONCEPT_JUDGE_VERSION,\n score: 0,\n presentCount: 0,\n totalCount,\n findings: [],\n summary: '',\n durationMs: Date.now() - start,\n costUsd: receipt && !receipt.costUnknown ? receipt.costUsd : null,\n available: false,\n error: err instanceof Error ? err.message : String(err),\n }\n }\n}\n\n/**\n * Factory: pin LLM options once, return a closure that accepts inputs.\n * Convenient for pipelines that want to share a single LlmClient config.\n */\nexport function createSemanticConceptJudge(\n options: SemanticConceptJudgeOptions = {},\n): (input: SemanticConceptJudgeInput) => Promise<SemanticConceptJudgeResult> {\n return (input) => runSemanticConceptJudge(input, options)\n}\n"],"mappings":";;;;;;;;;;;;;;;;AAeA,MAAM,0BAAU,IAAI,IAAmB;AAEvC,SAAS,SAAS,MAAqB;CACrC,IAAI,IAAI,QAAQ,IAAI,IAAI;CACxB,IAAI,CAAC,GAAG;EACN,IAAI,IAAI,MAAM;EACd,QAAQ,IAAI,MAAM,CAAC;CACrB;CACA,OAAO;AACT;AAEA,IAAa,sBAAb,MAAiC;CAEH;CAD5B;CACA,YAAY,MAA8B;EAAd,KAAA,OAAA;EAC1B,KAAK,QAAQ,SAAS,IAAI;EAC1B,IAAI,CAAC,WAAW,QAAQ,IAAI,CAAC,GAC3B,UAAU,QAAQ,IAAI,GAAG,EAAE,WAAW,KAAK,CAAC;CAEhD;CAEA,MAAM,OAAO,OAA+B;EAC1C,MAAM,OAAO,GAAG,KAAK,UAAU,KAAK,EAAE;EACtC,MAAM,KAAK,MAAM,mBAAmB;GAClC,eAAe,KAAK,MAAM,IAAI;EAChC,CAAC;CACH;AACF;;;;;;;;;;;;;;;;;;;;;;ACPA,IAAa,gBAAb,MAA2B;CAGG;CAF5B;CAEA,YAAY,MAA8B;EAAd,KAAA,OAAA;EAC1B,KAAK,WAAW,IAAI,oBAAoB,IAAI;CAC9C;CAEA,MAAM,OAAO,OAAe,UAA2C;EACrE,KAAK,MAAM,KAAK,UAAU;GACxB,MAAM,MAAwB;IAAE,GAAG;IAAG,QAAQ;GAAM;GACpD,MAAM,KAAK,SAAS,OAAO,GAAG;EAChC;CACF;;CAGA,UAA8B;EAC5B,IAAI,CAAC,WAAW,KAAK,IAAI,GAAG,OAAO,CAAC;EACpC,MAAM,MAAM,aAAa,KAAK,MAAM,MAAM;EAC1C,IAAI,CAAC,KAAK,OAAO,CAAC;EAClB,MAAM,MAA0B,CAAC;EACjC,KAAK,MAAM,QAAQ,IAAI,MAAM,IAAI,GAAG;GAClC,IAAI,CAAC,MAAM;GACX,IAAI;IACF,IAAI,KAAK,KAAK,MAAM,IAAI,CAAqB;GAC/C,QAAQ,CAGR;EACF;EACA,OAAO;CACT;;CAGA,QAAQ,OAAmC;EACzC,OAAO,KAAK,QAAQ,CAAC,CAAC,QAAQ,MAAM,EAAE,WAAW,KAAK;CACxD;AACF;;;;;AAiCA,SAAgB,kBAAkB,GAAmB,GAA4B;CAC/E,IAAI,EAAE,aAAa,EAAE,UAAU,OAAO;CACtC,IAAI,KAAK,KAAK,EAAE,cAAc,MAAM,EAAE,cAAc,EAAE,IAAI,KAAM,OAAO;CACvE,IAAI,EAAE,cAAc,WAAW,EAAE,cAAc,QAAQ,OAAO;CAC9D,OAAO;AACT;;;;;AAMA,SAAgB,aACd,UACA,SACA,SAAqB,CAAC,GACR;CACd,MAAM,aAAa,OAAO,cAAc;CACxC,MAAM,WAAW,IAAI,IAAI,SAAS,KAAK,MAAM,CAAC,EAAE,YAAY,CAAC,CAAC,CAAC;CAC/D,MAAM,UAAU,IAAI,IAAI,QAAQ,KAAK,MAAM,CAAC,EAAE,YAAY,CAAC,CAAC,CAAC;CAE7D,MAAM,WAA+B,CAAC;CACtC,MAAM,cAAkC,CAAC;CACzC,MAAM,YAAgC,CAAC;CACvC,MAAM,UAAmC,CAAC;CAE1C,KAAK,MAAM,CAAC,IAAI,QAAQ,SAAS;EAC/B,MAAM,OAAO,SAAS,IAAI,EAAE;EAC5B,IAAI,CAAC,MAAM;GACT,SAAS,KAAK,GAAG;GACjB;EACF;EACA,IAAI,WAAW,MAAM,GAAG,GACtB,QAAQ,KAAK;GAAE,UAAU;GAAM,SAAS;EAAI,CAAC;OAE7C,UAAU,KAAK,GAAG;CAEtB;CACA,KAAK,MAAM,CAAC,IAAI,SAAS,UACvB,IAAI,CAAC,QAAQ,IAAI,EAAE,GAAG,YAAY,KAAK,IAAI;CAE7C,OAAO;EAAE;EAAU;EAAa;EAAW;CAAQ;AACrD;;;;;;;;;;;;;;;;;;;;;;;AC9BA,MAAa,6BAAgE;CAC3E,QAAQ;CACR,WAAW;CACX,SAAS;AACX;AAiCA,MAAa,iCAAiC;AAE9C,MAAM,qBAAqB;AAC3B,MAAM,mBAAmB;AACzB,MAAM,uBAAuB;AAC7B,MAAM,kBAAkB;AACxB,MAAM,qBAAqB;AAC3B,MAAM,gBAAgB;AAEtB,MAAM,kBAAkB;CACtB,MAAM;CACN,sBAAsB;CACtB,UAAU,CAAC,WAAW,UAAU;CAChC,YAAY;EACV,SAAS;GAAE,MAAM;GAAU,WAAW;GAAI,WAAW;EAAI;EACzD,UAAU;GACR,MAAM;GACN,UAAU;GACV,OAAO;IACL,MAAM;IACN,sBAAsB;IACtB,UAAU;KAAC;KAAW;KAAW;KAAS;KAAY;IAAU;IAChE,YAAY;KACV,SAAS;MAAE,MAAM;MAAU,WAAW;MAAG,WAAW;KAAI;KACxD,SAAS,EAAE,MAAM,UAAU;KAC3B,OAAO;MAAE,MAAM;MAAU,SAAS;MAAG,SAAS;KAAG;KACjD,UAAU;MAAE,MAAM;MAAU,WAAW;MAAG,WAAW;KAAI;KACzD,UAAU;MAAE,MAAM;MAAU,MAAM;OAAC;OAAY;OAAS;OAAS;MAAM;KAAE;IAC3E;GACF;EACF;CACF;AACF;AAEA,SAAS,SAAS,MAAc,KAAa,OAAuB;CAClE,IAAI,KAAK,UAAU,KAAK,OAAO;CAC/B,OAAO,GAAG,KAAK,MAAM,GAAG,GAAG,EAAE,iBAAiB,KAAK,SAAS,IAAI,YAAY,MAAM;AACpF;AAEA,SAAS,YACP,OACA,MACQ;CACR,MAAM,aAAa,MAAM,YACtB,QAAQ,MAAM,EAAE,QAAQ,UAAU,KAAK,eAAe,CAAC,CACvD,KAAK,MAAM,aAAa,EAAE,KAAK,QAAQ,EAAE,SAAS,CAAC,CACnD,KAAK,MAAM;CAEd,MAAM,OAAO,MAAM,cAAc;CAEjC,OAAO;;;;;;;;;;EAUP,MAAM,YAAY;;EAElB,MAAM,gBAAgB,+BAA+B,MAAM,cAAc,mBAAmB,MAAM,uBAAuB,GAAG,QAAQ,GAAG;EACvI,MAAM,iBACL,KACE,GAAG,MACF,KAAK,IAAI,EAAE,KAAK,EAAE,KAAK,GAAG,EAAE,UAAU,SAAS,cAAc,EAAE,SAAS,MAAM,GAAG,CAAC,CAAC,CAAC,KAAK,KAAK,EAAE,KAAK,IACzG,CAAC,CACA,KAAK,IAAI,EAAE;;EAEZ,OAAO,qDAAqD,SAAS,MAAM,KAAK,cAAc,MAAM,EAAE,QAAQ,GAAG;EACjH,SAAS,YAAY,KAAK,gBAAgB,QAAQ,EAAE;;;;;;;;;;;;;;;;;;AAkBtD;;;;;;AASA,eAAsB,wBACpB,OACA,UAAuC,CAAC,GACH;CACrC,MAAM,QAAQ,KAAK,IAAI;CACvB,MAAM,aAAa,MAAM,iBAAiB;CAE1C,IAAI,eAAe,GACjB,OAAO;EACL,MAAM;EACN,SAAS;EACT,OAAO;EACP,cAAc;EACd,YAAY;EACZ,UAAU,CAAC;EACX,SAAS;EACT,YAAY;EACZ,SAAS;EACT,WAAW;EACX,OAAO;CACT;CAGF,MAAM,OAA8C;EAClD,OAAO,QAAQ,SAAS;EACxB,WAAW,QAAQ,aAAa;EAChC,WAAW,QAAQ,aAAa;EAChC,gBAAgB,QAAQ,kBAAkB;EAC1C,iBAAiB,QAAQ,mBAAmB;EAC5C,cAAc,QAAQ,gBAAgB;EACtC,KAAK,QAAQ,OAAO,CAAC;EACrB,YAAY,QAAQ,cAAc,IAAI,WAAW;EACjD,WAAW,QAAQ,aAAa;EAChC,UAAU,QAAQ,YAAY,CAAC;EAC/B,QAAQ,QAAQ,UAAU,IAAI,gBAAgB,CAAC,CAAC;EAChD,gBAAgB,QAAQ,kBAAkB;EAC1C,mBAAmB;GAAE,GAAG;GAA4B,GAAI,QAAQ,qBAAqB,CAAC;EAAG;CAC3F;CAMA,MAAM,oBAAoB,SAA8B;EACtD,IAAI,KAAK,mBAAmB,QAAQ,OAAO;EAC3C,IAAI,KAAK,UAAU,MAAM,OAAO,KAAK;EACrC,IAAI,KAAK,mBAAmB,cAC1B,OAAO,KAAK,kBAAkB,KAAK,cAAc,aAAa;EAEhE,OAAO;CACT;CACA,MAAM,eAAe,IAAI,IACvB,MAAM,iBAAiB,KAAK,MAAM,CAAC,EAAE,MAAM,iBAAiB,CAAC,CAAC,CAAC,CACjE;CAEA,IAAI;CACJ,IAAI;EACF,MAAM,UAAU;GACd,OAAO,KAAK;GACZ,UAAU,CACR;IACE,MAAM;IACN,SACE;GACJ,GACA;IAAE,MAAM;IAAiB,SAAS,YAAY,OAAO,IAAI;GAAE,CAC7D;GACA,YAAY;IAAE,MAAM;IAA0B,QAAQ;GAAgB;GACtE,aAAa;GACb,WAAW,KAAK;GAChB,WAAW,KAAK;EAClB;EACA,MAAM,OAAO,MAAM,KAAK,WAAW,YAAY;GAC7C,SAAS;GACT,OAAO,KAAK;GACZ,OAAO;GACP,OAAO,KAAK;GACZ,GAAI,OAAO,KAAK,KAAK,QAAQ,CAAC,CAAC,SAAS,IAAI,EAAE,MAAM,KAAK,SAAS,IAAI,CAAC;GACvE,eAAe,2BAA2B,SAAS,KAAK,GAAG;GAC3D,QAAQ,KAAK;GACb,UAAU,QAAQ,WAChB,YAA6D,SAAS;IACpE,GAAG,KAAK;IACR;IACA,gBAAgB;GAClB,CAAC;GACH,UAAU,EAAE,aAAa,mBAAmB,MAAM;GAClD,kBAAkB;EACpB,CAAC;EACD,UAAU,KAAK;EACf,IAAI,CAAC,KAAK,WAAW,MAAM,KAAK;EAChC,MAAM,EAAE,UAAU,KAAK;EAEvB,IAAI,CAAC,OAAO,YAAY,CAAC,MAAM,QAAQ,MAAM,QAAQ,GACnD,MAAM,IAAI,MAAM,uEAAqE;EAGvF,MAAM,WAA6B,MAAM,SAAS,KAAK,OAAO;GAC5D,SAAS,OAAO,EAAE,OAAO;GACzB,SAAS,QAAQ,EAAE,OAAO;GAC1B,OAAO,KAAK,IAAI,GAAG,KAAK,IAAI,IAAI,OAAO,EAAE,SAAS,CAAC,CAAC,CAAC;GACrD,UAAU,OAAO,EAAE,YAAY,EAAE;GACjC,UAAW;IAAC;IAAY;IAAS;IAAS;GAAM,CAAC,CAAW,SAAS,EAAE,QAAQ,IAC3E,EAAE,WACF;EACN,EAAE;EAEF,MAAM,eAAe,SAAS,QAAQ,MAAM,EAAE,WAAW,EAAE,SAAS,CAAC,CAAC,CAAC;EACvE,IAAI,YAAY;EAChB,IAAI,mBAAmB;EACvB,KAAK,MAAM,KAAK,UAAU;GACxB,MAAM,IAAI,aAAa,IAAI,EAAE,OAAO,KAAK;GACzC,aAAa;GACb,oBAAoB,IAAI,EAAE;EAC5B;EACA,MAAM,WACJ,YAAY,IACR,mBAAmB,YACnB,SAAS,QAAQ,GAAG,MAAM,IAAI,EAAE,OAAO,CAAC,IAAI,KAAK,IAAI,GAAG,SAAS,MAAM;EAE7E,OAAO;GACL,MAAM;GACN,SAAS;GACT,OAAO,QAAQ,WAAW,GAAA,CAAI,QAAQ,CAAC,CAAC;GACxC;GACA;GACA;GACA,SAAS,OAAO,MAAM,WAAW,EAAE;GACnC,YAAY,KAAK,IAAI,IAAI;GACzB,SAAS,KAAK,QAAQ,cAAc,OAAO,KAAK,QAAQ;GACxD,WAAW;EACb;CACF,SAAS,KAAK;EACZ,OAAO;GACL,MAAM;GACN,SAAS;GACT,OAAO;GACP,cAAc;GACd;GACA,UAAU,CAAC;GACX,SAAS;GACT,YAAY,KAAK,IAAI,IAAI;GACzB,SAAS,WAAW,CAAC,QAAQ,cAAc,QAAQ,UAAU;GAC7D,WAAW;GACX,OAAO,eAAe,QAAQ,IAAI,UAAU,OAAO,GAAG;EACxD;CACF;AACF"}
|
|
1
|
+
{"version":3,"file":"semantic-concept-judge-BI7Rrl5-.js","names":[],"sources":["../src/locked-jsonl-appender.ts","../src/analyst/findings-store.ts","../src/semantic-concept-judge.ts"],"sourcesContent":["/**\n * LockedJsonlAppender — mutex-serialized JSONL append helper for arbitrary\n * payloads. The reference-replay store does the same thing for typed\n * `ReferenceReplayRun` rows; this is the generic version used by\n * `MutationTelemetry`, `TrialTelemetry`, and any other consumer that wants\n * append-only durable telemetry without rolling its own lock.\n *\n * Locks are per absolute file path (process-local). Cross-process\n * concurrency is NOT addressed — that's an fcntl/flock problem.\n */\n\nimport { appendFileSync, existsSync, mkdirSync } from 'node:fs'\nimport { dirname } from 'node:path'\nimport { Mutex } from './concurrency'\n\nconst mutexes = new Map<string, Mutex>()\n\nfunction getMutex(path: string): Mutex {\n let m = mutexes.get(path)\n if (!m) {\n m = new Mutex()\n mutexes.set(path, m)\n }\n return m\n}\n\nexport class LockedJsonlAppender {\n private readonly mutex: Mutex\n constructor(public readonly path: string) {\n this.mutex = getMutex(path)\n if (!existsSync(dirname(path))) {\n mkdirSync(dirname(path), { recursive: true })\n }\n }\n\n async append(entry: unknown): Promise<void> {\n const line = `${JSON.stringify(entry)}\\n`\n await this.mutex.runExclusive(() => {\n appendFileSync(this.path, line)\n })\n }\n}\n\n/** Reset all internal mutex state — tests only. */\nexport function resetLockedAppendersForTesting(): void {\n mutexes.clear()\n}\n","/**\n * FindingsStore — durable persistence for AnalystFinding rows + a diff\n * helper so we can answer \"what changed since the last run?\" without\n * recomputing analysts.\n *\n * On-disk shape is JSONL: one finding per line, append-only, locked via\n * LockedJsonlAppender. Operators get crash-safety (no partial JSON),\n * cheap reads (sequential parse), and trivial backup (rsync the file).\n *\n * Reads are non-locking: a reader sees a consistent snapshot of all\n * fully-written lines and skips an incomplete trailing line if the\n * writer is mid-append. Cross-process locking is intentionally out of\n * scope (see locked-jsonl-appender.ts).\n *\n * The store is run-scoped: callers pass `runId` on append and on load,\n * which keeps multi-run files cleanly partitioned. The `diffFindings`\n * helper compares two run-id sets using stable `finding_id` semantics —\n * the diff is the cross-run signal the regression dashboard renders.\n */\n\nimport { existsSync, readFileSync } from 'node:fs'\n\nimport { LockedJsonlAppender } from '../locked-jsonl-appender'\nimport type { AnalystFinding } from './types'\n\n/**\n * One persisted row. We attach `run_id` on disk so a single file can\n * hold multiple runs and the diff helper can query without re-walking\n * separate files.\n */\nexport interface PersistedFinding extends AnalystFinding {\n run_id: string\n}\n\nexport class FindingsStore {\n private readonly appender: LockedJsonlAppender\n\n constructor(public readonly path: string) {\n this.appender = new LockedJsonlAppender(path)\n }\n\n async append(runId: string, findings: AnalystFinding[]): Promise<void> {\n for (const f of findings) {\n const row: PersistedFinding = { ...f, run_id: runId }\n await this.appender.append(row)\n }\n }\n\n /** Load every persisted finding. Discards malformed trailing lines silently. */\n loadAll(): PersistedFinding[] {\n if (!existsSync(this.path)) return []\n const raw = readFileSync(this.path, 'utf8')\n if (!raw) return []\n const out: PersistedFinding[] = []\n for (const line of raw.split('\\n')) {\n if (!line) continue\n try {\n out.push(JSON.parse(line) as PersistedFinding)\n } catch {\n // Skip torn trailing line — the lock guarantees no torn lines\n // mid-file, only at EOF when a writer is in-flight.\n }\n }\n return out\n }\n\n /** Filter to a single run. */\n loadRun(runId: string): PersistedFinding[] {\n return this.loadAll().filter((r) => r.run_id === runId)\n }\n}\n\n// ── Cross-run diff ──────────────────────────────────────────────────\n\nexport interface FindingsDiff {\n /** New finding ids in `current` that weren't in `previous`. */\n appeared: PersistedFinding[]\n /** Finding ids in `previous` that aren't in `current`. */\n disappeared: PersistedFinding[]\n /** Same finding id present in both runs and unchanged per the materiality test. */\n persisted: PersistedFinding[]\n /**\n * Same finding id in both runs but at least one non-identity field\n * shifted per `DiffPolicy.isMaterial`. Reported as [previous, current].\n */\n changed: Array<{ previous: PersistedFinding; current: PersistedFinding }>\n}\n\nexport interface DiffPolicy {\n /**\n * Predicate that decides whether two findings (same finding_id) count\n * as a material change. Defaults to {@link defaultIsMaterial}: severity\n * shift, confidence Δ > 0.05, or evidence count change. Compliance /\n * perf consumers MAY supply a stricter predicate (e.g. rationale text\n * diff, metric Δ thresholds).\n */\n isMaterial?: (previous: AnalystFinding, current: AnalystFinding) => boolean\n}\n\n/**\n * Default materiality test. Deliberately narrow so LLM-reword churn\n * doesn't flood the diff. Stricter tests are opt-in via DiffPolicy.\n */\nexport function defaultIsMaterial(a: AnalystFinding, b: AnalystFinding): boolean {\n if (a.severity !== b.severity) return true\n if (Math.abs((a.confidence ?? 0) - (b.confidence ?? 0)) > 0.05) return true\n if (a.evidence_refs.length !== b.evidence_refs.length) return true\n return false\n}\n\n/**\n * Diff two findings sets by stable finding_id. Callers typically load\n * the two run-id slices from the same store and pass them in.\n */\nexport function diffFindings(\n previous: PersistedFinding[],\n current: PersistedFinding[],\n policy: DiffPolicy = {},\n): FindingsDiff {\n const isMaterial = policy.isMaterial ?? defaultIsMaterial\n const prevById = new Map(previous.map((f) => [f.finding_id, f]))\n const curById = new Map(current.map((f) => [f.finding_id, f]))\n\n const appeared: PersistedFinding[] = []\n const disappeared: PersistedFinding[] = []\n const persisted: PersistedFinding[] = []\n const changed: FindingsDiff['changed'] = []\n\n for (const [id, cur] of curById) {\n const prev = prevById.get(id)\n if (!prev) {\n appeared.push(cur)\n continue\n }\n if (isMaterial(prev, cur)) {\n changed.push({ previous: prev, current: cur })\n } else {\n persisted.push(cur)\n }\n }\n for (const [id, prev] of prevById) {\n if (!curById.has(id)) disappeared.push(prev)\n }\n return { appeared, disappeared, persisted, changed }\n}\n","/**\n * Semantic concept judge — \"does the built artifact actually implement\n * the features the user asked for?\"\n *\n * Distinct from the domain/code/coherence judges in `judges.ts`:\n * - those judges score free-form conversational agent outputs along\n * quality dimensions (accuracy, depth, etc.)\n * - this judge scores a *built artifact* (served HTML + source files)\n * against an explicit list of expected concepts, returning per-concept\n * {present, score 0-10, evidence, severity}.\n *\n * The judge is strict about distinguishing (a) a working implementation\n * from (b) a keyword-present stub. \"// TODO: mint button\" is NOT present.\n * Only real, functional, wired-up code counts.\n *\n * Use via {@link createSemanticConceptJudge} or directly via\n * {@link runSemanticConceptJudge}. Soft-fails (available=false) on LLM\n * or JSON-parse errors so the caller can treat that as \"layer skipped\"\n * rather than \"layer failed\" in a multi-layer pipeline.\n */\n\nimport { CostLedger, type CostLedgerHandle, type CostReceipt } from './cost-ledger'\nimport {\n callLlmJson,\n costReceiptFromLlm,\n costReceiptFromLlmError,\n type LlmCallRequest,\n type LlmClientOptions,\n maximumChargeForLlmRequest,\n} from './llm-client'\nimport type { Severity } from './multi-layer-verifier'\n\n// ─── Types ──────────────────────────────────────────────────────────────\n\n/**\n * Implementation complexity class for weighted scoring.\n *\n * - `render` (default): the concept is a UI surface that displays static\n * data — render a list, show a counter, lay out a button. Single-file\n * work, no external integration.\n * - `integrate`: the concept requires wiring a real external system —\n * wallet connect (wagmi + RainbowKit + chain config), payment provider\n * (Stripe Elements + intent + webhook), an API client with auth.\n * Multi-file, library-knowledge, runtime correctness matters.\n * - `compute`: the concept requires algorithmic work — solver, simulator,\n * constraint propagation, ML inference. Correctness > UI polish.\n *\n * Default weights (when applied via `weightConcepts: 'complexity'`):\n * render=1.0, integrate=2.0, compute=2.5\n *\n * Cross-vertical scoring without complexity weighting silently inflates\n * the rate of UI-heavy verticals (healthcare, fintech dashboards) vs\n * integration-heavy verticals (DeFi, wallets) — all concepts treated\n * equally even though the agent does 2-3x the work for `integrate`.\n */\nexport type ConceptComplexity = 'render' | 'integrate' | 'compute'\n\nexport interface ConceptSpec {\n name: string\n /** Short hints that help the judge; not used for matching. */\n keywords?: string[]\n /** Optional explicit weight; default 1.0. Overrides complexity-derived weight. */\n weight?: number\n /** Implementation complexity class. Default `render`. */\n complexity?: ConceptComplexity\n}\n\nexport interface ConceptFinding {\n concept: string\n present: boolean\n /** 0..10. 10 = production-ready; 7 = functional thin; 4 = partial; 0 = absent. */\n score: number\n evidence: string\n severity: Severity\n}\n\nexport interface SemanticConceptJudgeInput {\n /** Full natural-language prompt the agent was handed. */\n userRequest: string\n /** Rendered HTML the preview returns (UI artifacts). Optional. */\n servedHtml?: string\n /** Top-level source files from the agent's workdir. */\n sourceFiles: Array<{ path: string; content: string }>\n /** The expected concept list. */\n expectedConcepts: ConceptSpec[]\n /** Free-form metadata (id, difficulty) to inject into the prompt. */\n artifactLabel?: string\n artifactDescription?: string\n}\n\nexport interface SemanticConceptJudgeResult {\n kind: 'semantic-concept'\n version: string\n /** Normalized 0..1 score — mean of per-concept scores / 10. */\n score: number\n presentCount: number\n totalCount: number\n findings: ConceptFinding[]\n summary: string\n durationMs: number\n costUsd: number | null\n /** False on LLM/JSON error — treat as \"skipped / unable to judge\" in pipelines. */\n available: boolean\n error?: string\n}\n\n/**\n * Score-aggregation strategy. `mean` averages 0-10 scores uniformly.\n * `complexity` applies the default weight table (render=1, integrate=2,\n * compute=2.5) unless a concept has an explicit `weight`. `explicit`\n * honors only `weight` (defaulting to 1 for unspecified).\n */\nexport type ConceptWeightStrategy = 'mean' | 'complexity' | 'explicit'\n\nexport const DEFAULT_COMPLEXITY_WEIGHTS: Record<ConceptComplexity, number> = {\n render: 1.0,\n integrate: 2.0,\n compute: 2.5,\n}\n\nexport interface SemanticConceptJudgeOptions {\n /** Model id to call. Default 'claude-sonnet-4-6' via agent-eval defaults. */\n model?: string\n /** Per-call timeout. Default 300s. */\n timeoutMs?: number\n /** Provider-enforced output limit. Default 16000. */\n maxTokens?: number\n /** Pipeline budget for the prompt (source blob truncation). Default 45000. */\n maxSourceChars?: number\n /** Per-file cap before inclusion. Default 20000. */\n maxPerFileChars?: number\n /** HTML cap. Default 30000. */\n maxHtmlChars?: number\n /** LlmClient config (baseUrl, apiKey, authHeader, …). */\n llm?: LlmClientOptions\n costLedger?: CostLedgerHandle\n costPhase?: string\n costTags?: Record<string, string>\n signal?: AbortSignal\n /**\n * Score aggregation strategy. Default `mean` — uniform average across\n * concepts. Cross-vertical comparisons should use `complexity` to\n * neutralize the integrate-vs-render asymmetry.\n */\n weightConcepts?: ConceptWeightStrategy\n /** Override the default complexity → weight table. */\n complexityWeights?: Partial<Record<ConceptComplexity, number>>\n}\n\n// ─── Prompt assembly ────────────────────────────────────────────────────\n\nexport const SEMANTIC_CONCEPT_JUDGE_VERSION = 'semantic-concept-judge-v1-2026-04-24'\n\nconst DEFAULT_MAX_SOURCE = 45_000\nconst DEFAULT_MAX_HTML = 30_000\nconst DEFAULT_MAX_PER_FILE = 20_000\nconst DEFAULT_TIMEOUT = 300_000\nconst DEFAULT_MAX_TOKENS = 16_000\nconst DEFAULT_MODEL = 'claude-sonnet-4-6'\n\nconst SEMANTIC_SCHEMA = {\n type: 'object',\n additionalProperties: false,\n required: ['summary', 'concepts'],\n properties: {\n summary: { type: 'string', minLength: 20, maxLength: 600 },\n concepts: {\n type: 'array',\n minItems: 1,\n items: {\n type: 'object',\n additionalProperties: false,\n required: ['concept', 'present', 'score', 'evidence', 'severity'],\n properties: {\n concept: { type: 'string', minLength: 1, maxLength: 120 },\n present: { type: 'boolean' },\n score: { type: 'number', minimum: 0, maximum: 10 },\n evidence: { type: 'string', minLength: 5, maxLength: 400 },\n severity: { type: 'string', enum: ['critical', 'major', 'minor', 'info'] },\n },\n },\n },\n },\n}\n\nfunction truncate(body: string, cap: number, label: string): string {\n if (body.length <= cap) return body\n return `${body.slice(0, cap)}\\n… [truncated ${body.length - cap} chars of ${label}]`\n}\n\nfunction buildPrompt(\n input: SemanticConceptJudgeInput,\n opts: Required<SemanticConceptJudgeOptions>,\n): string {\n const sourceBlob = input.sourceFiles\n .filter((f) => f.content.length <= opts.maxPerFileChars)\n .map((f) => `--- FILE: ${f.path} ---\\n${f.content}`)\n .join('\\n\\n')\n\n const html = input.servedHtml ?? ''\n\n return `You are a strict code-review judge evaluating whether an agent's 0-to-1 build actually implements the features the user asked for.\n\nYou MUST distinguish:\n (a) WORKING code that implements the concept (rendered UI, wired handler, real API call),\n (b) KEYWORD-PRESENT stub (comments mentioning the concept, variable names, TODOs),\n (c) ABSENT (concept nowhere).\n\nA comment like \"// TODO: add mint button\" is NOT present — score 2-3. Only count a concept as present if there is real functional code: a rendered component, a call handler wired to state or a network call, a computed value actually used.\n\nUSER REQUEST (what the agent was asked to build):\n${input.userRequest}\n\n${input.artifactLabel ? `ARTIFACT METADATA:\\n name: ${input.artifactLabel}\\n description: ${input.artifactDescription ?? ''}\\n\\n` : ''}EXPECTED CONCEPTS (each must be graded independently):\n${input.expectedConcepts\n .map(\n (c, i) =>\n ` ${i + 1}. \"${c.name}\"${c.keywords?.length ? ` — hints: [${c.keywords.slice(0, 6).join(' | ')}]` : ''}`,\n )\n .join('\\n')}\n\n${html ? `SERVED HTML (what the preview returns when hit):\\n${truncate(html, opts.maxHtmlChars, 'HTML')}\\n\\n` : ''}SOURCE FILES (the agent's workdir):\n${truncate(sourceBlob, opts.maxSourceChars, 'source')}\n\nFor EACH concept, return:\n - concept: the concept name as given (match exactly)\n - present: boolean — does a working implementation exist?\n - score: 0-10 — 10 = production-ready; 7 = functional but thin; 4 = partial/stubbed; 2 = keyword-only comment; 0 = absent\n - evidence: cite \"<file>:<line>\" or \"served-html:<selector>\" pointing at the strongest supporting code. If the concept is absent or stubbed, explain what's missing.\n - severity:\n \"info\" when present: true AND score >= 7\n \"minor\" when present: true AND 4 <= score < 7\n \"major\" when present: false OR score < 4\n \"critical\" when the concept is not only absent but a core user flow depends on it\n\nAlso produce a \"summary\" (one sentence, 20-600 chars): overall verdict on whether this is a shippable implementation of the user request vs a keyword-dense placeholder.\n\nBE SKEPTICAL. Keyword matching already passed — your job is to catch what keyword matching misses. If the agent shipped a working build, say so. If it shipped a stub, say so. Don't grade on effort.\n\nReturn STRICT JSON. No prose outside the JSON.`\n}\n\n// ─── Runner ─────────────────────────────────────────────────────────────\n\n/**\n * Run the semantic concept judge. Soft-fails to available=false on\n * LLM/JSON errors — callers in a MultiLayerVerifier pipeline can treat\n * that as \"skip\" rather than \"fail.\"\n */\nexport async function runSemanticConceptJudge(\n input: SemanticConceptJudgeInput,\n options: SemanticConceptJudgeOptions = {},\n): Promise<SemanticConceptJudgeResult> {\n const start = Date.now()\n const totalCount = input.expectedConcepts.length\n\n if (totalCount === 0) {\n return {\n kind: 'semantic-concept',\n version: SEMANTIC_CONCEPT_JUDGE_VERSION,\n score: 0,\n presentCount: 0,\n totalCount: 0,\n findings: [],\n summary: 'no expected concepts declared',\n durationMs: 0,\n costUsd: null,\n available: false,\n error: 'no expected concepts declared',\n }\n }\n\n const opts: Required<SemanticConceptJudgeOptions> = {\n model: options.model ?? DEFAULT_MODEL,\n timeoutMs: options.timeoutMs ?? DEFAULT_TIMEOUT,\n maxTokens: options.maxTokens ?? DEFAULT_MAX_TOKENS,\n maxSourceChars: options.maxSourceChars ?? DEFAULT_MAX_SOURCE,\n maxPerFileChars: options.maxPerFileChars ?? DEFAULT_MAX_PER_FILE,\n maxHtmlChars: options.maxHtmlChars ?? DEFAULT_MAX_HTML,\n llm: options.llm ?? {},\n costLedger: options.costLedger ?? new CostLedger(),\n costPhase: options.costPhase ?? 'judge.semantic-concept',\n costTags: options.costTags ?? {},\n signal: options.signal ?? new AbortController().signal,\n weightConcepts: options.weightConcepts ?? 'mean',\n complexityWeights: { ...DEFAULT_COMPLEXITY_WEIGHTS, ...(options.complexityWeights ?? {}) },\n }\n\n // Build a name → weight map for aggregation. Mean strategy keeps every\n // weight at 1 (uniform average). Complexity strategy reads the table\n // and lets an explicit `weight` override. Explicit strategy uses ONLY\n // the spec's `weight` (defaulting to 1).\n const weightForConcept = (spec: ConceptSpec): number => {\n if (opts.weightConcepts === 'mean') return 1\n if (spec.weight != null) return spec.weight\n if (opts.weightConcepts === 'complexity') {\n return opts.complexityWeights[spec.complexity ?? 'render'] ?? 1\n }\n return 1\n }\n const weightByName = new Map<string, number>(\n input.expectedConcepts.map((c) => [c.name, weightForConcept(c)]),\n )\n\n let receipt: CostReceipt | undefined\n try {\n const request = {\n model: opts.model,\n messages: [\n {\n role: 'system' as const,\n content:\n 'You are a strict code-review judge. Return strict JSON only. No prose outside the JSON. A keyword in a comment is NOT a working implementation.',\n },\n { role: 'user' as const, content: buildPrompt(input, opts) },\n ],\n jsonSchema: { name: 'semantic_concept_judge', schema: SEMANTIC_SCHEMA },\n temperature: 0,\n maxTokens: opts.maxTokens,\n timeoutMs: opts.timeoutMs,\n } satisfies LlmCallRequest\n const paid = await opts.costLedger.runPaidCall({\n channel: 'judge',\n phase: opts.costPhase,\n actor: 'semantic-concept',\n model: opts.model,\n ...(Object.keys(opts.costTags).length > 0 ? { tags: opts.costTags } : {}),\n maximumCharge: maximumChargeForLlmRequest(request, opts.llm),\n signal: opts.signal,\n execute: (signal, callId) =>\n callLlmJson<{ summary: string; concepts: ConceptFinding[] }>(request, {\n ...opts.llm,\n signal,\n idempotencyKey: callId,\n }),\n receipt: ({ result }) => costReceiptFromLlm(result),\n receiptFromError: costReceiptFromLlmError,\n })\n receipt = paid.receipt\n if (!paid.succeeded) throw paid.error\n const { value } = paid.value\n\n if (!value?.concepts || !Array.isArray(value.concepts)) {\n throw new Error('judge returned malformed response — expected array under \"concepts\"')\n }\n\n const findings: ConceptFinding[] = value.concepts.map((c) => ({\n concept: String(c.concept),\n present: Boolean(c.present),\n score: Math.max(0, Math.min(10, Number(c.score ?? 0))),\n evidence: String(c.evidence ?? ''),\n severity: (['critical', 'major', 'minor', 'info'] as const).includes(c.severity)\n ? c.severity\n : 'info',\n }))\n\n const presentCount = findings.filter((f) => f.present && f.score >= 7).length\n let weightSum = 0\n let weightedScoreSum = 0\n for (const f of findings) {\n const w = weightByName.get(f.concept) ?? 1\n weightSum += w\n weightedScoreSum += w * f.score\n }\n const scoreAvg =\n weightSum > 0\n ? weightedScoreSum / weightSum\n : findings.reduce((a, f) => a + f.score, 0) / Math.max(1, findings.length)\n\n return {\n kind: 'semantic-concept',\n version: SEMANTIC_CONCEPT_JUDGE_VERSION,\n score: Number((scoreAvg / 10).toFixed(3)),\n presentCount,\n totalCount,\n findings,\n summary: String(value.summary ?? ''),\n durationMs: Date.now() - start,\n costUsd: paid.receipt.costUnknown ? null : paid.receipt.costUsd,\n available: true,\n }\n } catch (err) {\n return {\n kind: 'semantic-concept',\n version: SEMANTIC_CONCEPT_JUDGE_VERSION,\n score: 0,\n presentCount: 0,\n totalCount,\n findings: [],\n summary: '',\n durationMs: Date.now() - start,\n costUsd: receipt && !receipt.costUnknown ? receipt.costUsd : null,\n available: false,\n error: err instanceof Error ? err.message : String(err),\n }\n }\n}\n\n/**\n * Factory: pin LLM options once, return a closure that accepts inputs.\n * Convenient for pipelines that want to share a single LlmClient config.\n */\nexport function createSemanticConceptJudge(\n options: SemanticConceptJudgeOptions = {},\n): (input: SemanticConceptJudgeInput) => Promise<SemanticConceptJudgeResult> {\n return (input) => runSemanticConceptJudge(input, options)\n}\n"],"mappings":";;;;;;;;;;;;;;;;AAeA,MAAM,0BAAU,IAAI,IAAmB;AAEvC,SAAS,SAAS,MAAqB;CACrC,IAAI,IAAI,QAAQ,IAAI,IAAI;CACxB,IAAI,CAAC,GAAG;EACN,IAAI,IAAI,MAAM;EACd,QAAQ,IAAI,MAAM,CAAC;CACrB;CACA,OAAO;AACT;AAEA,IAAa,sBAAb,MAAiC;CAEH;CAD5B;CACA,YAAY,MAA8B;EAAd,KAAA,OAAA;EAC1B,KAAK,QAAQ,SAAS,IAAI;EAC1B,IAAI,CAAC,WAAW,QAAQ,IAAI,CAAC,GAC3B,UAAU,QAAQ,IAAI,GAAG,EAAE,WAAW,KAAK,CAAC;CAEhD;CAEA,MAAM,OAAO,OAA+B;EAC1C,MAAM,OAAO,GAAG,KAAK,UAAU,KAAK,EAAE;EACtC,MAAM,KAAK,MAAM,mBAAmB;GAClC,eAAe,KAAK,MAAM,IAAI;EAChC,CAAC;CACH;AACF;;;;;;;;;;;;;;;;;;;;;;ACPA,IAAa,gBAAb,MAA2B;CAGG;CAF5B;CAEA,YAAY,MAA8B;EAAd,KAAA,OAAA;EAC1B,KAAK,WAAW,IAAI,oBAAoB,IAAI;CAC9C;CAEA,MAAM,OAAO,OAAe,UAA2C;EACrE,KAAK,MAAM,KAAK,UAAU;GACxB,MAAM,MAAwB;IAAE,GAAG;IAAG,QAAQ;GAAM;GACpD,MAAM,KAAK,SAAS,OAAO,GAAG;EAChC;CACF;;CAGA,UAA8B;EAC5B,IAAI,CAAC,WAAW,KAAK,IAAI,GAAG,OAAO,CAAC;EACpC,MAAM,MAAM,aAAa,KAAK,MAAM,MAAM;EAC1C,IAAI,CAAC,KAAK,OAAO,CAAC;EAClB,MAAM,MAA0B,CAAC;EACjC,KAAK,MAAM,QAAQ,IAAI,MAAM,IAAI,GAAG;GAClC,IAAI,CAAC,MAAM;GACX,IAAI;IACF,IAAI,KAAK,KAAK,MAAM,IAAI,CAAqB;GAC/C,QAAQ,CAGR;EACF;EACA,OAAO;CACT;;CAGA,QAAQ,OAAmC;EACzC,OAAO,KAAK,QAAQ,CAAC,CAAC,QAAQ,MAAM,EAAE,WAAW,KAAK;CACxD;AACF;;;;;AAiCA,SAAgB,kBAAkB,GAAmB,GAA4B;CAC/E,IAAI,EAAE,aAAa,EAAE,UAAU,OAAO;CACtC,IAAI,KAAK,KAAK,EAAE,cAAc,MAAM,EAAE,cAAc,EAAE,IAAI,KAAM,OAAO;CACvE,IAAI,EAAE,cAAc,WAAW,EAAE,cAAc,QAAQ,OAAO;CAC9D,OAAO;AACT;;;;;AAMA,SAAgB,aACd,UACA,SACA,SAAqB,CAAC,GACR;CACd,MAAM,aAAa,OAAO,cAAc;CACxC,MAAM,WAAW,IAAI,IAAI,SAAS,KAAK,MAAM,CAAC,EAAE,YAAY,CAAC,CAAC,CAAC;CAC/D,MAAM,UAAU,IAAI,IAAI,QAAQ,KAAK,MAAM,CAAC,EAAE,YAAY,CAAC,CAAC,CAAC;CAE7D,MAAM,WAA+B,CAAC;CACtC,MAAM,cAAkC,CAAC;CACzC,MAAM,YAAgC,CAAC;CACvC,MAAM,UAAmC,CAAC;CAE1C,KAAK,MAAM,CAAC,IAAI,QAAQ,SAAS;EAC/B,MAAM,OAAO,SAAS,IAAI,EAAE;EAC5B,IAAI,CAAC,MAAM;GACT,SAAS,KAAK,GAAG;GACjB;EACF;EACA,IAAI,WAAW,MAAM,GAAG,GACtB,QAAQ,KAAK;GAAE,UAAU;GAAM,SAAS;EAAI,CAAC;OAE7C,UAAU,KAAK,GAAG;CAEtB;CACA,KAAK,MAAM,CAAC,IAAI,SAAS,UACvB,IAAI,CAAC,QAAQ,IAAI,EAAE,GAAG,YAAY,KAAK,IAAI;CAE7C,OAAO;EAAE;EAAU;EAAa;EAAW;CAAQ;AACrD;;;;;;;;;;;;;;;;;;;;;;;AC9BA,MAAa,6BAAgE;CAC3E,QAAQ;CACR,WAAW;CACX,SAAS;AACX;AAiCA,MAAa,iCAAiC;AAE9C,MAAM,qBAAqB;AAC3B,MAAM,mBAAmB;AACzB,MAAM,uBAAuB;AAC7B,MAAM,kBAAkB;AACxB,MAAM,qBAAqB;AAC3B,MAAM,gBAAgB;AAEtB,MAAM,kBAAkB;CACtB,MAAM;CACN,sBAAsB;CACtB,UAAU,CAAC,WAAW,UAAU;CAChC,YAAY;EACV,SAAS;GAAE,MAAM;GAAU,WAAW;GAAI,WAAW;EAAI;EACzD,UAAU;GACR,MAAM;GACN,UAAU;GACV,OAAO;IACL,MAAM;IACN,sBAAsB;IACtB,UAAU;KAAC;KAAW;KAAW;KAAS;KAAY;IAAU;IAChE,YAAY;KACV,SAAS;MAAE,MAAM;MAAU,WAAW;MAAG,WAAW;KAAI;KACxD,SAAS,EAAE,MAAM,UAAU;KAC3B,OAAO;MAAE,MAAM;MAAU,SAAS;MAAG,SAAS;KAAG;KACjD,UAAU;MAAE,MAAM;MAAU,WAAW;MAAG,WAAW;KAAI;KACzD,UAAU;MAAE,MAAM;MAAU,MAAM;OAAC;OAAY;OAAS;OAAS;MAAM;KAAE;IAC3E;GACF;EACF;CACF;AACF;AAEA,SAAS,SAAS,MAAc,KAAa,OAAuB;CAClE,IAAI,KAAK,UAAU,KAAK,OAAO;CAC/B,OAAO,GAAG,KAAK,MAAM,GAAG,GAAG,EAAE,iBAAiB,KAAK,SAAS,IAAI,YAAY,MAAM;AACpF;AAEA,SAAS,YACP,OACA,MACQ;CACR,MAAM,aAAa,MAAM,YACtB,QAAQ,MAAM,EAAE,QAAQ,UAAU,KAAK,eAAe,CAAC,CACvD,KAAK,MAAM,aAAa,EAAE,KAAK,QAAQ,EAAE,SAAS,CAAC,CACnD,KAAK,MAAM;CAEd,MAAM,OAAO,MAAM,cAAc;CAEjC,OAAO;;;;;;;;;;EAUP,MAAM,YAAY;;EAElB,MAAM,gBAAgB,+BAA+B,MAAM,cAAc,mBAAmB,MAAM,uBAAuB,GAAG,QAAQ,GAAG;EACvI,MAAM,iBACL,KACE,GAAG,MACF,KAAK,IAAI,EAAE,KAAK,EAAE,KAAK,GAAG,EAAE,UAAU,SAAS,cAAc,EAAE,SAAS,MAAM,GAAG,CAAC,CAAC,CAAC,KAAK,KAAK,EAAE,KAAK,IACzG,CAAC,CACA,KAAK,IAAI,EAAE;;EAEZ,OAAO,qDAAqD,SAAS,MAAM,KAAK,cAAc,MAAM,EAAE,QAAQ,GAAG;EACjH,SAAS,YAAY,KAAK,gBAAgB,QAAQ,EAAE;;;;;;;;;;;;;;;;;;AAkBtD;;;;;;AASA,eAAsB,wBACpB,OACA,UAAuC,CAAC,GACH;CACrC,MAAM,QAAQ,KAAK,IAAI;CACvB,MAAM,aAAa,MAAM,iBAAiB;CAE1C,IAAI,eAAe,GACjB,OAAO;EACL,MAAM;EACN,SAAS;EACT,OAAO;EACP,cAAc;EACd,YAAY;EACZ,UAAU,CAAC;EACX,SAAS;EACT,YAAY;EACZ,SAAS;EACT,WAAW;EACX,OAAO;CACT;CAGF,MAAM,OAA8C;EAClD,OAAO,QAAQ,SAAS;EACxB,WAAW,QAAQ,aAAa;EAChC,WAAW,QAAQ,aAAa;EAChC,gBAAgB,QAAQ,kBAAkB;EAC1C,iBAAiB,QAAQ,mBAAmB;EAC5C,cAAc,QAAQ,gBAAgB;EACtC,KAAK,QAAQ,OAAO,CAAC;EACrB,YAAY,QAAQ,cAAc,IAAI,WAAW;EACjD,WAAW,QAAQ,aAAa;EAChC,UAAU,QAAQ,YAAY,CAAC;EAC/B,QAAQ,QAAQ,UAAU,IAAI,gBAAgB,CAAC,CAAC;EAChD,gBAAgB,QAAQ,kBAAkB;EAC1C,mBAAmB;GAAE,GAAG;GAA4B,GAAI,QAAQ,qBAAqB,CAAC;EAAG;CAC3F;CAMA,MAAM,oBAAoB,SAA8B;EACtD,IAAI,KAAK,mBAAmB,QAAQ,OAAO;EAC3C,IAAI,KAAK,UAAU,MAAM,OAAO,KAAK;EACrC,IAAI,KAAK,mBAAmB,cAC1B,OAAO,KAAK,kBAAkB,KAAK,cAAc,aAAa;EAEhE,OAAO;CACT;CACA,MAAM,eAAe,IAAI,IACvB,MAAM,iBAAiB,KAAK,MAAM,CAAC,EAAE,MAAM,iBAAiB,CAAC,CAAC,CAAC,CACjE;CAEA,IAAI;CACJ,IAAI;EACF,MAAM,UAAU;GACd,OAAO,KAAK;GACZ,UAAU,CACR;IACE,MAAM;IACN,SACE;GACJ,GACA;IAAE,MAAM;IAAiB,SAAS,YAAY,OAAO,IAAI;GAAE,CAC7D;GACA,YAAY;IAAE,MAAM;IAA0B,QAAQ;GAAgB;GACtE,aAAa;GACb,WAAW,KAAK;GAChB,WAAW,KAAK;EAClB;EACA,MAAM,OAAO,MAAM,KAAK,WAAW,YAAY;GAC7C,SAAS;GACT,OAAO,KAAK;GACZ,OAAO;GACP,OAAO,KAAK;GACZ,GAAI,OAAO,KAAK,KAAK,QAAQ,CAAC,CAAC,SAAS,IAAI,EAAE,MAAM,KAAK,SAAS,IAAI,CAAC;GACvE,eAAe,2BAA2B,SAAS,KAAK,GAAG;GAC3D,QAAQ,KAAK;GACb,UAAU,QAAQ,WAChB,YAA6D,SAAS;IACpE,GAAG,KAAK;IACR;IACA,gBAAgB;GAClB,CAAC;GACH,UAAU,EAAE,aAAa,mBAAmB,MAAM;GAClD,kBAAkB;EACpB,CAAC;EACD,UAAU,KAAK;EACf,IAAI,CAAC,KAAK,WAAW,MAAM,KAAK;EAChC,MAAM,EAAE,UAAU,KAAK;EAEvB,IAAI,CAAC,OAAO,YAAY,CAAC,MAAM,QAAQ,MAAM,QAAQ,GACnD,MAAM,IAAI,MAAM,uEAAqE;EAGvF,MAAM,WAA6B,MAAM,SAAS,KAAK,OAAO;GAC5D,SAAS,OAAO,EAAE,OAAO;GACzB,SAAS,QAAQ,EAAE,OAAO;GAC1B,OAAO,KAAK,IAAI,GAAG,KAAK,IAAI,IAAI,OAAO,EAAE,SAAS,CAAC,CAAC,CAAC;GACrD,UAAU,OAAO,EAAE,YAAY,EAAE;GACjC,UAAW;IAAC;IAAY;IAAS;IAAS;GAAM,CAAC,CAAW,SAAS,EAAE,QAAQ,IAC3E,EAAE,WACF;EACN,EAAE;EAEF,MAAM,eAAe,SAAS,QAAQ,MAAM,EAAE,WAAW,EAAE,SAAS,CAAC,CAAC,CAAC;EACvE,IAAI,YAAY;EAChB,IAAI,mBAAmB;EACvB,KAAK,MAAM,KAAK,UAAU;GACxB,MAAM,IAAI,aAAa,IAAI,EAAE,OAAO,KAAK;GACzC,aAAa;GACb,oBAAoB,IAAI,EAAE;EAC5B;EACA,MAAM,WACJ,YAAY,IACR,mBAAmB,YACnB,SAAS,QAAQ,GAAG,MAAM,IAAI,EAAE,OAAO,CAAC,IAAI,KAAK,IAAI,GAAG,SAAS,MAAM;EAE7E,OAAO;GACL,MAAM;GACN,SAAS;GACT,OAAO,QAAQ,WAAW,GAAA,CAAI,QAAQ,CAAC,CAAC;GACxC;GACA;GACA;GACA,SAAS,OAAO,MAAM,WAAW,EAAE;GACnC,YAAY,KAAK,IAAI,IAAI;GACzB,SAAS,KAAK,QAAQ,cAAc,OAAO,KAAK,QAAQ;GACxD,WAAW;EACb;CACF,SAAS,KAAK;EACZ,OAAO;GACL,MAAM;GACN,SAAS;GACT,OAAO;GACP,cAAc;GACd;GACA,UAAU,CAAC;GACX,SAAS;GACT,YAAY,KAAK,IAAI,IAAI;GACzB,SAAS,WAAW,CAAC,QAAQ,cAAc,QAAQ,UAAU;GAC7D,WAAW;GACX,OAAO,eAAe,QAAQ,IAAI,UAAU,OAAO,GAAG;EACxD;CACF;AACF"}
|
|
@@ -92,6 +92,35 @@ interface PairedEvalueSequence {
|
|
|
92
92
|
* non-negative martingale; once it crosses the threshold, it's crossed).
|
|
93
93
|
*/
|
|
94
94
|
declare function pairedEvalueSequence(deltas: number[], opts?: PairedEvalueOptions): PairedEvalueSequence;
|
|
95
|
+
/** Configuration for the best-case reachability preflight. */
|
|
96
|
+
interface SequentialCrossingHorizonOptions extends Pick<PairedEvalueOptions, 'alpha' | 'bound' | 'initialBetShrinkage'> {
|
|
97
|
+
/** Maximum pairs to simulate before returning `null`. Default 10,000. */
|
|
98
|
+
maxPairs?: number;
|
|
99
|
+
}
|
|
100
|
+
/**
|
|
101
|
+
* Best-case horizons under the exact betting policy used by
|
|
102
|
+
* {@link pairedEvalueSequence}.
|
|
103
|
+
*
|
|
104
|
+
* `evidencePairs` is the first all-at-bound pair count whose e-value reaches
|
|
105
|
+
* `1/alpha`. `decisionPairs` is the first count whose actual two-sided
|
|
106
|
+
* sequence verdict fires (`2/alpha` in the current policy). A design whose
|
|
107
|
+
* pair cap is below either relevant horizon is undecidable by construction and
|
|
108
|
+
* should be refused before model spend.
|
|
109
|
+
*/
|
|
110
|
+
interface SequentialCrossingHorizon {
|
|
111
|
+
readonly evidencePairs: number | null;
|
|
112
|
+
readonly decisionPairs: number | null;
|
|
113
|
+
readonly evidenceThreshold: number;
|
|
114
|
+
readonly decisionThreshold: number;
|
|
115
|
+
readonly maxPairs: number;
|
|
116
|
+
}
|
|
117
|
+
/**
|
|
118
|
+
* Compute the minimum pair count at which even perfect, all-at-bound data can
|
|
119
|
+
* reach the evidence and decision thresholds. The calculation intentionally
|
|
120
|
+
* calls the canonical sequence implementation rather than maintaining a
|
|
121
|
+
* second approximation that can drift from its bet schedule.
|
|
122
|
+
*/
|
|
123
|
+
declare function sequentialCrossingHorizon(opts?: SequentialCrossingHorizonOptions): SequentialCrossingHorizon;
|
|
95
124
|
interface InterimReleaseConfidenceInput {
|
|
96
125
|
/**
|
|
97
126
|
* One delta series per candidate (paired deltas vs comparator). Order
|
|
@@ -137,5 +166,5 @@ interface InterimReleaseConfidence {
|
|
|
137
166
|
*/
|
|
138
167
|
declare function evaluateInterimReleaseConfidence(input: InterimReleaseConfidenceInput): InterimReleaseConfidence;
|
|
139
168
|
//#endregion
|
|
140
|
-
export { PairedEvalueStep as a,
|
|
141
|
-
//# sourceMappingURL=sequential-
|
|
169
|
+
export { PairedEvalueStep as a, SequentialDecision as c, sequentialCrossingHorizon as d, PairedEvalueSequence as i, evaluateInterimReleaseConfidence as l, InterimReleaseConfidenceInput as n, SequentialCrossingHorizon as o, PairedEvalueOptions as r, SequentialCrossingHorizonOptions as s, InterimReleaseConfidence as t, pairedEvalueSequence as u };
|
|
170
|
+
//# sourceMappingURL=sequential-BhsrMupG.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"sequential-BhsrMupG.d.ts","names":[],"sources":["../src/sequential.ts"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;KAuCY;UAEK;;;;;;EAMf;;EAEA;;;;;;EAMA;IAAS;IAAa;;;EAEtB;;UAGe;;EAEf;EACA;;EAEA;;EAEA;;EAEA;EACA;;EAEA,UAAU;;UAGK;EACf,OAAO;;EAEP,eAAe;;EAEf;;EAEA;;;;;;;;;;;iBAYc,qBACd,kBACA,OAAM,sBACL;;UA0Ec,yCACP,KAAK;;EAEb;;;;;;;;;;;;UAae;WACN;WACA;WACA;WACA;WACA;;;;;;;;iBAWK,0BACd,OAAM,mCACL;UAmCc;;;;;EAKf,aAAa;IAAQ;IAAqB;;EAC1C;EACA;EACA;IAAS;IAAa;;;UAGP;EACf,YAAY;IACV;IACA,UAAU;IACV;IACA;IACA;IACA;IACA;IACA;;;;;;;EAOF;IAAkB,UAAU;IAAoB;;;;;;;;;iBASlC,iCACd,OAAO,gCACN"}
|
|
@@ -13,9 +13,9 @@ function pairedEvalueSequence(deltas, opts = {}) {
|
|
|
13
13
|
const alpha = opts.alpha ?? .05;
|
|
14
14
|
const initialShrink = opts.initialBetShrinkage ?? .5;
|
|
15
15
|
const rope = opts.rope ?? null;
|
|
16
|
-
|
|
17
|
-
if (alpha <= 0 || alpha >= 1) throw new Error("pairedEvalueSequence: alpha must be in (0,1)");
|
|
16
|
+
assertSequentialScalarConfiguration(c, alpha, initialShrink, "pairedEvalueSequence");
|
|
18
17
|
if (rope && !(Number.isFinite(rope.low) && Number.isFinite(rope.high) && rope.low <= rope.high)) throw new Error("pairedEvalueSequence: rope must satisfy low ≤ high");
|
|
18
|
+
const { decisionThreshold } = sequentialThresholds(alpha);
|
|
19
19
|
const steps = [];
|
|
20
20
|
let clipped = false;
|
|
21
21
|
let evalue = 1;
|
|
@@ -25,6 +25,7 @@ function pairedEvalueSequence(deltas, opts = {}) {
|
|
|
25
25
|
let count = 0;
|
|
26
26
|
for (let i = 0; i < deltas.length; i++) {
|
|
27
27
|
let d = deltas[i];
|
|
28
|
+
if (!Number.isFinite(d)) throw new Error(`pairedEvalueSequence: delta[${i}] must be finite`);
|
|
28
29
|
if (d < -c || d > c) {
|
|
29
30
|
d = Math.max(-c, Math.min(c, d));
|
|
30
31
|
clipped = true;
|
|
@@ -46,8 +47,8 @@ function pairedEvalueSequence(deltas, opts = {}) {
|
|
|
46
47
|
const cs = empiricalBernsteinCs(sum, sumSq, count, c, alpha);
|
|
47
48
|
let decision = "continue";
|
|
48
49
|
if (rope && cs.low >= rope.low && cs.high <= rope.high) decision = "equivalent";
|
|
49
|
-
else if (evalue >=
|
|
50
|
-
else if (evalue >=
|
|
50
|
+
else if (evalue >= decisionThreshold && muHat > 0) decision = "promote_now";
|
|
51
|
+
else if (evalue >= decisionThreshold && muHat < 0) decision = "reject_now";
|
|
51
52
|
else if (rope && cs.high < rope.low) decision = "reject_now";
|
|
52
53
|
if (decision !== "continue" && decisionFiredAt === null) decisionFiredAt = t;
|
|
53
54
|
steps.push({
|
|
@@ -67,6 +68,36 @@ function pairedEvalueSequence(deltas, opts = {}) {
|
|
|
67
68
|
clipped
|
|
68
69
|
};
|
|
69
70
|
}
|
|
71
|
+
const MAX_SEQUENTIAL_CROSSING_PAIRS = 1e6;
|
|
72
|
+
/**
|
|
73
|
+
* Compute the minimum pair count at which even perfect, all-at-bound data can
|
|
74
|
+
* reach the evidence and decision thresholds. The calculation intentionally
|
|
75
|
+
* calls the canonical sequence implementation rather than maintaining a
|
|
76
|
+
* second approximation that can drift from its bet schedule.
|
|
77
|
+
*/
|
|
78
|
+
function sequentialCrossingHorizon(opts = {}) {
|
|
79
|
+
const bound = opts.bound ?? 1;
|
|
80
|
+
const alpha = opts.alpha ?? .05;
|
|
81
|
+
const initialBetShrinkage = opts.initialBetShrinkage ?? .5;
|
|
82
|
+
const maxPairs = opts.maxPairs ?? 1e4;
|
|
83
|
+
assertSequentialScalarConfiguration(bound, alpha, initialBetShrinkage, "sequentialCrossingHorizon");
|
|
84
|
+
if (!Number.isSafeInteger(maxPairs) || maxPairs < 1) throw new Error("sequentialCrossingHorizon: maxPairs must be a positive safe integer");
|
|
85
|
+
if (maxPairs > MAX_SEQUENTIAL_CROSSING_PAIRS) throw new Error(`sequentialCrossingHorizon: maxPairs must be <= ${MAX_SEQUENTIAL_CROSSING_PAIRS}`);
|
|
86
|
+
const sequence = pairedEvalueSequence(Array(maxPairs).fill(bound), {
|
|
87
|
+
bound,
|
|
88
|
+
alpha,
|
|
89
|
+
initialBetShrinkage
|
|
90
|
+
});
|
|
91
|
+
const { evidenceThreshold, decisionThreshold } = sequentialThresholds(alpha);
|
|
92
|
+
const evidencePairs = sequence.steps.find((step) => step.evalue >= evidenceThreshold)?.t ?? null;
|
|
93
|
+
return Object.freeze({
|
|
94
|
+
evidencePairs,
|
|
95
|
+
decisionPairs: sequence.decisionFiredAt,
|
|
96
|
+
evidenceThreshold,
|
|
97
|
+
decisionThreshold,
|
|
98
|
+
maxPairs
|
|
99
|
+
});
|
|
100
|
+
}
|
|
70
101
|
/**
|
|
71
102
|
* Run interim sequential analyses across many candidates at once,
|
|
72
103
|
* preserving the time-uniform α guarantee for each candidate's series and
|
|
@@ -123,6 +154,17 @@ function evaluateInterimReleaseConfidence(input) {
|
|
|
123
154
|
}
|
|
124
155
|
};
|
|
125
156
|
}
|
|
157
|
+
function assertSequentialScalarConfiguration(bound, alpha, initialBetShrinkage, context) {
|
|
158
|
+
if (!Number.isFinite(bound) || bound <= 0) throw new Error(`${context}: bound must be a finite number > 0`);
|
|
159
|
+
if (!Number.isFinite(alpha) || alpha <= 0 || alpha >= 1) throw new Error(`${context}: alpha must be a finite number in (0,1)`);
|
|
160
|
+
if (!Number.isFinite(initialBetShrinkage) || initialBetShrinkage <= 0 || initialBetShrinkage > 1) throw new Error(`${context}: initialBetShrinkage must be a finite number in (0,1]`);
|
|
161
|
+
}
|
|
162
|
+
function sequentialThresholds(alpha) {
|
|
163
|
+
return {
|
|
164
|
+
evidenceThreshold: 1 / alpha,
|
|
165
|
+
decisionThreshold: 2 / alpha
|
|
166
|
+
};
|
|
167
|
+
}
|
|
126
168
|
/**
|
|
127
169
|
* Empirical Bernstein confidence sequence on the mean of bounded variables.
|
|
128
170
|
* Adapted from Howard et al. (2021) §4.4. Provides a time-uniform CI on
|
|
@@ -143,6 +185,6 @@ function empiricalBernsteinCs(sum, sumSq, n, bound, alpha) {
|
|
|
143
185
|
};
|
|
144
186
|
}
|
|
145
187
|
//#endregion
|
|
146
|
-
export { pairedEvalueSequence as n, evaluateInterimReleaseConfidence as t };
|
|
188
|
+
export { pairedEvalueSequence as n, sequentialCrossingHorizon as r, evaluateInterimReleaseConfidence as t };
|
|
147
189
|
|
|
148
|
-
//# sourceMappingURL=sequential-
|
|
190
|
+
//# sourceMappingURL=sequential-CzK5DarL.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"sequential-CzK5DarL.js","names":[],"sources":["../src/sequential.ts"],"sourcesContent":["/**\n * Always-valid sequential evaluation.\n *\n * `researchReport` assumes a single pre-specified analysis. Real\n * consumers run campaigns weekly / nightly / per-PR; each new run silently\n * inflates the false-discovery rate, because the BH-FDR guarantee is for\n * the *first* look, not the 47th. Without time-uniform inference,\n * launch-decision teams either (a) don't peek, which forfeits the cost\n * advantage of stop-when-decisive, or (b) peek and pretend they didn't,\n * which forfeits scientific validity.\n *\n * This module ships **e-value-based confidence sequences** for paired\n * bounded outcomes. The methodology is the predictable plug-in betting\n * martingale of Waudby-Smith & Ramdas (2024) — provably valid at *any*\n * stopping time. Concretely:\n *\n * For paired deltas D_1, D_2, … ∈ [-c, c] with the null H_0: E[D] ≤ 0,\n * a betting fraction λ_i is chosen using only D_{1..i-1} (predictable\n * plug-in), and the running e-value is\n *\n * E_t = ∏_{i=1}^{t} (1 + λ_i · D_i)\n *\n * E_t is a non-negative martingale under H_0 with E[E_t] ≤ 1, so by\n * Ville's inequality, P(∃ t : E_t ≥ 1/α) ≤ α — we can reject the null\n * at any time without inflating the type-I error.\n *\n * Combined with `runEvalCampaign`, every consumer running rolling\n * campaigns gains the ability to ship the moment evidence is decisive,\n * stop-early on dead-on-arrival variants, and accumulate evidence across\n * partial runs without spending the FDR budget. No new sweep is wasted.\n *\n * References:\n * - Howard, S. R., Ramdas, A., McAuliffe, J., Sekhon, J. (2021).\n * Time-uniform, nonparametric, nonasymptotic confidence sequences.\n * Annals of Statistics, 49(2), 1055–1080.\n * - Waudby-Smith, I., Ramdas, A. (2024). Estimating means of bounded\n * random variables by betting. JRSS B, 86(1), 1–27.\n */\n\nexport type SequentialDecision = 'promote_now' | 'continue' | 'reject_now' | 'equivalent'\n\nexport interface PairedEvalueOptions {\n /**\n * Bound on |delta|. Default 1 (matching most score scales). Must satisfy\n * c > 0; deltas outside [-c, c] are clipped with a warning attached to\n * the return value.\n */\n bound?: number\n /** Target Type-I error. Default 0.05. */\n alpha?: number\n /**\n * Region of Practical Equivalence on the *mean* paired delta. When\n * supplied, the verdict can return `'equivalent'` once the running\n * confidence sequence on the mean is fully contained in [low, high].\n */\n rope?: { low: number; high: number }\n /** Initial bet shrinkage (0 < scale ≤ 1). Default 0.5 — empirically robust. */\n initialBetShrinkage?: number\n}\n\nexport interface PairedEvalueStep {\n /** 1-indexed observation count. */\n t: number\n delta: number\n /** Running e-value E_t = ∏ (1 + λ_i · D_i). */\n evalue: number\n /** Time-uniform p-value at stopping time t. */\n pValue: number\n /** Lower bound of the empirical Bernstein confidence sequence at level 1-α. */\n csLow: number\n csHigh: number\n /** Verdict at this stopping time. */\n decision: SequentialDecision\n}\n\nexport interface PairedEvalueSequence {\n steps: PairedEvalueStep[]\n /** The decision at the final step. */\n finalDecision: SequentialDecision\n /** Index (1-based) at which a non-`continue` decision first fired, or null. */\n decisionFiredAt: number | null\n /** True if any deltas were clipped to [-bound, bound]. */\n clipped: boolean\n}\n\n/**\n * Run the paired e-value sequence over an in-order delta stream.\n *\n * Use for *streaming* / interim analyses: pass the deltas you have so\n * far, get the verdict at every prefix length. The decision is\n * monotone-stable in the sense that once `'reject_now'` or `'promote_now'`\n * fires, the verdict at later steps remains decisive (the e-value is a\n * non-negative martingale; once it crosses the threshold, it's crossed).\n */\nexport function pairedEvalueSequence(\n deltas: number[],\n opts: PairedEvalueOptions = {},\n): PairedEvalueSequence {\n const c = opts.bound ?? 1\n const alpha = opts.alpha ?? 0.05\n const initialShrink = opts.initialBetShrinkage ?? 0.5\n const rope = opts.rope ?? null\n assertSequentialScalarConfiguration(c, alpha, initialShrink, 'pairedEvalueSequence')\n if (rope && !(Number.isFinite(rope.low) && Number.isFinite(rope.high) && rope.low <= rope.high)) {\n throw new Error('pairedEvalueSequence: rope must satisfy low ≤ high')\n }\n const { decisionThreshold } = sequentialThresholds(alpha)\n\n const steps: PairedEvalueStep[] = []\n let clipped = false\n let evalue = 1\n let decisionFiredAt: number | null = null\n\n // Running statistics (using only D_{1..i-1} for the bet → predictable plug-in).\n let sum = 0\n let sumSq = 0\n let count = 0\n\n for (let i = 0; i < deltas.length; i++) {\n let d = deltas[i]!\n if (!Number.isFinite(d)) {\n throw new Error(`pairedEvalueSequence: delta[${i}] must be finite`)\n }\n if (d < -c || d > c) {\n d = Math.max(-c, Math.min(c, d))\n clipped = true\n }\n\n // Predictable plug-in bet (positive λ tests for E[D] > 0; we run a two-sided\n // test by tracking the symmetric e-value via |bet|).\n // λ_i ∝ mean / (variance + bound^2). Shrink early to avoid overbetting.\n const muHat = count === 0 ? 0 : sum / count\n const varHat = count === 0 ? c * c : Math.max(1e-12, sumSq / count - muHat * muHat)\n const t = i + 1\n const shrink = initialShrink * Math.min(1, count / 32) // anneal toward 1\n let lambda = (muHat / (varHat + c * c)) * shrink\n // Clip to ensure 1 + λ·D > 0 for all |D| ≤ c (so the e-value stays non-negative).\n const lambdaMax = 0.99 / c\n if (lambda > lambdaMax) lambda = lambdaMax\n if (lambda < -lambdaMax) lambda = -lambdaMax\n\n evalue = evalue * (1 + lambda * d)\n if (!Number.isFinite(evalue) || evalue < 0) evalue = 0\n\n sum += d\n sumSq += d * d\n count += 1\n\n const pValue = Math.min(1, 1 / Math.max(evalue, 1e-300))\n\n // Empirical Bernstein confidence sequence on the mean. Howard et al.\n // (2021), Theorem 4.4 with σ̂² the running sample variance and a\n // calibration constant tuned for two-sided coverage at level 1 - α.\n const cs = empiricalBernsteinCs(sum, sumSq, count, c, alpha)\n\n let decision: SequentialDecision = 'continue'\n if (rope && cs.low >= rope.low && cs.high <= rope.high) decision = 'equivalent'\n else if (evalue >= decisionThreshold && muHat > 0) decision = 'promote_now'\n else if (evalue >= decisionThreshold && muHat < 0) decision = 'reject_now'\n else if (rope && cs.high < rope.low) decision = 'reject_now'\n\n if (decision !== 'continue' && decisionFiredAt === null) decisionFiredAt = t\n\n steps.push({ t, delta: d, evalue, pValue, csLow: cs.low, csHigh: cs.high, decision })\n }\n\n const finalDecision = steps.length === 0 ? 'continue' : steps[steps.length - 1]!.decision\n return { steps, finalDecision, decisionFiredAt, clipped }\n}\n\n/** Configuration for the best-case reachability preflight. */\nexport interface SequentialCrossingHorizonOptions\n extends Pick<PairedEvalueOptions, 'alpha' | 'bound' | 'initialBetShrinkage'> {\n /** Maximum pairs to simulate before returning `null`. Default 10,000. */\n maxPairs?: number\n}\n\n/**\n * Best-case horizons under the exact betting policy used by\n * {@link pairedEvalueSequence}.\n *\n * `evidencePairs` is the first all-at-bound pair count whose e-value reaches\n * `1/alpha`. `decisionPairs` is the first count whose actual two-sided\n * sequence verdict fires (`2/alpha` in the current policy). A design whose\n * pair cap is below either relevant horizon is undecidable by construction and\n * should be refused before model spend.\n */\nexport interface SequentialCrossingHorizon {\n readonly evidencePairs: number | null\n readonly decisionPairs: number | null\n readonly evidenceThreshold: number\n readonly decisionThreshold: number\n readonly maxPairs: number\n}\n\nconst MAX_SEQUENTIAL_CROSSING_PAIRS = 1_000_000\n\n/**\n * Compute the minimum pair count at which even perfect, all-at-bound data can\n * reach the evidence and decision thresholds. The calculation intentionally\n * calls the canonical sequence implementation rather than maintaining a\n * second approximation that can drift from its bet schedule.\n */\nexport function sequentialCrossingHorizon(\n opts: SequentialCrossingHorizonOptions = {},\n): SequentialCrossingHorizon {\n const bound = opts.bound ?? 1\n const alpha = opts.alpha ?? 0.05\n const initialBetShrinkage = opts.initialBetShrinkage ?? 0.5\n const maxPairs = opts.maxPairs ?? 10_000\n assertSequentialScalarConfiguration(\n bound,\n alpha,\n initialBetShrinkage,\n 'sequentialCrossingHorizon',\n )\n if (!Number.isSafeInteger(maxPairs) || maxPairs < 1) {\n throw new Error('sequentialCrossingHorizon: maxPairs must be a positive safe integer')\n }\n if (maxPairs > MAX_SEQUENTIAL_CROSSING_PAIRS) {\n throw new Error(\n `sequentialCrossingHorizon: maxPairs must be <= ${MAX_SEQUENTIAL_CROSSING_PAIRS}`,\n )\n }\n const sequence = pairedEvalueSequence(Array<number>(maxPairs).fill(bound), {\n bound,\n alpha,\n initialBetShrinkage,\n })\n const { evidenceThreshold, decisionThreshold } = sequentialThresholds(alpha)\n const evidencePairs = sequence.steps.find((step) => step.evalue >= evidenceThreshold)?.t ?? null\n return Object.freeze({\n evidencePairs,\n decisionPairs: sequence.decisionFiredAt,\n evidenceThreshold,\n decisionThreshold,\n maxPairs,\n })\n}\n\nexport interface InterimReleaseConfidenceInput {\n /**\n * One delta series per candidate (paired deltas vs comparator). Order\n * within a series is the order the campaigns were run.\n */\n deltaSeries: Array<{ candidateId: string; deltas: number[] }>\n alpha?: number\n bound?: number\n rope?: { low: number; high: number }\n}\n\nexport interface InterimReleaseConfidence {\n candidates: Array<{\n candidateId: string\n decision: SequentialDecision\n decisionFiredAt: number | null\n finalEvalue: number\n finalPValue: number\n pairs: number\n csLow: number\n csHigh: number\n }>\n /**\n * Campaign-level recommendation: pick the strongest 'promote_now', else\n * 'continue' if any candidate is still live, else 'reject_now' if every\n * candidate is dead, else 'equivalent'.\n */\n recommendation: { decision: SequentialDecision; candidateId: string | null }\n}\n\n/**\n * Run interim sequential analyses across many candidates at once,\n * preserving the time-uniform α guarantee for each candidate's series and\n * synthesising a campaign-level recommendation. Designed to be called on\n * every campaign tick — the recommendation is anytime-valid.\n */\nexport function evaluateInterimReleaseConfidence(\n input: InterimReleaseConfidenceInput,\n): InterimReleaseConfidence {\n const candidates = input.deltaSeries.map((s) => {\n const seq = pairedEvalueSequence(s.deltas, {\n alpha: input.alpha,\n bound: input.bound,\n rope: input.rope,\n })\n const last = seq.steps[seq.steps.length - 1]\n return {\n candidateId: s.candidateId,\n decision: seq.finalDecision,\n decisionFiredAt: seq.decisionFiredAt,\n finalEvalue: last?.evalue ?? 1,\n finalPValue: last?.pValue ?? 1,\n pairs: seq.steps.length,\n csLow: last?.csLow ?? Number.NEGATIVE_INFINITY,\n csHigh: last?.csHigh ?? Number.POSITIVE_INFINITY,\n }\n })\n\n const promote = candidates.find((c) => c.decision === 'promote_now')\n if (promote)\n return {\n candidates,\n recommendation: { decision: 'promote_now', candidateId: promote.candidateId },\n }\n const live = candidates.find((c) => c.decision === 'continue')\n if (live) return { candidates, recommendation: { decision: 'continue', candidateId: null } }\n const equiv = candidates.find((c) => c.decision === 'equivalent')\n if (equiv)\n return {\n candidates,\n recommendation: { decision: 'equivalent', candidateId: equiv.candidateId },\n }\n return { candidates, recommendation: { decision: 'reject_now', candidateId: null } }\n}\n\n// ── Internals ────────────────────────────────────────────────────────────\n\nfunction assertSequentialScalarConfiguration(\n bound: number,\n alpha: number,\n initialBetShrinkage: number,\n context: string,\n): void {\n if (!Number.isFinite(bound) || bound <= 0) {\n throw new Error(`${context}: bound must be a finite number > 0`)\n }\n if (!Number.isFinite(alpha) || alpha <= 0 || alpha >= 1) {\n throw new Error(`${context}: alpha must be a finite number in (0,1)`)\n }\n if (\n !Number.isFinite(initialBetShrinkage) ||\n initialBetShrinkage <= 0 ||\n initialBetShrinkage > 1\n ) {\n throw new Error(`${context}: initialBetShrinkage must be a finite number in (0,1]`)\n }\n}\n\nfunction sequentialThresholds(alpha: number): {\n evidenceThreshold: number\n decisionThreshold: number\n} {\n return { evidenceThreshold: 1 / alpha, decisionThreshold: 2 / alpha }\n}\n\n/**\n * Empirical Bernstein confidence sequence on the mean of bounded variables.\n * Adapted from Howard et al. (2021) §4.4. Provides a time-uniform CI on\n * the running mean; valid at every stopping time.\n */\nfunction empiricalBernsteinCs(\n sum: number,\n sumSq: number,\n n: number,\n bound: number,\n alpha: number,\n): { low: number; high: number } {\n if (n === 0) return { low: -bound, high: bound }\n const mean = sum / n\n const variance = Math.max(0, sumSq / n - mean * mean)\n // Iterated-log calibration constant. The 1.7 exponent matches the\n // recommended choice in Howard et al. for two-sided coverage at level\n // 1 - α with mild log-corrections; tightening further requires a\n // tuned mixture and is out of scope.\n const psi = Math.log(2 / alpha) + 1.7 * Math.log(Math.log(Math.max(Math.E, n)) + 1)\n const radius = Math.sqrt((2 * variance * psi) / n) + (3 * bound * psi) / n\n return { low: mean - radius, high: mean + radius }\n}\n"],"mappings":";;;;;;;;;;AA8FA,SAAgB,qBACd,QACA,OAA4B,CAAC,GACP;CACtB,MAAM,IAAI,KAAK,SAAS;CACxB,MAAM,QAAQ,KAAK,SAAS;CAC5B,MAAM,gBAAgB,KAAK,uBAAuB;CAClD,MAAM,OAAO,KAAK,QAAQ;CAC1B,oCAAoC,GAAG,OAAO,eAAe,sBAAsB;CACnF,IAAI,QAAQ,EAAE,OAAO,SAAS,KAAK,GAAG,KAAK,OAAO,SAAS,KAAK,IAAI,KAAK,KAAK,OAAO,KAAK,OACxF,MAAM,IAAI,MAAM,oDAAoD;CAEtE,MAAM,EAAE,sBAAsB,qBAAqB,KAAK;CAExD,MAAM,QAA4B,CAAC;CACnC,IAAI,UAAU;CACd,IAAI,SAAS;CACb,IAAI,kBAAiC;CAGrC,IAAI,MAAM;CACV,IAAI,QAAQ;CACZ,IAAI,QAAQ;CAEZ,KAAK,IAAI,IAAI,GAAG,IAAI,OAAO,QAAQ,KAAK;EACtC,IAAI,IAAI,OAAO;EACf,IAAI,CAAC,OAAO,SAAS,CAAC,GACpB,MAAM,IAAI,MAAM,+BAA+B,EAAE,iBAAiB;EAEpE,IAAI,IAAI,CAAC,KAAK,IAAI,GAAG;GACnB,IAAI,KAAK,IAAI,CAAC,GAAG,KAAK,IAAI,GAAG,CAAC,CAAC;GAC/B,UAAU;EACZ;EAKA,MAAM,QAAQ,UAAU,IAAI,IAAI,MAAM;EACtC,MAAM,SAAS,UAAU,IAAI,IAAI,IAAI,KAAK,IAAI,OAAO,QAAQ,QAAQ,QAAQ,KAAK;EAClF,MAAM,IAAI,IAAI;EACd,MAAM,SAAS,gBAAgB,KAAK,IAAI,GAAG,QAAQ,EAAE;EACrD,IAAI,SAAU,SAAS,SAAS,IAAI,KAAM;EAE1C,MAAM,YAAY,MAAO;EACzB,IAAI,SAAS,WAAW,SAAS;EACjC,IAAI,SAAS,CAAC,WAAW,SAAS,CAAC;EAEnC,SAAS,UAAU,IAAI,SAAS;EAChC,IAAI,CAAC,OAAO,SAAS,MAAM,KAAK,SAAS,GAAG,SAAS;EAErD,OAAO;EACP,SAAS,IAAI;EACb,SAAS;EAET,MAAM,SAAS,KAAK,IAAI,GAAG,IAAI,KAAK,IAAI,QAAQ,MAAM,CAAC;EAKvD,MAAM,KAAK,qBAAqB,KAAK,OAAO,OAAO,GAAG,KAAK;EAE3D,IAAI,WAA+B;EACnC,IAAI,QAAQ,GAAG,OAAO,KAAK,OAAO,GAAG,QAAQ,KAAK,MAAM,WAAW;OAC9D,IAAI,UAAU,qBAAqB,QAAQ,GAAG,WAAW;OACzD,IAAI,UAAU,qBAAqB,QAAQ,GAAG,WAAW;OACzD,IAAI,QAAQ,GAAG,OAAO,KAAK,KAAK,WAAW;EAEhD,IAAI,aAAa,cAAc,oBAAoB,MAAM,kBAAkB;EAE3E,MAAM,KAAK;GAAE;GAAG,OAAO;GAAG;GAAQ;GAAQ,OAAO,GAAG;GAAK,QAAQ,GAAG;GAAM;EAAS,CAAC;CACtF;CAGA,OAAO;EAAE;EAAO,eADM,MAAM,WAAW,IAAI,aAAa,MAAM,MAAM,SAAS,EAAE,CAAE;EAClD;EAAiB;CAAQ;AAC1D;AA2BA,MAAM,gCAAgC;;;;;;;AAQtC,SAAgB,0BACd,OAAyC,CAAC,GACf;CAC3B,MAAM,QAAQ,KAAK,SAAS;CAC5B,MAAM,QAAQ,KAAK,SAAS;CAC5B,MAAM,sBAAsB,KAAK,uBAAuB;CACxD,MAAM,WAAW,KAAK,YAAY;CAClC,oCACE,OACA,OACA,qBACA,2BACF;CACA,IAAI,CAAC,OAAO,cAAc,QAAQ,KAAK,WAAW,GAChD,MAAM,IAAI,MAAM,qEAAqE;CAEvF,IAAI,WAAW,+BACb,MAAM,IAAI,MACR,kDAAkD,+BACpD;CAEF,MAAM,WAAW,qBAAqB,MAAc,QAAQ,CAAC,CAAC,KAAK,KAAK,GAAG;EACzE;EACA;EACA;CACF,CAAC;CACD,MAAM,EAAE,mBAAmB,sBAAsB,qBAAqB,KAAK;CAC3E,MAAM,gBAAgB,SAAS,MAAM,MAAM,SAAS,KAAK,UAAU,iBAAiB,CAAC,EAAE,KAAK;CAC5F,OAAO,OAAO,OAAO;EACnB;EACA,eAAe,SAAS;EACxB;EACA;EACA;CACF,CAAC;AACH;;;;;;;AAsCA,SAAgB,iCACd,OAC0B;CAC1B,MAAM,aAAa,MAAM,YAAY,KAAK,MAAM;EAC9C,MAAM,MAAM,qBAAqB,EAAE,QAAQ;GACzC,OAAO,MAAM;GACb,OAAO,MAAM;GACb,MAAM,MAAM;EACd,CAAC;EACD,MAAM,OAAO,IAAI,MAAM,IAAI,MAAM,SAAS;EAC1C,OAAO;GACL,aAAa,EAAE;GACf,UAAU,IAAI;GACd,iBAAiB,IAAI;GACrB,aAAa,MAAM,UAAU;GAC7B,aAAa,MAAM,UAAU;GAC7B,OAAO,IAAI,MAAM;GACjB,OAAO,MAAM,SAAS,OAAO;GAC7B,QAAQ,MAAM,UAAU,OAAO;EACjC;CACF,CAAC;CAED,MAAM,UAAU,WAAW,MAAM,MAAM,EAAE,aAAa,aAAa;CACnE,IAAI,SACF,OAAO;EACL;EACA,gBAAgB;GAAE,UAAU;GAAe,aAAa,QAAQ;EAAY;CAC9E;CAEF,IADa,WAAW,MAAM,MAAM,EAAE,aAAa,UAC5C,GAAG,OAAO;EAAE;EAAY,gBAAgB;GAAE,UAAU;GAAY,aAAa;EAAK;CAAE;CAC3F,MAAM,QAAQ,WAAW,MAAM,MAAM,EAAE,aAAa,YAAY;CAChE,IAAI,OACF,OAAO;EACL;EACA,gBAAgB;GAAE,UAAU;GAAc,aAAa,MAAM;EAAY;CAC3E;CACF,OAAO;EAAE;EAAY,gBAAgB;GAAE,UAAU;GAAc,aAAa;EAAK;CAAE;AACrF;AAIA,SAAS,oCACP,OACA,OACA,qBACA,SACM;CACN,IAAI,CAAC,OAAO,SAAS,KAAK,KAAK,SAAS,GACtC,MAAM,IAAI,MAAM,GAAG,QAAQ,oCAAoC;CAEjE,IAAI,CAAC,OAAO,SAAS,KAAK,KAAK,SAAS,KAAK,SAAS,GACpD,MAAM,IAAI,MAAM,GAAG,QAAQ,yCAAyC;CAEtE,IACE,CAAC,OAAO,SAAS,mBAAmB,KACpC,uBAAuB,KACvB,sBAAsB,GAEtB,MAAM,IAAI,MAAM,GAAG,QAAQ,uDAAuD;AAEtF;AAEA,SAAS,qBAAqB,OAG5B;CACA,OAAO;EAAE,mBAAmB,IAAI;EAAO,mBAAmB,IAAI;CAAM;AACtE;;;;;;AAOA,SAAS,qBACP,KACA,OACA,GACA,OACA,OAC+B;CAC/B,IAAI,MAAM,GAAG,OAAO;EAAE,KAAK,CAAC;EAAO,MAAM;CAAM;CAC/C,MAAM,OAAO,MAAM;CACnB,MAAM,WAAW,KAAK,IAAI,GAAG,QAAQ,IAAI,OAAO,IAAI;CAKpD,MAAM,MAAM,KAAK,IAAI,IAAI,KAAK,IAAI,MAAM,KAAK,IAAI,KAAK,IAAI,KAAK,IAAI,KAAK,GAAG,CAAC,CAAC,IAAI,CAAC;CAClF,MAAM,SAAS,KAAK,KAAM,IAAI,WAAW,MAAO,CAAC,IAAK,IAAI,QAAQ,MAAO;CACzE,OAAO;EAAE,KAAK,OAAO;EAAQ,MAAM,OAAO;CAAO;AACnD"}
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"series-convergence-CjO2QdRW.js","names":[],"sources":["../src/series-convergence.ts"],"sourcesContent":["/**\n * Series convergence — detects whether a sequence of scalar measurements\n * is stabilizing, drifting, or noisy
|
|
1
|
+
{"version":3,"file":"series-convergence-CjO2QdRW.js","names":[],"sources":["../src/series-convergence.ts"],"sourcesContent":["/**\n * Series convergence — detects whether a sequence of scalar measurements\n * is stabilizing, drifting, or noisy. It reads drift *across* runs, e.g.\n * \"are my nightly eval scores stabilizing?\".\n *\n * Three signals:\n * - stabilized: last K values have low variance (< epsilon) — done\n * - drifting: recent trend is monotonic and beyond noise — regressing or improving\n * - noisy: neither — keep iterating, but flag as untrustworthy for gating\n */\n\nexport interface SeriesConvergenceOptions {\n /** Window size for \"recent\" analysis (default 5). */\n window?: number\n /** Coefficient-of-variation threshold below which the window is stabilized (default 0.05 = 5%). */\n stableCv?: number\n /** Minimum monotone run length to call drift (default 3). */\n driftRun?: number\n}\n\nexport interface SeriesConvergenceResult {\n state: 'stabilized' | 'drifting-up' | 'drifting-down' | 'noisy' | 'insufficient-data'\n windowMean: number\n windowCv: number\n /** Longest monotonic run at the tail of the series (positive for up, negative for down). */\n tailRun: number\n /** True when n ≥ window AND windowCv ≤ stableCv. */\n stable: boolean\n}\n\nexport function analyzeSeries(\n values: number[],\n options: SeriesConvergenceOptions = {},\n): SeriesConvergenceResult {\n const window = options.window ?? 5\n const stableCv = options.stableCv ?? 0.05\n const driftRun = options.driftRun ?? 3\n\n if (values.length < Math.max(2, Math.min(window, 3))) {\n return { state: 'insufficient-data', windowMean: 0, windowCv: 0, tailRun: 0, stable: false }\n }\n\n const tail = values.slice(-window)\n const mean = tail.reduce((a, b) => a + b, 0) / tail.length\n const variance = tail.reduce((acc, v) => acc + (v - mean) ** 2, 0) / tail.length\n const stdDev = Math.sqrt(variance)\n const refMean = Math.abs(mean) > 1e-9 ? Math.abs(mean) : 1\n const cv = stdDev / refMean\n const stable = tail.length >= window && cv <= stableCv\n\n // Tail monotonic run: count how many consecutive strictly-increasing (or decreasing)\n // steps end at the final value.\n let tailRun = 0\n let direction: 1 | -1 | 0 = 0\n for (let i = values.length - 1; i > 0; i--) {\n const delta = values[i]! - values[i - 1]!\n if (delta === 0) break\n const dir = delta > 0 ? 1 : -1\n if (direction === 0) direction = dir\n if (dir !== direction) break\n tailRun += dir\n }\n\n let state: SeriesConvergenceResult['state']\n if (stable) {\n state = 'stabilized'\n } else if (Math.abs(tailRun) >= driftRun) {\n state = tailRun > 0 ? 'drifting-up' : 'drifting-down'\n } else {\n state = 'noisy'\n }\n\n return { state, windowMean: mean, windowCv: cv, tailRun, stable }\n}\n"],"mappings":";AA8BA,SAAgB,cACd,QACA,UAAoC,CAAC,GACZ;CACzB,MAAM,SAAS,QAAQ,UAAU;CACjC,MAAM,WAAW,QAAQ,YAAY;CACrC,MAAM,WAAW,QAAQ,YAAY;CAErC,IAAI,OAAO,SAAS,KAAK,IAAI,GAAG,KAAK,IAAI,QAAQ,CAAC,CAAC,GACjD,OAAO;EAAE,OAAO;EAAqB,YAAY;EAAG,UAAU;EAAG,SAAS;EAAG,QAAQ;CAAM;CAG7F,MAAM,OAAO,OAAO,MAAM,CAAC,MAAM;CACjC,MAAM,OAAO,KAAK,QAAQ,GAAG,MAAM,IAAI,GAAG,CAAC,IAAI,KAAK;CACpD,MAAM,WAAW,KAAK,QAAQ,KAAK,MAAM,OAAO,IAAI,SAAS,GAAG,CAAC,IAAI,KAAK;CAG1E,MAAM,KAFS,KAAK,KAAK,QAET,KADA,KAAK,IAAI,IAAI,IAAI,OAAO,KAAK,IAAI,IAAI,IAAI;CAEzD,MAAM,SAAS,KAAK,UAAU,UAAU,MAAM;CAI9C,IAAI,UAAU;CACd,IAAI,YAAwB;CAC5B,KAAK,IAAI,IAAI,OAAO,SAAS,GAAG,IAAI,GAAG,KAAK;EAC1C,MAAM,QAAQ,OAAO,KAAM,OAAO,IAAI;EACtC,IAAI,UAAU,GAAG;EACjB,MAAM,MAAM,QAAQ,IAAI,IAAI;EAC5B,IAAI,cAAc,GAAG,YAAY;EACjC,IAAI,QAAQ,WAAW;EACvB,WAAW;CACb;CAEA,IAAI;CACJ,IAAI,QACF,QAAQ;MACH,IAAI,KAAK,IAAI,OAAO,KAAK,UAC9B,QAAQ,UAAU,IAAI,gBAAgB;MAEtC,QAAQ;CAGV,OAAO;EAAE;EAAO,YAAY;EAAM,UAAU;EAAI;EAAS;CAAO;AAClE"}
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { s as JudgeScore } from "./types-
|
|
1
|
+
import { s as JudgeScore } from "./types-DOhGq4S1.js";
|
|
2
2
|
import { i as ContinuousAgreementOptions, r as ContinuousAgreement } from "./judge-calibration-C5CbMYce.js";
|
|
3
3
|
//#region src/statistics/agreement-irr.d.ts
|
|
4
4
|
/**
|
|
@@ -95,11 +95,8 @@ declare function corpusInterRaterAgreementFromJudgeScores(itemsScores: Array<{
|
|
|
95
95
|
//#region src/series-convergence.d.ts
|
|
96
96
|
/**
|
|
97
97
|
* Series convergence — detects whether a sequence of scalar measurements
|
|
98
|
-
* is stabilizing, drifting, or noisy.
|
|
99
|
-
*
|
|
100
|
-
* Lifted from ADC convergence.ts. The per-turn `ConvergenceTracker` is
|
|
101
|
-
* about progress *within* a single run; this module is about drift
|
|
102
|
-
* *across* runs (e.g. "are my nightly eval scores stabilizing?").
|
|
98
|
+
* is stabilizing, drifting, or noisy. It reads drift *across* runs, e.g.
|
|
99
|
+
* "are my nightly eval scores stabilizing?".
|
|
103
100
|
*
|
|
104
101
|
* Three signals:
|
|
105
102
|
* - stabilized: last K values have low variance (< epsilon) — done
|
|
@@ -126,4 +123,4 @@ interface SeriesConvergenceResult {
|
|
|
126
123
|
declare function analyzeSeries(values: number[], options?: SeriesConvergenceOptions): SeriesConvergenceResult;
|
|
127
124
|
//#endregion
|
|
128
125
|
export { CorpusAgreementPerDimension as a, corpusInterRaterAgreement as c, CorpusAgreementOptions as i, corpusInterRaterAgreementFromJudgeScores as l, SeriesConvergenceResult as n, CorpusAgreementReport as o, analyzeSeries as r, CorpusScoreRecord as s, SeriesConvergenceOptions as t, interRaterReliability as u };
|
|
129
|
-
//# sourceMappingURL=series-convergence-
|
|
126
|
+
//# sourceMappingURL=series-convergence-D2fsoJ2w.d.ts.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"series-convergence-
|
|
1
|
+
{"version":3,"file":"series-convergence-D2fsoJ2w.d.ts","names":[],"sources":["../src/statistics/agreement-irr.ts","../src/series-convergence.ts"],"mappings":";;;;;;;;;;;;;;;;;;;;iBAyBgB,sBAAsB,aAAa;UA0ElC;;EAEf;;EAEA;;EAEA;;EAEA;;UAGe,oCAAoC;EACnD;;EAEA;;EAEA;;UAGe;;EAEf,cAAc;;EAEd;;EAEA;;EAEA;;EAEA;;UAGe,+BAA+B;;;;;;EAM9C;;;;;;EAMA;;;;;;;;;;;;;;;;;;iBAmBc,0BACd,SAAS,qBACT,OAAM,yBACL;;;;;;;;;iBA0Ha,yCACd,aAAa;EAAQ;EAAgB,QAAQ;IAC7C,OAAM,yBACL;;;;;;;;;;;;;UCvRc;;EAEf;;EAEA;;EAEA;;UAGe;EACf;EACA;EACA;;EAEA;;EAEA;;iBAGc,cACd,kBACA,UAAS,2BACR"}
|
|
@@ -1,6 +1,6 @@
|
|
|
1
|
-
import { i as CostLedger } from "./cost-ledger-
|
|
1
|
+
import { i as CostLedger } from "./cost-ledger-B1qx30B4.js";
|
|
2
2
|
import { t as packageVersion } from "./package-version-D7lQHt_-.js";
|
|
3
|
-
import { a as assertLlmRoute, c as callLlmJson, f as maximumChargeForLlmRequest, i as LlmRouteAssertionError, l as costReceiptFromLlm, u as costReceiptFromLlmError } from "./llm-client-
|
|
3
|
+
import { a as assertLlmRoute, c as callLlmJson, f as maximumChargeForLlmRequest, i as LlmRouteAssertionError, l as costReceiptFromLlm, u as costReceiptFromLlmError } from "./llm-client-BMuxYoZy.js";
|
|
4
4
|
import { z } from "zod";
|
|
5
5
|
import { OpenAPIRegistry, OpenApiGeneratorV31, extendZodWithOpenApi } from "@asteasolutions/zod-to-openapi";
|
|
6
6
|
import { serve } from "@hono/node-server";
|
|
@@ -1021,4 +1021,4 @@ function startServerAsync(opts = {}) {
|
|
|
1021
1021
|
//#endregion
|
|
1022
1022
|
export { TraceEventSchema as A, HealthResponseSchema as C, RubricDimensionSchema as D, ListRubricsResponseSchema as E, hashRubric as F, TracesIngestResponseSchema as M, VersionResponseSchema as N, RubricInfoSchema as O, WIRE_VERSION as P, FeedbackTrajectorySchema as S, JudgeResultSchema as T, ErrorResponseSchema as _, runRpcBatch as a, FeedbackIngestResponseSchema as b, WireError as c, handleListRubrics as d, handleTracesIngest as f, listBuiltinRubrics as g, getBuiltinRubric as h, dispatchRpc as i, TracesIngestRequestSchema as j, RubricSchema as k, handleFeedbackIngest as l, BUILTIN_RUBRICS as m, startServer as n, runRpcOnce as o, handleVersion as p, startServerAsync as r, buildOpenApi as s, createApp as t, handleJudge as u, FailureModeSchema as v, JudgeRequestSchema as w, FeedbackLabelSchema as x, FeedbackAttemptSchema as y };
|
|
1023
1023
|
|
|
1024
|
-
//# sourceMappingURL=server-
|
|
1024
|
+
//# sourceMappingURL=server-D9wQclzG.js.map
|