@tangle-network/agent-eval 0.172.1 → 0.173.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +19 -0
- package/README.md +17 -2
- package/dist/adapters/http.d.ts +2 -2
- package/dist/analyst/index.d.ts +13 -13
- package/dist/analyst/index.js +7 -6
- package/dist/analyst/index.js.map +1 -1
- package/dist/{attestation-CJBGmMVh.d.ts → attestation-c1QvaBdX.d.ts} +2 -2
- package/dist/{attestation-CJBGmMVh.d.ts.map → attestation-c1QvaBdX.d.ts.map} +1 -1
- package/dist/{backend-integrity-e79K3UPD.d.ts → backend-integrity-CeuTgqsd.d.ts} +3 -4
- package/dist/backend-integrity-CeuTgqsd.d.ts.map +1 -0
- package/dist/{benchmark-h-h4bfqj.d.ts → benchmark-BjLGkfnN.d.ts} +3 -3
- package/dist/{benchmark-h-h4bfqj.d.ts.map → benchmark-BjLGkfnN.d.ts.map} +1 -1
- package/dist/{benchmark-command--qeZUHbu.js → benchmark-command-9S20PRel.js} +9 -9
- package/dist/{benchmark-command--qeZUHbu.js.map → benchmark-command-9S20PRel.js.map} +1 -1
- package/dist/benchmarks/index.d.ts +5 -5
- package/dist/benchmarks/index.js +3 -3
- package/dist/{bounded-process-VIi0KSL2.js → bounded-process-BBZob7vl.js} +30 -7
- package/dist/bounded-process-BBZob7vl.js.map +1 -0
- package/dist/builder-eval/index.d.ts +3 -3
- package/dist/builder-eval/index.js +2 -2
- package/dist/campaign/index.d.ts +9 -9
- package/dist/campaign/index.js +6 -6
- package/dist/{campaign-Dp35pBbS.js → campaign-pxS0wmo4.js} +8 -8
- package/dist/{campaign-Dp35pBbS.js.map → campaign-pxS0wmo4.js.map} +1 -1
- package/dist/chat-client-Db4bqYfA.js +115 -0
- package/dist/chat-client-Db4bqYfA.js.map +1 -0
- package/dist/{chat-json-call-5Jxna-aV.js → chat-json-call-C26igCih.js} +16 -6
- package/dist/chat-json-call-C26igCih.js.map +1 -0
- package/dist/cli.js +31 -17
- package/dist/cli.js.map +1 -1
- package/dist/{client-BvwNkIRN.js → client-BlLY6o2w.js} +2 -2
- package/dist/{client-BvwNkIRN.js.map → client-BlLY6o2w.js.map} +1 -1
- package/dist/{client-Df7wdslk.d.ts → client-DlqdbM7n.d.ts} +4 -4
- package/dist/{client-Df7wdslk.d.ts.map → client-DlqdbM7n.d.ts.map} +1 -1
- package/dist/contract/index.d.ts +13 -13
- package/dist/contract/index.js +202 -10
- package/dist/contract/index.js.map +1 -1
- package/dist/{counterfactual-Bee5_BIn.d.ts → counterfactual-CLgrwhkY.d.ts} +4 -4
- package/dist/{counterfactual-Bee5_BIn.d.ts.map → counterfactual-CLgrwhkY.d.ts.map} +1 -1
- package/dist/{chat-client-DI79OPye.js → default-registry-B0bKikCb.js} +3 -39
- package/dist/default-registry-B0bKikCb.js.map +1 -0
- package/dist/{default-registry-XxedTLwu.d.ts → default-registry-BKwc8bN5.d.ts} +6 -6
- package/dist/{default-registry-XxedTLwu.d.ts.map → default-registry-BKwc8bN5.d.ts.map} +1 -1
- package/dist/{define-agent-eval-0wW7gFhr.d.ts → define-agent-eval-CY6qdlGV.d.ts} +6 -6
- package/dist/{define-agent-eval-0wW7gFhr.d.ts.map → define-agent-eval-CY6qdlGV.d.ts.map} +1 -1
- package/dist/{define-agent-eval-jS8xj_Q_.js → define-agent-eval-D_i_s69h.js} +7 -7
- package/dist/{define-agent-eval-jS8xj_Q_.js.map → define-agent-eval-D_i_s69h.js.map} +1 -1
- package/dist/{dspy-rlm-engine-CS3qcCEk.js → dspy-rlm-engine-D5byiHn9.js} +5 -6
- package/dist/dspy-rlm-engine-D5byiHn9.js.map +1 -0
- package/dist/{emitter-Bvnu0VzL.d.ts → emitter-Cs0egaFd.d.ts} +3 -3
- package/dist/{emitter-Bvnu0VzL.d.ts.map → emitter-Cs0egaFd.d.ts.map} +1 -1
- package/dist/{engine-BfRay1qD.d.ts → engine-DhFir3Ys.d.ts} +23 -8
- package/dist/{engine-BfRay1qD.d.ts.map → engine-DhFir3Ys.d.ts.map} +1 -1
- package/dist/{eval-campaign-JDTeE6Pl.js → eval-campaign-BeAjdhzC.js} +2 -2
- package/dist/{eval-campaign-JDTeE6Pl.js.map → eval-campaign-BeAjdhzC.js.map} +1 -1
- package/dist/{exact-types-BEecmnWm.d.ts → exact-types-BKOEILRP.d.ts} +2 -2
- package/dist/{exact-types-BEecmnWm.d.ts.map → exact-types-BKOEILRP.d.ts.map} +1 -1
- package/dist/experiment/index.d.ts +4 -4
- package/dist/{external-optimizer-contracts-szBJ_1vh.d.ts → external-optimizer-contracts-CQCpyrIL.d.ts} +2 -2
- package/dist/{external-optimizer-contracts-szBJ_1vh.d.ts.map → external-optimizer-contracts-CQCpyrIL.d.ts.map} +1 -1
- package/dist/{external-optimizer-process-CQxylYeG.js → external-optimizer-process-Cq_Pg15r.js} +2 -2
- package/dist/{external-optimizer-process-CQxylYeG.js.map → external-optimizer-process-Cq_Pg15r.js.map} +1 -1
- package/dist/{external-optimizer-subprocess-Cex8Da2i.js → external-optimizer-subprocess-DgNebftP.js} +2 -2
- package/dist/{external-optimizer-subprocess-Cex8Da2i.js.map → external-optimizer-subprocess-DgNebftP.js.map} +1 -1
- package/dist/failure-cluster-OldNRoAt.d.ts +154 -0
- package/dist/failure-cluster-OldNRoAt.d.ts.map +1 -0
- package/dist/{feedback-trajectory-DIqpCyF0.d.ts → feedback-trajectory-CMnv_uYs.d.ts} +6 -6
- package/dist/{feedback-trajectory-DIqpCyF0.d.ts.map → feedback-trajectory-CMnv_uYs.d.ts.map} +1 -1
- package/dist/{heldout-gate-JgNRDZwZ.d.ts → heldout-gate-Df5hsqmm.d.ts} +7 -7
- package/dist/{heldout-gate-JgNRDZwZ.d.ts.map → heldout-gate-Df5hsqmm.d.ts.map} +1 -1
- package/dist/hosted/index.d.ts +2 -2
- package/dist/hosted/index.js +1 -1
- package/dist/{index-DnglhM0A.d.ts → index-BQqOjerE.d.ts} +94 -13
- package/dist/index-BQqOjerE.d.ts.map +1 -0
- package/dist/{index-_vPrVMRX.d.ts → index-CFDffsKz.d.ts} +11 -11
- package/dist/{index-_vPrVMRX.d.ts.map → index-CFDffsKz.d.ts.map} +1 -1
- package/dist/{index-DDAPhUJJ.d.ts → index-D0Db5X-4.d.ts} +45 -9
- package/dist/index-D0Db5X-4.d.ts.map +1 -0
- package/dist/{index-DMoxLG8P.d.ts → index-e7LXeRVa.d.ts} +3 -3
- package/dist/{index-DMoxLG8P.d.ts.map → index-e7LXeRVa.d.ts.map} +1 -1
- package/dist/index.d.ts +130 -42
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +64 -41
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-08F022xN.d.ts → insight-report-DETqPc_A.d.ts} +4 -4
- package/dist/{insight-report-08F022xN.d.ts.map → insight-report-DETqPc_A.d.ts.map} +1 -1
- package/dist/{integrity-B_EDELom.d.ts → integrity-BKTcA-HP.d.ts} +3 -3
- package/dist/{integrity-B_EDELom.d.ts.map → integrity-BKTcA-HP.d.ts.map} +1 -1
- package/dist/{kind-factory-DMeEoMQZ.js → kind-factory-gP6lDySe.js} +164 -149
- package/dist/kind-factory-gP6lDySe.js.map +1 -0
- package/dist/{llm-client-BFMRpmqb.js → llm-client-CxQtdtd6.js} +12 -5
- package/dist/llm-client-CxQtdtd6.js.map +1 -0
- package/dist/{llm-judge-aQHIk5_-.js → llm-judge-B2YxbAJb.js} +81 -7
- package/dist/{llm-judge-aQHIk5_-.js.map → llm-judge-B2YxbAJb.js.map} +1 -1
- package/dist/{matrix-Ch8JO1pG.d.ts → matrix-DGu8KhSs.d.ts} +2 -2
- package/dist/{matrix-Ch8JO1pG.d.ts.map → matrix-DGu8KhSs.d.ts.map} +1 -1
- package/dist/meta-eval/index.d.ts +100 -5
- package/dist/meta-eval/index.d.ts.map +1 -1
- package/dist/meta-eval/index.js +200 -2
- package/dist/meta-eval/index.js.map +1 -1
- package/dist/{mint-DjfDUMHr.js → mint-vWOdD8Ae.js} +2 -2
- package/dist/{mint-DjfDUMHr.js.map → mint-vWOdD8Ae.js.map} +1 -1
- package/dist/multishot/golden/index.d.ts +1 -1
- package/dist/multishot/index.d.ts +2 -2
- package/dist/openapi.json +1 -1
- package/dist/pipelines/index.d.ts +5 -5
- package/dist/pipelines/index.js +3 -3
- package/dist/{pre-registration-DHz6P_6f.d.ts → pre-registration-BoI4ucR3.d.ts} +2 -2
- package/dist/{pre-registration-DHz6P_6f.d.ts.map → pre-registration-BoI4ucR3.d.ts.map} +1 -1
- package/dist/{produced-state-CxmbFxFd.js → produced-state-7VYDwtkk.js} +3 -3
- package/dist/{produced-state-CxmbFxFd.js.map → produced-state-7VYDwtkk.js.map} +1 -1
- package/dist/{promotion-policy-BBBcz5_3.d.ts → promotion-policy-CvMda3kU.d.ts} +2 -2
- package/dist/{promotion-policy-BBBcz5_3.d.ts.map → promotion-policy-CvMda3kU.d.ts.map} +1 -1
- package/dist/{provenance-Dp-vvyrU.d.ts → provenance-CRY67X50.d.ts} +39 -165
- package/dist/provenance-CRY67X50.d.ts.map +1 -0
- package/dist/{query-BPGMVlbM.js → query-D1nLIKt7.js} +2 -2
- package/dist/{query-BPGMVlbM.js.map → query-D1nLIKt7.js.map} +1 -1
- package/dist/{query-Na5gEIGd.d.ts → query-D6W6MaGx.d.ts} +3 -3
- package/dist/{query-Na5gEIGd.d.ts.map → query-D6W6MaGx.d.ts.map} +1 -1
- package/dist/{registry-xEb_xfns.d.ts → registry-7pOUBrtX.d.ts} +4 -4
- package/dist/{registry-xEb_xfns.d.ts.map → registry-7pOUBrtX.d.ts.map} +1 -1
- package/dist/{release-confidence-D6lQw_o7.d.ts → release-confidence-BAcNYOf1.d.ts} +4 -4
- package/dist/{release-confidence-D6lQw_o7.d.ts.map → release-confidence-BAcNYOf1.d.ts.map} +1 -1
- package/dist/{release-confidence-CzUHc4z4.js → release-confidence-BsGEg_xg.js} +3 -3
- package/dist/{release-confidence-CzUHc4z4.js.map → release-confidence-BsGEg_xg.js.map} +1 -1
- package/dist/reporting.d.ts +3 -3
- package/dist/reporting.js +1 -1
- package/dist/{researcher-CMUTQXD7.d.ts → researcher-jsW1X94L.d.ts} +7 -8
- package/dist/researcher-jsW1X94L.d.ts.map +1 -0
- package/dist/{reward-hacking-SkxYgT0x.js → reward-hacking-CKW4teig.js} +2 -2
- package/dist/{reward-hacking-SkxYgT0x.js.map → reward-hacking-CKW4teig.js.map} +1 -1
- package/dist/{reward-hacking-CgPRUesA.d.ts → reward-hacking-ZXEi9VCq.d.ts} +2 -2
- package/dist/{reward-hacking-CgPRUesA.d.ts.map → reward-hacking-ZXEi9VCq.d.ts.map} +1 -1
- package/dist/rl.d.ts +7 -7
- package/dist/rl.js +4 -4
- package/dist/rollout/index.d.ts +1 -1
- package/dist/rollout/index.js +2 -2
- package/dist/{rollout-Crypdx8s.js → rollout-C-znbbYg.js} +2 -2
- package/dist/{rollout-Crypdx8s.js.map → rollout-C-znbbYg.js.map} +1 -1
- package/dist/{rubric-predictive-validity-DluJLCKQ.d.ts → rubric-predictive-validity-Dl1dvKCv.d.ts} +2 -2
- package/dist/{rubric-predictive-validity-DluJLCKQ.d.ts.map → rubric-predictive-validity-Dl1dvKCv.d.ts.map} +1 -1
- package/dist/{run-record-DQjRcYwA.d.ts → run-record-DTv1MdjK.d.ts} +2 -2
- package/dist/{run-record-DQjRcYwA.d.ts.map → run-record-DTv1MdjK.d.ts.map} +1 -1
- package/dist/{run-record-DLORoL7t.js → run-record-ZIsR9Fif.js} +2 -2
- package/dist/{run-record-DLORoL7t.js.map → run-record-ZIsR9Fif.js.map} +1 -1
- package/dist/{schema-DID1Cqct.d.ts → schema-CR5cpjQ3.d.ts} +57 -2
- package/dist/{schema-DID1Cqct.d.ts.map → schema-CR5cpjQ3.d.ts.map} +1 -1
- package/dist/{schema-CdIX2aHu.js → schema-CSf6qWgZ.js} +35 -1
- package/dist/{schema-CdIX2aHu.js.map → schema-CSf6qWgZ.js.map} +1 -1
- package/dist/semantic-concept-judge-Ct3QU7t5.js +780 -0
- package/dist/semantic-concept-judge-Ct3QU7t5.js.map +1 -0
- package/dist/{series-convergence-D9WgpXGi.d.ts → series-convergence-DeG33RpC.d.ts} +2 -2
- package/dist/{series-convergence-D9WgpXGi.d.ts.map → series-convergence-DeG33RpC.d.ts.map} +1 -1
- package/dist/{server-CCEnywOR.js → server-BR6onwZB.js} +2 -2
- package/dist/{server-CCEnywOR.js.map → server-BR6onwZB.js.map} +1 -1
- package/dist/{skillopt-optimization-method-LHi02MzH.js → skillopt-optimization-method-BzdphODy.js} +6 -6
- package/dist/{skillopt-optimization-method-LHi02MzH.js.map → skillopt-optimization-method-BzdphODy.js.map} +1 -1
- package/dist/{statistical-heldout-Yldkntvy.d.ts → statistical-heldout-DTyB_6-1.d.ts} +3 -3
- package/dist/{statistical-heldout-Yldkntvy.d.ts.map → statistical-heldout-DTyB_6-1.d.ts.map} +1 -1
- package/dist/{store-Cq9oOrI1.d.ts → store-BErPvYBr.d.ts} +2 -2
- package/dist/{store-Cq9oOrI1.d.ts.map → store-BErPvYBr.d.ts.map} +1 -1
- package/dist/{store-otlp-CHjBvWQY.js → store-otlp-Dow0pk_5.js} +2 -2
- package/dist/{store-otlp-CHjBvWQY.js.map → store-otlp-Dow0pk_5.js.map} +1 -1
- package/dist/{store-tool-spans-BvdUbeOB.d.ts → store-tool-spans-CCZNsihA.d.ts} +8 -8
- package/dist/{store-tool-spans-BvdUbeOB.d.ts.map → store-tool-spans-CCZNsihA.d.ts.map} +1 -1
- package/dist/{store-tool-spans-B9o6tU8f.js → store-tool-spans-CeNj_m2L.js} +3 -3
- package/dist/{store-tool-spans-B9o6tU8f.js.map → store-tool-spans-CeNj_m2L.js.map} +1 -1
- package/dist/storyboard/index.d.ts +1 -1
- package/dist/{summary-report-DRstQNBX.d.ts → summary-report-gMrbYawB.d.ts} +3 -3
- package/dist/{summary-report-DRstQNBX.d.ts.map → summary-report-gMrbYawB.d.ts.map} +1 -1
- package/dist/{task-failure-attributes-CBGtLS_H.js → task-failure-attributes-CZjZeBsY.js} +3 -3
- package/dist/{task-failure-attributes-CBGtLS_H.js.map → task-failure-attributes-CZjZeBsY.js.map} +1 -1
- package/dist/{tool-groups-DjwlMBvW.d.ts → tool-groups-Cp4Xdzrp.d.ts} +3 -3
- package/dist/tool-groups-Cp4Xdzrp.d.ts.map +1 -0
- package/dist/{tool-waste-CwGHzBzX.js → tool-waste-B9tdWV6g.js} +253 -6
- package/dist/tool-waste-B9tdWV6g.js.map +1 -0
- package/dist/{tool-waste-BrmLKxMw.d.ts → tool-waste-D23I0zWm.d.ts} +4 -4
- package/dist/{tool-waste-BrmLKxMw.d.ts.map → tool-waste-D23I0zWm.d.ts.map} +1 -1
- package/dist/trace-repair/index.d.ts +2 -2
- package/dist/traces.d.ts +10 -10
- package/dist/traces.js +7 -7
- package/dist/{trajectory-r1bQqvBQ.d.ts → trajectory-D7qrNvaN.d.ts} +3 -3
- package/dist/{trajectory-r1bQqvBQ.d.ts.map → trajectory-D7qrNvaN.d.ts.map} +1 -1
- package/dist/trajectory-replay/index.d.ts +3 -3
- package/dist/{types-CCZ34qmV.d.ts → types-BDV4PiMR.d.ts} +3 -3
- package/dist/{types-CCZ34qmV.d.ts.map → types-BDV4PiMR.d.ts.map} +1 -1
- package/dist/{types-nokrtr7M.d.ts → types-Ba5UQyVD.d.ts} +4 -4
- package/dist/{types-nokrtr7M.d.ts.map → types-Ba5UQyVD.d.ts.map} +1 -1
- package/dist/{types-DMoNFDWi.d.ts → types-DN2WdT5S.d.ts} +3 -3
- package/dist/{types-DMoNFDWi.d.ts.map → types-DN2WdT5S.d.ts.map} +1 -1
- package/dist/types-gvRsyJLh.d.ts +831 -0
- package/dist/types-gvRsyJLh.d.ts.map +1 -0
- package/dist/wire/index.d.ts +3 -3
- package/dist/wire/index.js +1 -1
- package/docs/plants.md +69 -1
- package/docs/public-api.md +6 -4
- package/docs/trace-analysis.md +55 -8
- package/package.json +1 -1
- package/dist/backend-integrity-e79K3UPD.d.ts.map +0 -1
- package/dist/bounded-process-VIi0KSL2.js.map +0 -1
- package/dist/chat-client-DI79OPye.js.map +0 -1
- package/dist/chat-json-call-5Jxna-aV.js.map +0 -1
- package/dist/dspy-rlm-engine-CS3qcCEk.js.map +0 -1
- package/dist/failure-cluster-6YSvsKlp.d.ts +0 -58
- package/dist/failure-cluster-6YSvsKlp.d.ts.map +0 -1
- package/dist/index-DDAPhUJJ.d.ts.map +0 -1
- package/dist/index-DnglhM0A.d.ts.map +0 -1
- package/dist/kind-factory-DMeEoMQZ.js.map +0 -1
- package/dist/llm-client-BFMRpmqb.js.map +0 -1
- package/dist/provenance-Dp-vvyrU.d.ts.map +0 -1
- package/dist/raw-provider-sink-BU29Sh8h.d.ts +0 -134
- package/dist/raw-provider-sink-BU29Sh8h.d.ts.map +0 -1
- package/dist/researcher-CMUTQXD7.d.ts.map +0 -1
- package/dist/semantic-concept-judge-I36eejJx.js +0 -382
- package/dist/semantic-concept-judge-I36eejJx.js.map +0 -1
- package/dist/tool-groups-DjwlMBvW.d.ts.map +0 -1
- package/dist/tool-waste-CwGHzBzX.js.map +0 -1
- package/dist/types-Bfk0uxRj.d.ts +0 -443
- package/dist/types-Bfk0uxRj.d.ts.map +0 -1
|
@@ -1,382 +0,0 @@
|
|
|
1
|
-
import { i as CostLedger } from "./cost-ledger-B1qx30B4.js";
|
|
2
|
-
import { l as Mutex } from "./ledger-core-PIfjCbKn.js";
|
|
3
|
-
import { t as paidJsonChat } from "./chat-json-call-5Jxna-aV.js";
|
|
4
|
-
import { appendFileSync, existsSync, mkdirSync, readFileSync } from "node:fs";
|
|
5
|
-
import { dirname } from "node:path";
|
|
6
|
-
//#region src/locked-jsonl-appender.ts
|
|
7
|
-
/**
|
|
8
|
-
* LockedJsonlAppender — mutex-serialized JSONL append helper for arbitrary
|
|
9
|
-
* payloads. The reference-replay store does the same thing for typed
|
|
10
|
-
* `ReferenceReplayRun` rows; this is the generic version used by
|
|
11
|
-
* `MutationTelemetry`, `TrialTelemetry`, and any other consumer that wants
|
|
12
|
-
* append-only durable telemetry without rolling its own lock.
|
|
13
|
-
*
|
|
14
|
-
* Locks are per absolute file path (process-local). Cross-process
|
|
15
|
-
* concurrency is NOT addressed — that's an fcntl/flock problem.
|
|
16
|
-
*/
|
|
17
|
-
const mutexes = /* @__PURE__ */ new Map();
|
|
18
|
-
function getMutex(path) {
|
|
19
|
-
let m = mutexes.get(path);
|
|
20
|
-
if (!m) {
|
|
21
|
-
m = new Mutex();
|
|
22
|
-
mutexes.set(path, m);
|
|
23
|
-
}
|
|
24
|
-
return m;
|
|
25
|
-
}
|
|
26
|
-
var LockedJsonlAppender = class {
|
|
27
|
-
path;
|
|
28
|
-
mutex;
|
|
29
|
-
constructor(path) {
|
|
30
|
-
this.path = path;
|
|
31
|
-
this.mutex = getMutex(path);
|
|
32
|
-
if (!existsSync(dirname(path))) mkdirSync(dirname(path), { recursive: true });
|
|
33
|
-
}
|
|
34
|
-
async append(entry) {
|
|
35
|
-
const line = `${JSON.stringify(entry)}\n`;
|
|
36
|
-
await this.mutex.runExclusive(() => {
|
|
37
|
-
appendFileSync(this.path, line);
|
|
38
|
-
});
|
|
39
|
-
}
|
|
40
|
-
};
|
|
41
|
-
//#endregion
|
|
42
|
-
//#region src/analyst/findings-store.ts
|
|
43
|
-
/**
|
|
44
|
-
* FindingsStore — durable persistence for AnalystFinding rows + a diff
|
|
45
|
-
* helper so we can answer "what changed since the last run?" without
|
|
46
|
-
* recomputing analysts.
|
|
47
|
-
*
|
|
48
|
-
* On-disk shape is JSONL: one finding per line, append-only, locked via
|
|
49
|
-
* LockedJsonlAppender. Operators get crash-safety (no partial JSON),
|
|
50
|
-
* cheap reads (sequential parse), and trivial backup (rsync the file).
|
|
51
|
-
*
|
|
52
|
-
* Reads are non-locking: a reader sees a consistent snapshot of all
|
|
53
|
-
* fully-written lines and skips an incomplete trailing line if the
|
|
54
|
-
* writer is mid-append. Cross-process locking is intentionally out of
|
|
55
|
-
* scope (see locked-jsonl-appender.ts).
|
|
56
|
-
*
|
|
57
|
-
* The store is run-scoped: callers pass `runId` on append and on load,
|
|
58
|
-
* which keeps multi-run files cleanly partitioned. The `diffFindings`
|
|
59
|
-
* helper compares two run-id sets using stable `finding_id` semantics —
|
|
60
|
-
* the diff is the cross-run signal the regression dashboard renders.
|
|
61
|
-
*/
|
|
62
|
-
var FindingsStore = class {
|
|
63
|
-
path;
|
|
64
|
-
appender;
|
|
65
|
-
constructor(path) {
|
|
66
|
-
this.path = path;
|
|
67
|
-
this.appender = new LockedJsonlAppender(path);
|
|
68
|
-
}
|
|
69
|
-
async append(runId, findings) {
|
|
70
|
-
for (const f of findings) {
|
|
71
|
-
const row = {
|
|
72
|
-
...f,
|
|
73
|
-
run_id: runId
|
|
74
|
-
};
|
|
75
|
-
await this.appender.append(row);
|
|
76
|
-
}
|
|
77
|
-
}
|
|
78
|
-
/** Load every persisted finding. Discards malformed trailing lines silently. */
|
|
79
|
-
loadAll() {
|
|
80
|
-
if (!existsSync(this.path)) return [];
|
|
81
|
-
const raw = readFileSync(this.path, "utf8");
|
|
82
|
-
if (!raw) return [];
|
|
83
|
-
const out = [];
|
|
84
|
-
for (const line of raw.split("\n")) {
|
|
85
|
-
if (!line) continue;
|
|
86
|
-
try {
|
|
87
|
-
out.push(JSON.parse(line));
|
|
88
|
-
} catch {}
|
|
89
|
-
}
|
|
90
|
-
return out;
|
|
91
|
-
}
|
|
92
|
-
/** Filter to a single run. */
|
|
93
|
-
loadRun(runId) {
|
|
94
|
-
return this.loadAll().filter((r) => r.run_id === runId);
|
|
95
|
-
}
|
|
96
|
-
};
|
|
97
|
-
/**
|
|
98
|
-
* Default materiality test. Deliberately narrow so LLM-reword churn
|
|
99
|
-
* doesn't flood the diff. Stricter tests are opt-in via DiffPolicy.
|
|
100
|
-
*/
|
|
101
|
-
function defaultIsMaterial(a, b) {
|
|
102
|
-
if (a.severity !== b.severity) return true;
|
|
103
|
-
if (Math.abs((a.confidence ?? 0) - (b.confidence ?? 0)) > .05) return true;
|
|
104
|
-
if (a.evidence_refs.length !== b.evidence_refs.length) return true;
|
|
105
|
-
return false;
|
|
106
|
-
}
|
|
107
|
-
/**
|
|
108
|
-
* Diff two findings sets by stable finding_id. Callers typically load
|
|
109
|
-
* the two run-id slices from the same store and pass them in.
|
|
110
|
-
*/
|
|
111
|
-
function diffFindings(previous, current, policy = {}) {
|
|
112
|
-
const isMaterial = policy.isMaterial ?? defaultIsMaterial;
|
|
113
|
-
const prevById = new Map(previous.map((f) => [f.finding_id, f]));
|
|
114
|
-
const curById = new Map(current.map((f) => [f.finding_id, f]));
|
|
115
|
-
const appeared = [];
|
|
116
|
-
const disappeared = [];
|
|
117
|
-
const persisted = [];
|
|
118
|
-
const changed = [];
|
|
119
|
-
for (const [id, cur] of curById) {
|
|
120
|
-
const prev = prevById.get(id);
|
|
121
|
-
if (!prev) {
|
|
122
|
-
appeared.push(cur);
|
|
123
|
-
continue;
|
|
124
|
-
}
|
|
125
|
-
if (isMaterial(prev, cur)) changed.push({
|
|
126
|
-
previous: prev,
|
|
127
|
-
current: cur
|
|
128
|
-
});
|
|
129
|
-
else persisted.push(cur);
|
|
130
|
-
}
|
|
131
|
-
for (const [id, prev] of prevById) if (!curById.has(id)) disappeared.push(prev);
|
|
132
|
-
return {
|
|
133
|
-
appeared,
|
|
134
|
-
disappeared,
|
|
135
|
-
persisted,
|
|
136
|
-
changed
|
|
137
|
-
};
|
|
138
|
-
}
|
|
139
|
-
//#endregion
|
|
140
|
-
//#region src/semantic-concept-judge.ts
|
|
141
|
-
const DEFAULT_COMPLEXITY_WEIGHTS = {
|
|
142
|
-
render: 1,
|
|
143
|
-
integrate: 2,
|
|
144
|
-
compute: 2.5
|
|
145
|
-
};
|
|
146
|
-
const SEMANTIC_CONCEPT_JUDGE_VERSION = "semantic-concept-judge-v1-2026-04-24";
|
|
147
|
-
const DEFAULT_MAX_SOURCE = 45e3;
|
|
148
|
-
const DEFAULT_MAX_HTML = 3e4;
|
|
149
|
-
const DEFAULT_MAX_PER_FILE = 2e4;
|
|
150
|
-
const DEFAULT_TIMEOUT = 3e5;
|
|
151
|
-
const DEFAULT_MAX_TOKENS = 16e3;
|
|
152
|
-
const DEFAULT_MODEL = "claude-sonnet-4-6";
|
|
153
|
-
const SEMANTIC_SCHEMA = {
|
|
154
|
-
type: "object",
|
|
155
|
-
additionalProperties: false,
|
|
156
|
-
required: ["summary", "concepts"],
|
|
157
|
-
properties: {
|
|
158
|
-
summary: {
|
|
159
|
-
type: "string",
|
|
160
|
-
minLength: 20,
|
|
161
|
-
maxLength: 600
|
|
162
|
-
},
|
|
163
|
-
concepts: {
|
|
164
|
-
type: "array",
|
|
165
|
-
minItems: 1,
|
|
166
|
-
items: {
|
|
167
|
-
type: "object",
|
|
168
|
-
additionalProperties: false,
|
|
169
|
-
required: [
|
|
170
|
-
"concept",
|
|
171
|
-
"present",
|
|
172
|
-
"score",
|
|
173
|
-
"evidence",
|
|
174
|
-
"severity"
|
|
175
|
-
],
|
|
176
|
-
properties: {
|
|
177
|
-
concept: {
|
|
178
|
-
type: "string",
|
|
179
|
-
minLength: 1,
|
|
180
|
-
maxLength: 120
|
|
181
|
-
},
|
|
182
|
-
present: { type: "boolean" },
|
|
183
|
-
score: {
|
|
184
|
-
type: "number",
|
|
185
|
-
minimum: 0,
|
|
186
|
-
maximum: 10
|
|
187
|
-
},
|
|
188
|
-
evidence: {
|
|
189
|
-
type: "string",
|
|
190
|
-
minLength: 5,
|
|
191
|
-
maxLength: 400
|
|
192
|
-
},
|
|
193
|
-
severity: {
|
|
194
|
-
type: "string",
|
|
195
|
-
enum: [
|
|
196
|
-
"critical",
|
|
197
|
-
"major",
|
|
198
|
-
"minor",
|
|
199
|
-
"info"
|
|
200
|
-
]
|
|
201
|
-
}
|
|
202
|
-
}
|
|
203
|
-
}
|
|
204
|
-
}
|
|
205
|
-
}
|
|
206
|
-
};
|
|
207
|
-
function truncate(body, cap, label) {
|
|
208
|
-
if (body.length <= cap) return body;
|
|
209
|
-
return `${body.slice(0, cap)}\n… [truncated ${body.length - cap} chars of ${label}]`;
|
|
210
|
-
}
|
|
211
|
-
function buildPrompt(input, opts) {
|
|
212
|
-
const sourceBlob = input.sourceFiles.filter((f) => f.content.length <= opts.maxPerFileChars).map((f) => `--- FILE: ${f.path} ---\n${f.content}`).join("\n\n");
|
|
213
|
-
const html = input.servedHtml ?? "";
|
|
214
|
-
return `You are a strict code-review judge evaluating whether an agent's 0-to-1 build actually implements the features the user asked for.
|
|
215
|
-
|
|
216
|
-
You MUST distinguish:
|
|
217
|
-
(a) WORKING code that implements the concept (rendered UI, wired handler, real API call),
|
|
218
|
-
(b) KEYWORD-PRESENT stub (comments mentioning the concept, variable names, TODOs),
|
|
219
|
-
(c) ABSENT (concept nowhere).
|
|
220
|
-
|
|
221
|
-
A comment like "// TODO: add mint button" is NOT present — score 2-3. Only count a concept as present if there is real functional code: a rendered component, a call handler wired to state or a network call, a computed value actually used.
|
|
222
|
-
|
|
223
|
-
USER REQUEST (what the agent was asked to build):
|
|
224
|
-
${input.userRequest}
|
|
225
|
-
|
|
226
|
-
${input.artifactLabel ? `ARTIFACT METADATA:\n name: ${input.artifactLabel}\n description: ${input.artifactDescription ?? ""}\n\n` : ""}EXPECTED CONCEPTS (each must be graded independently):
|
|
227
|
-
${input.expectedConcepts.map((c, i) => ` ${i + 1}. "${c.name}"${c.keywords?.length ? ` — hints: [${c.keywords.slice(0, 6).join(" | ")}]` : ""}`).join("\n")}
|
|
228
|
-
|
|
229
|
-
${html ? `SERVED HTML (what the preview returns when hit):\n${truncate(html, opts.maxHtmlChars, "HTML")}\n\n` : ""}SOURCE FILES (the agent's workdir):
|
|
230
|
-
${truncate(sourceBlob, opts.maxSourceChars, "source")}
|
|
231
|
-
|
|
232
|
-
For EACH concept, return:
|
|
233
|
-
- concept: the concept name as given (match exactly)
|
|
234
|
-
- present: boolean — does a working implementation exist?
|
|
235
|
-
- score: 0-10 — 10 = production-ready; 7 = functional but thin; 4 = partial/stubbed; 2 = keyword-only comment; 0 = absent
|
|
236
|
-
- evidence: cite "<file>:<line>" or "served-html:<selector>" pointing at the strongest supporting code. If the concept is absent or stubbed, explain what's missing.
|
|
237
|
-
- severity:
|
|
238
|
-
"info" when present: true AND score >= 7
|
|
239
|
-
"minor" when present: true AND 4 <= score < 7
|
|
240
|
-
"major" when present: false OR score < 4
|
|
241
|
-
"critical" when the concept is not only absent but a core user flow depends on it
|
|
242
|
-
|
|
243
|
-
Also produce a "summary" (one sentence, 20-600 chars): overall verdict on whether this is a shippable implementation of the user request vs a keyword-dense placeholder.
|
|
244
|
-
|
|
245
|
-
BE SKEPTICAL. Keyword matching already passed — your job is to catch what keyword matching misses. If the agent shipped a working build, say so. If it shipped a stub, say so. Don't grade on effort.
|
|
246
|
-
|
|
247
|
-
Return STRICT JSON. No prose outside the JSON.`;
|
|
248
|
-
}
|
|
249
|
-
/**
|
|
250
|
-
* Run the semantic concept judge. Soft-fails to available=false on
|
|
251
|
-
* LLM/JSON errors — callers in a MultiLayerVerifier pipeline can treat
|
|
252
|
-
* that as "skip" rather than "fail."
|
|
253
|
-
*/
|
|
254
|
-
async function runSemanticConceptJudge(input, options) {
|
|
255
|
-
const start = Date.now();
|
|
256
|
-
const totalCount = input.expectedConcepts.length;
|
|
257
|
-
if (totalCount === 0) return {
|
|
258
|
-
kind: "semantic-concept",
|
|
259
|
-
version: SEMANTIC_CONCEPT_JUDGE_VERSION,
|
|
260
|
-
score: 0,
|
|
261
|
-
presentCount: 0,
|
|
262
|
-
totalCount: 0,
|
|
263
|
-
findings: [],
|
|
264
|
-
summary: "no expected concepts declared",
|
|
265
|
-
durationMs: 0,
|
|
266
|
-
costUsd: null,
|
|
267
|
-
available: false,
|
|
268
|
-
error: "no expected concepts declared"
|
|
269
|
-
};
|
|
270
|
-
const opts = {
|
|
271
|
-
chat: options.chat,
|
|
272
|
-
model: options.model ?? options.chat.defaultModel ?? DEFAULT_MODEL,
|
|
273
|
-
timeoutMs: options.timeoutMs ?? DEFAULT_TIMEOUT,
|
|
274
|
-
maxTokens: options.maxTokens ?? DEFAULT_MAX_TOKENS,
|
|
275
|
-
maxSourceChars: options.maxSourceChars ?? DEFAULT_MAX_SOURCE,
|
|
276
|
-
maxPerFileChars: options.maxPerFileChars ?? DEFAULT_MAX_PER_FILE,
|
|
277
|
-
maxHtmlChars: options.maxHtmlChars ?? DEFAULT_MAX_HTML,
|
|
278
|
-
...options.pricing ? { pricing: options.pricing } : {},
|
|
279
|
-
costLedger: options.costLedger ?? new CostLedger(),
|
|
280
|
-
costPhase: options.costPhase ?? "judge.semantic-concept",
|
|
281
|
-
costTags: options.costTags ?? {},
|
|
282
|
-
signal: options.signal ?? new AbortController().signal,
|
|
283
|
-
weightConcepts: options.weightConcepts ?? "mean",
|
|
284
|
-
complexityWeights: {
|
|
285
|
-
...DEFAULT_COMPLEXITY_WEIGHTS,
|
|
286
|
-
...options.complexityWeights ?? {}
|
|
287
|
-
}
|
|
288
|
-
};
|
|
289
|
-
const weightForConcept = (spec) => {
|
|
290
|
-
if (opts.weightConcepts === "mean") return 1;
|
|
291
|
-
if (spec.weight != null) return spec.weight;
|
|
292
|
-
if (opts.weightConcepts === "complexity") return opts.complexityWeights[spec.complexity ?? "render"] ?? 1;
|
|
293
|
-
return 1;
|
|
294
|
-
};
|
|
295
|
-
const weightByName = new Map(input.expectedConcepts.map((c) => [c.name, weightForConcept(c)]));
|
|
296
|
-
let receipt;
|
|
297
|
-
try {
|
|
298
|
-
const request = {
|
|
299
|
-
model: opts.model,
|
|
300
|
-
messages: [{
|
|
301
|
-
role: "system",
|
|
302
|
-
content: "You are a strict code-review judge. Return strict JSON only. No prose outside the JSON. A keyword in a comment is NOT a working implementation."
|
|
303
|
-
}, {
|
|
304
|
-
role: "user",
|
|
305
|
-
content: buildPrompt(input, opts)
|
|
306
|
-
}],
|
|
307
|
-
jsonSchema: {
|
|
308
|
-
name: "semantic_concept_judge",
|
|
309
|
-
schema: SEMANTIC_SCHEMA
|
|
310
|
-
},
|
|
311
|
-
temperature: 0,
|
|
312
|
-
maxTokens: opts.maxTokens,
|
|
313
|
-
timeoutMs: opts.timeoutMs
|
|
314
|
-
};
|
|
315
|
-
const paid = await paidJsonChat({
|
|
316
|
-
chat: opts.chat,
|
|
317
|
-
request,
|
|
318
|
-
ledger: opts.costLedger,
|
|
319
|
-
channel: "judge",
|
|
320
|
-
phase: opts.costPhase,
|
|
321
|
-
actor: "semantic-concept",
|
|
322
|
-
tags: opts.costTags,
|
|
323
|
-
signal: opts.signal,
|
|
324
|
-
...opts.pricing ? { pricing: opts.pricing } : {}
|
|
325
|
-
});
|
|
326
|
-
receipt = paid.receipt;
|
|
327
|
-
if (!paid.succeeded) throw paid.error;
|
|
328
|
-
const { value } = paid;
|
|
329
|
-
if (!value?.concepts || !Array.isArray(value.concepts)) throw new Error("judge returned malformed response — expected array under \"concepts\"");
|
|
330
|
-
const findings = value.concepts.map((c) => ({
|
|
331
|
-
concept: String(c.concept),
|
|
332
|
-
present: Boolean(c.present),
|
|
333
|
-
score: Math.max(0, Math.min(10, Number(c.score ?? 0))),
|
|
334
|
-
evidence: String(c.evidence ?? ""),
|
|
335
|
-
severity: [
|
|
336
|
-
"critical",
|
|
337
|
-
"major",
|
|
338
|
-
"minor",
|
|
339
|
-
"info"
|
|
340
|
-
].includes(c.severity) ? c.severity : "info"
|
|
341
|
-
}));
|
|
342
|
-
const presentCount = findings.filter((f) => f.present && f.score >= 7).length;
|
|
343
|
-
let weightSum = 0;
|
|
344
|
-
let weightedScoreSum = 0;
|
|
345
|
-
for (const f of findings) {
|
|
346
|
-
const w = weightByName.get(f.concept) ?? 1;
|
|
347
|
-
weightSum += w;
|
|
348
|
-
weightedScoreSum += w * f.score;
|
|
349
|
-
}
|
|
350
|
-
const scoreAvg = weightSum > 0 ? weightedScoreSum / weightSum : findings.reduce((a, f) => a + f.score, 0) / Math.max(1, findings.length);
|
|
351
|
-
return {
|
|
352
|
-
kind: "semantic-concept",
|
|
353
|
-
version: SEMANTIC_CONCEPT_JUDGE_VERSION,
|
|
354
|
-
score: Number((scoreAvg / 10).toFixed(3)),
|
|
355
|
-
presentCount,
|
|
356
|
-
totalCount,
|
|
357
|
-
findings,
|
|
358
|
-
summary: String(value.summary ?? ""),
|
|
359
|
-
durationMs: Date.now() - start,
|
|
360
|
-
costUsd: paid.receipt.costUnknown ? null : paid.receipt.costUsd,
|
|
361
|
-
available: true
|
|
362
|
-
};
|
|
363
|
-
} catch (err) {
|
|
364
|
-
return {
|
|
365
|
-
kind: "semantic-concept",
|
|
366
|
-
version: SEMANTIC_CONCEPT_JUDGE_VERSION,
|
|
367
|
-
score: 0,
|
|
368
|
-
presentCount: 0,
|
|
369
|
-
totalCount,
|
|
370
|
-
findings: [],
|
|
371
|
-
summary: "",
|
|
372
|
-
durationMs: Date.now() - start,
|
|
373
|
-
costUsd: receipt && !receipt.costUnknown ? receipt.costUsd : null,
|
|
374
|
-
available: false,
|
|
375
|
-
error: err instanceof Error ? err.message : String(err)
|
|
376
|
-
};
|
|
377
|
-
}
|
|
378
|
-
}
|
|
379
|
-
//#endregion
|
|
380
|
-
export { diffFindings as a, defaultIsMaterial as i, runSemanticConceptJudge as n, FindingsStore as r, SEMANTIC_CONCEPT_JUDGE_VERSION as t };
|
|
381
|
-
|
|
382
|
-
//# sourceMappingURL=semantic-concept-judge-I36eejJx.js.map
|
|
@@ -1 +0,0 @@
|
|
|
1
|
-
{"version":3,"file":"semantic-concept-judge-I36eejJx.js","names":[],"sources":["../src/locked-jsonl-appender.ts","../src/analyst/findings-store.ts","../src/semantic-concept-judge.ts"],"sourcesContent":["/**\n * LockedJsonlAppender — mutex-serialized JSONL append helper for arbitrary\n * payloads. The reference-replay store does the same thing for typed\n * `ReferenceReplayRun` rows; this is the generic version used by\n * `MutationTelemetry`, `TrialTelemetry`, and any other consumer that wants\n * append-only durable telemetry without rolling its own lock.\n *\n * Locks are per absolute file path (process-local). Cross-process\n * concurrency is NOT addressed — that's an fcntl/flock problem.\n */\n\nimport { appendFileSync, existsSync, mkdirSync } from 'node:fs'\nimport { dirname } from 'node:path'\nimport { Mutex } from './concurrency'\n\nconst mutexes = new Map<string, Mutex>()\n\nfunction getMutex(path: string): Mutex {\n let m = mutexes.get(path)\n if (!m) {\n m = new Mutex()\n mutexes.set(path, m)\n }\n return m\n}\n\nexport class LockedJsonlAppender {\n private readonly mutex: Mutex\n constructor(public readonly path: string) {\n this.mutex = getMutex(path)\n if (!existsSync(dirname(path))) {\n mkdirSync(dirname(path), { recursive: true })\n }\n }\n\n async append(entry: unknown): Promise<void> {\n const line = `${JSON.stringify(entry)}\\n`\n await this.mutex.runExclusive(() => {\n appendFileSync(this.path, line)\n })\n }\n}\n\n/** Reset all internal mutex state — tests only. */\nexport function resetLockedAppendersForTesting(): void {\n mutexes.clear()\n}\n","/**\n * FindingsStore — durable persistence for AnalystFinding rows + a diff\n * helper so we can answer \"what changed since the last run?\" without\n * recomputing analysts.\n *\n * On-disk shape is JSONL: one finding per line, append-only, locked via\n * LockedJsonlAppender. Operators get crash-safety (no partial JSON),\n * cheap reads (sequential parse), and trivial backup (rsync the file).\n *\n * Reads are non-locking: a reader sees a consistent snapshot of all\n * fully-written lines and skips an incomplete trailing line if the\n * writer is mid-append. Cross-process locking is intentionally out of\n * scope (see locked-jsonl-appender.ts).\n *\n * The store is run-scoped: callers pass `runId` on append and on load,\n * which keeps multi-run files cleanly partitioned. The `diffFindings`\n * helper compares two run-id sets using stable `finding_id` semantics —\n * the diff is the cross-run signal the regression dashboard renders.\n */\n\nimport { existsSync, readFileSync } from 'node:fs'\n\nimport { LockedJsonlAppender } from '../locked-jsonl-appender'\nimport type { AnalystFinding } from './types'\n\n/**\n * One persisted row. We attach `run_id` on disk so a single file can\n * hold multiple runs and the diff helper can query without re-walking\n * separate files.\n */\nexport interface PersistedFinding extends AnalystFinding {\n run_id: string\n}\n\nexport class FindingsStore {\n private readonly appender: LockedJsonlAppender\n\n constructor(public readonly path: string) {\n this.appender = new LockedJsonlAppender(path)\n }\n\n async append(runId: string, findings: AnalystFinding[]): Promise<void> {\n for (const f of findings) {\n const row: PersistedFinding = { ...f, run_id: runId }\n await this.appender.append(row)\n }\n }\n\n /** Load every persisted finding. Discards malformed trailing lines silently. */\n loadAll(): PersistedFinding[] {\n if (!existsSync(this.path)) return []\n const raw = readFileSync(this.path, 'utf8')\n if (!raw) return []\n const out: PersistedFinding[] = []\n for (const line of raw.split('\\n')) {\n if (!line) continue\n try {\n out.push(JSON.parse(line) as PersistedFinding)\n } catch {\n // Skip torn trailing line — the lock guarantees no torn lines\n // mid-file, only at EOF when a writer is in-flight.\n }\n }\n return out\n }\n\n /** Filter to a single run. */\n loadRun(runId: string): PersistedFinding[] {\n return this.loadAll().filter((r) => r.run_id === runId)\n }\n}\n\n// ── Cross-run diff ──────────────────────────────────────────────────\n\nexport interface FindingsDiff {\n /** New finding ids in `current` that weren't in `previous`. */\n appeared: PersistedFinding[]\n /** Finding ids in `previous` that aren't in `current`. */\n disappeared: PersistedFinding[]\n /** Same finding id present in both runs and unchanged per the materiality test. */\n persisted: PersistedFinding[]\n /**\n * Same finding id in both runs but at least one non-identity field\n * shifted per `DiffPolicy.isMaterial`. Reported as [previous, current].\n */\n changed: Array<{ previous: PersistedFinding; current: PersistedFinding }>\n}\n\nexport interface DiffPolicy {\n /**\n * Predicate that decides whether two findings (same finding_id) count\n * as a material change. Defaults to {@link defaultIsMaterial}: severity\n * shift, confidence Δ > 0.05, or evidence count change. Compliance /\n * perf consumers MAY supply a stricter predicate (e.g. rationale text\n * diff, metric Δ thresholds).\n */\n isMaterial?: (previous: AnalystFinding, current: AnalystFinding) => boolean\n}\n\n/**\n * Default materiality test. Deliberately narrow so LLM-reword churn\n * doesn't flood the diff. Stricter tests are opt-in via DiffPolicy.\n */\nexport function defaultIsMaterial(a: AnalystFinding, b: AnalystFinding): boolean {\n if (a.severity !== b.severity) return true\n if (Math.abs((a.confidence ?? 0) - (b.confidence ?? 0)) > 0.05) return true\n if (a.evidence_refs.length !== b.evidence_refs.length) return true\n return false\n}\n\n/**\n * Diff two findings sets by stable finding_id. Callers typically load\n * the two run-id slices from the same store and pass them in.\n */\nexport function diffFindings(\n previous: PersistedFinding[],\n current: PersistedFinding[],\n policy: DiffPolicy = {},\n): FindingsDiff {\n const isMaterial = policy.isMaterial ?? defaultIsMaterial\n const prevById = new Map(previous.map((f) => [f.finding_id, f]))\n const curById = new Map(current.map((f) => [f.finding_id, f]))\n\n const appeared: PersistedFinding[] = []\n const disappeared: PersistedFinding[] = []\n const persisted: PersistedFinding[] = []\n const changed: FindingsDiff['changed'] = []\n\n for (const [id, cur] of curById) {\n const prev = prevById.get(id)\n if (!prev) {\n appeared.push(cur)\n continue\n }\n if (isMaterial(prev, cur)) {\n changed.push({ previous: prev, current: cur })\n } else {\n persisted.push(cur)\n }\n }\n for (const [id, prev] of prevById) {\n if (!curById.has(id)) disappeared.push(prev)\n }\n return { appeared, disappeared, persisted, changed }\n}\n","/**\n * Semantic concept judge — \"does the built artifact actually implement\n * the features the user asked for?\"\n *\n * Distinct from the domain/code/coherence judges in `judges.ts`:\n * - those judges score free-form conversational agent outputs along\n * quality dimensions (accuracy, depth, etc.)\n * - this judge scores a *built artifact* (served HTML + source files)\n * against an explicit list of expected concepts, returning per-concept\n * {present, score 0-10, evidence, severity}.\n *\n * The judge is strict about distinguishing (a) a working implementation\n * from (b) a keyword-present stub. \"// TODO: mint button\" is NOT present.\n * Only real, functional, wired-up code counts.\n *\n * Use via {@link createSemanticConceptJudge} or directly via\n * {@link runSemanticConceptJudge}. Soft-fails (available=false) on LLM\n * or JSON-parse errors so the caller can treat that as \"layer skipped\"\n * rather than \"layer failed\" in a multi-layer pipeline.\n */\n\nimport type { ChatClient } from './analyst/chat-client'\nimport { paidJsonChat } from './chat-json-call'\nimport {\n CostLedger,\n type CostLedgerHandle,\n type CostReceipt,\n type CustomTokenPricing,\n} from './cost-ledger'\nimport type { LlmCallRequest } from './llm-client'\nimport type { Severity } from './multi-layer-verifier'\n\n// ─── Types ──────────────────────────────────────────────────────────────\n\n/**\n * Implementation complexity class for weighted scoring.\n *\n * - `render` (default): the concept is a UI surface that displays static\n * data — render a list, show a counter, lay out a button. Single-file\n * work, no external integration.\n * - `integrate`: the concept requires wiring a real external system —\n * wallet connect (wagmi + RainbowKit + chain config), payment provider\n * (Stripe Elements + intent + webhook), an API client with auth.\n * Multi-file, library-knowledge, runtime correctness matters.\n * - `compute`: the concept requires algorithmic work — solver, simulator,\n * constraint propagation, ML inference. Correctness > UI polish.\n *\n * Default weights (when applied via `weightConcepts: 'complexity'`):\n * render=1.0, integrate=2.0, compute=2.5\n *\n * Cross-vertical scoring without complexity weighting silently inflates\n * the rate of UI-heavy verticals (healthcare, fintech dashboards) vs\n * integration-heavy verticals (DeFi, wallets) — all concepts treated\n * equally even though the agent does 2-3x the work for `integrate`.\n */\nexport type ConceptComplexity = 'render' | 'integrate' | 'compute'\n\nexport interface ConceptSpec {\n name: string\n /** Short hints that help the judge; not used for matching. */\n keywords?: string[]\n /** Optional explicit weight; default 1.0. Overrides complexity-derived weight. */\n weight?: number\n /** Implementation complexity class. Default `render`. */\n complexity?: ConceptComplexity\n}\n\nexport interface ConceptFinding {\n concept: string\n present: boolean\n /** 0..10. 10 = production-ready; 7 = functional thin; 4 = partial; 0 = absent. */\n score: number\n evidence: string\n severity: Severity\n}\n\nexport interface SemanticConceptJudgeInput {\n /** Full natural-language prompt the agent was handed. */\n userRequest: string\n /** Rendered HTML the preview returns (UI artifacts). Optional. */\n servedHtml?: string\n /** Top-level source files from the agent's workdir. */\n sourceFiles: Array<{ path: string; content: string }>\n /** The expected concept list. */\n expectedConcepts: ConceptSpec[]\n /** Free-form metadata (id, difficulty) to inject into the prompt. */\n artifactLabel?: string\n artifactDescription?: string\n}\n\nexport interface SemanticConceptJudgeResult {\n kind: 'semantic-concept'\n version: string\n /** Normalized 0..1 score — mean of per-concept scores / 10. */\n score: number\n presentCount: number\n totalCount: number\n findings: ConceptFinding[]\n summary: string\n durationMs: number\n costUsd: number | null\n /** False on LLM/JSON error — treat as \"skipped / unable to judge\" in pipelines. */\n available: boolean\n error?: string\n}\n\n/**\n * Score-aggregation strategy. `mean` averages 0-10 scores uniformly.\n * `complexity` applies the default weight table (render=1, integrate=2,\n * compute=2.5) unless a concept has an explicit `weight`. `explicit`\n * honors only `weight` (defaulting to 1 for unspecified).\n */\nexport type ConceptWeightStrategy = 'mean' | 'complexity' | 'explicit'\n\nexport const DEFAULT_COMPLEXITY_WEIGHTS: Record<ConceptComplexity, number> = {\n render: 1.0,\n integrate: 2.0,\n compute: 2.5,\n}\n\nexport interface SemanticConceptJudgeOptions {\n /** Model id to call. Default 'claude-sonnet-4-6' via agent-eval defaults. */\n model?: string\n /** Per-call timeout. Default 300s. */\n timeoutMs?: number\n /** Provider-enforced output limit. Default 16000. */\n maxTokens?: number\n /** Pipeline budget for the prompt (source blob truncation). Default 45000. */\n maxSourceChars?: number\n /** Per-file cap before inclusion. Default 20000. */\n maxPerFileChars?: number\n /** HTML cap. Default 30000. */\n maxHtmlChars?: number\n /** Caller-owned transport. Required: agent-eval executes no paid model. */\n chat: ChatClient\n /** Endpoint rates used when the transport reports no billed amount. */\n pricing?: CustomTokenPricing\n costLedger?: CostLedgerHandle\n costPhase?: string\n costTags?: Record<string, string>\n signal?: AbortSignal\n /**\n * Score aggregation strategy. Default `mean` — uniform average across\n * concepts. Cross-vertical comparisons should use `complexity` to\n * neutralize the integrate-vs-render asymmetry.\n */\n weightConcepts?: ConceptWeightStrategy\n /** Override the default complexity → weight table. */\n complexityWeights?: Partial<Record<ConceptComplexity, number>>\n}\n\n// ─── Prompt assembly ────────────────────────────────────────────────────\n\nexport const SEMANTIC_CONCEPT_JUDGE_VERSION = 'semantic-concept-judge-v1-2026-04-24'\n\nconst DEFAULT_MAX_SOURCE = 45_000\nconst DEFAULT_MAX_HTML = 30_000\nconst DEFAULT_MAX_PER_FILE = 20_000\nconst DEFAULT_TIMEOUT = 300_000\nconst DEFAULT_MAX_TOKENS = 16_000\nconst DEFAULT_MODEL = 'claude-sonnet-4-6'\n\nconst SEMANTIC_SCHEMA = {\n type: 'object',\n additionalProperties: false,\n required: ['summary', 'concepts'],\n properties: {\n summary: { type: 'string', minLength: 20, maxLength: 600 },\n concepts: {\n type: 'array',\n minItems: 1,\n items: {\n type: 'object',\n additionalProperties: false,\n required: ['concept', 'present', 'score', 'evidence', 'severity'],\n properties: {\n concept: { type: 'string', minLength: 1, maxLength: 120 },\n present: { type: 'boolean' },\n score: { type: 'number', minimum: 0, maximum: 10 },\n evidence: { type: 'string', minLength: 5, maxLength: 400 },\n severity: { type: 'string', enum: ['critical', 'major', 'minor', 'info'] },\n },\n },\n },\n },\n}\n\nfunction truncate(body: string, cap: number, label: string): string {\n if (body.length <= cap) return body\n return `${body.slice(0, cap)}\\n… [truncated ${body.length - cap} chars of ${label}]`\n}\n\nfunction buildPrompt(\n input: SemanticConceptJudgeInput,\n opts: { maxPerFileChars: number; maxSourceChars: number; maxHtmlChars: number },\n): string {\n const sourceBlob = input.sourceFiles\n .filter((f) => f.content.length <= opts.maxPerFileChars)\n .map((f) => `--- FILE: ${f.path} ---\\n${f.content}`)\n .join('\\n\\n')\n\n const html = input.servedHtml ?? ''\n\n return `You are a strict code-review judge evaluating whether an agent's 0-to-1 build actually implements the features the user asked for.\n\nYou MUST distinguish:\n (a) WORKING code that implements the concept (rendered UI, wired handler, real API call),\n (b) KEYWORD-PRESENT stub (comments mentioning the concept, variable names, TODOs),\n (c) ABSENT (concept nowhere).\n\nA comment like \"// TODO: add mint button\" is NOT present — score 2-3. Only count a concept as present if there is real functional code: a rendered component, a call handler wired to state or a network call, a computed value actually used.\n\nUSER REQUEST (what the agent was asked to build):\n${input.userRequest}\n\n${input.artifactLabel ? `ARTIFACT METADATA:\\n name: ${input.artifactLabel}\\n description: ${input.artifactDescription ?? ''}\\n\\n` : ''}EXPECTED CONCEPTS (each must be graded independently):\n${input.expectedConcepts\n .map(\n (c, i) =>\n ` ${i + 1}. \"${c.name}\"${c.keywords?.length ? ` — hints: [${c.keywords.slice(0, 6).join(' | ')}]` : ''}`,\n )\n .join('\\n')}\n\n${html ? `SERVED HTML (what the preview returns when hit):\\n${truncate(html, opts.maxHtmlChars, 'HTML')}\\n\\n` : ''}SOURCE FILES (the agent's workdir):\n${truncate(sourceBlob, opts.maxSourceChars, 'source')}\n\nFor EACH concept, return:\n - concept: the concept name as given (match exactly)\n - present: boolean — does a working implementation exist?\n - score: 0-10 — 10 = production-ready; 7 = functional but thin; 4 = partial/stubbed; 2 = keyword-only comment; 0 = absent\n - evidence: cite \"<file>:<line>\" or \"served-html:<selector>\" pointing at the strongest supporting code. If the concept is absent or stubbed, explain what's missing.\n - severity:\n \"info\" when present: true AND score >= 7\n \"minor\" when present: true AND 4 <= score < 7\n \"major\" when present: false OR score < 4\n \"critical\" when the concept is not only absent but a core user flow depends on it\n\nAlso produce a \"summary\" (one sentence, 20-600 chars): overall verdict on whether this is a shippable implementation of the user request vs a keyword-dense placeholder.\n\nBE SKEPTICAL. Keyword matching already passed — your job is to catch what keyword matching misses. If the agent shipped a working build, say so. If it shipped a stub, say so. Don't grade on effort.\n\nReturn STRICT JSON. No prose outside the JSON.`\n}\n\n// ─── Runner ─────────────────────────────────────────────────────────────\n\n/**\n * Run the semantic concept judge. Soft-fails to available=false on\n * LLM/JSON errors — callers in a MultiLayerVerifier pipeline can treat\n * that as \"skip\" rather than \"fail.\"\n */\nexport async function runSemanticConceptJudge(\n input: SemanticConceptJudgeInput,\n options: SemanticConceptJudgeOptions,\n): Promise<SemanticConceptJudgeResult> {\n const start = Date.now()\n const totalCount = input.expectedConcepts.length\n\n if (totalCount === 0) {\n return {\n kind: 'semantic-concept',\n version: SEMANTIC_CONCEPT_JUDGE_VERSION,\n score: 0,\n presentCount: 0,\n totalCount: 0,\n findings: [],\n summary: 'no expected concepts declared',\n durationMs: 0,\n costUsd: null,\n available: false,\n error: 'no expected concepts declared',\n }\n }\n\n const opts = {\n chat: options.chat,\n model: options.model ?? options.chat.defaultModel ?? DEFAULT_MODEL,\n timeoutMs: options.timeoutMs ?? DEFAULT_TIMEOUT,\n maxTokens: options.maxTokens ?? DEFAULT_MAX_TOKENS,\n maxSourceChars: options.maxSourceChars ?? DEFAULT_MAX_SOURCE,\n maxPerFileChars: options.maxPerFileChars ?? DEFAULT_MAX_PER_FILE,\n maxHtmlChars: options.maxHtmlChars ?? DEFAULT_MAX_HTML,\n ...(options.pricing ? { pricing: options.pricing } : {}),\n costLedger: options.costLedger ?? new CostLedger(),\n costPhase: options.costPhase ?? 'judge.semantic-concept',\n costTags: options.costTags ?? {},\n signal: options.signal ?? new AbortController().signal,\n weightConcepts: options.weightConcepts ?? 'mean',\n complexityWeights: { ...DEFAULT_COMPLEXITY_WEIGHTS, ...(options.complexityWeights ?? {}) },\n }\n\n // Build a name → weight map for aggregation. Mean strategy keeps every\n // weight at 1 (uniform average). Complexity strategy reads the table\n // and lets an explicit `weight` override. Explicit strategy uses ONLY\n // the spec's `weight` (defaulting to 1).\n const weightForConcept = (spec: ConceptSpec): number => {\n if (opts.weightConcepts === 'mean') return 1\n if (spec.weight != null) return spec.weight\n if (opts.weightConcepts === 'complexity') {\n return opts.complexityWeights[spec.complexity ?? 'render'] ?? 1\n }\n return 1\n }\n const weightByName = new Map<string, number>(\n input.expectedConcepts.map((c) => [c.name, weightForConcept(c)]),\n )\n\n let receipt: CostReceipt | undefined\n try {\n const request = {\n model: opts.model,\n messages: [\n {\n role: 'system' as const,\n content:\n 'You are a strict code-review judge. Return strict JSON only. No prose outside the JSON. A keyword in a comment is NOT a working implementation.',\n },\n { role: 'user' as const, content: buildPrompt(input, opts) },\n ],\n jsonSchema: { name: 'semantic_concept_judge', schema: SEMANTIC_SCHEMA },\n temperature: 0,\n maxTokens: opts.maxTokens,\n timeoutMs: opts.timeoutMs,\n } satisfies LlmCallRequest\n const paid = await paidJsonChat<{ summary: string; concepts: ConceptFinding[] }>({\n chat: opts.chat,\n request,\n ledger: opts.costLedger,\n channel: 'judge',\n phase: opts.costPhase,\n actor: 'semantic-concept',\n tags: opts.costTags,\n signal: opts.signal,\n ...(opts.pricing ? { pricing: opts.pricing } : {}),\n })\n receipt = paid.receipt\n if (!paid.succeeded) throw paid.error\n const { value } = paid\n\n if (!value?.concepts || !Array.isArray(value.concepts)) {\n throw new Error('judge returned malformed response — expected array under \"concepts\"')\n }\n\n const findings: ConceptFinding[] = value.concepts.map((c) => ({\n concept: String(c.concept),\n present: Boolean(c.present),\n score: Math.max(0, Math.min(10, Number(c.score ?? 0))),\n evidence: String(c.evidence ?? ''),\n severity: (['critical', 'major', 'minor', 'info'] as const).includes(c.severity)\n ? c.severity\n : 'info',\n }))\n\n const presentCount = findings.filter((f) => f.present && f.score >= 7).length\n let weightSum = 0\n let weightedScoreSum = 0\n for (const f of findings) {\n const w = weightByName.get(f.concept) ?? 1\n weightSum += w\n weightedScoreSum += w * f.score\n }\n const scoreAvg =\n weightSum > 0\n ? weightedScoreSum / weightSum\n : findings.reduce((a, f) => a + f.score, 0) / Math.max(1, findings.length)\n\n return {\n kind: 'semantic-concept',\n version: SEMANTIC_CONCEPT_JUDGE_VERSION,\n score: Number((scoreAvg / 10).toFixed(3)),\n presentCount,\n totalCount,\n findings,\n summary: String(value.summary ?? ''),\n durationMs: Date.now() - start,\n costUsd: paid.receipt.costUnknown ? null : paid.receipt.costUsd,\n available: true,\n }\n } catch (err) {\n return {\n kind: 'semantic-concept',\n version: SEMANTIC_CONCEPT_JUDGE_VERSION,\n score: 0,\n presentCount: 0,\n totalCount,\n findings: [],\n summary: '',\n durationMs: Date.now() - start,\n costUsd: receipt && !receipt.costUnknown ? receipt.costUsd : null,\n available: false,\n error: err instanceof Error ? err.message : String(err),\n }\n }\n}\n\n/**\n * Factory: pin LLM options once, return a closure that accepts inputs.\n * Convenient for pipelines that want to share a single LlmClient config.\n */\nexport function createSemanticConceptJudge(\n options: SemanticConceptJudgeOptions,\n): (input: SemanticConceptJudgeInput) => Promise<SemanticConceptJudgeResult> {\n return (input) => runSemanticConceptJudge(input, options)\n}\n"],"mappings":";;;;;;;;;;;;;;;;AAeA,MAAM,0BAAU,IAAI,IAAmB;AAEvC,SAAS,SAAS,MAAqB;CACrC,IAAI,IAAI,QAAQ,IAAI,IAAI;CACxB,IAAI,CAAC,GAAG;EACN,IAAI,IAAI,MAAM;EACd,QAAQ,IAAI,MAAM,CAAC;CACrB;CACA,OAAO;AACT;AAEA,IAAa,sBAAb,MAAiC;CAEH;CAD5B;CACA,YAAY,MAA8B;EAAd,KAAA,OAAA;EAC1B,KAAK,QAAQ,SAAS,IAAI;EAC1B,IAAI,CAAC,WAAW,QAAQ,IAAI,CAAC,GAC3B,UAAU,QAAQ,IAAI,GAAG,EAAE,WAAW,KAAK,CAAC;CAEhD;CAEA,MAAM,OAAO,OAA+B;EAC1C,MAAM,OAAO,GAAG,KAAK,UAAU,KAAK,EAAE;EACtC,MAAM,KAAK,MAAM,mBAAmB;GAClC,eAAe,KAAK,MAAM,IAAI;EAChC,CAAC;CACH;AACF;;;;;;;;;;;;;;;;;;;;;;ACPA,IAAa,gBAAb,MAA2B;CAGG;CAF5B;CAEA,YAAY,MAA8B;EAAd,KAAA,OAAA;EAC1B,KAAK,WAAW,IAAI,oBAAoB,IAAI;CAC9C;CAEA,MAAM,OAAO,OAAe,UAA2C;EACrE,KAAK,MAAM,KAAK,UAAU;GACxB,MAAM,MAAwB;IAAE,GAAG;IAAG,QAAQ;GAAM;GACpD,MAAM,KAAK,SAAS,OAAO,GAAG;EAChC;CACF;;CAGA,UAA8B;EAC5B,IAAI,CAAC,WAAW,KAAK,IAAI,GAAG,OAAO,CAAC;EACpC,MAAM,MAAM,aAAa,KAAK,MAAM,MAAM;EAC1C,IAAI,CAAC,KAAK,OAAO,CAAC;EAClB,MAAM,MAA0B,CAAC;EACjC,KAAK,MAAM,QAAQ,IAAI,MAAM,IAAI,GAAG;GAClC,IAAI,CAAC,MAAM;GACX,IAAI;IACF,IAAI,KAAK,KAAK,MAAM,IAAI,CAAqB;GAC/C,QAAQ,CAGR;EACF;EACA,OAAO;CACT;;CAGA,QAAQ,OAAmC;EACzC,OAAO,KAAK,QAAQ,CAAC,CAAC,QAAQ,MAAM,EAAE,WAAW,KAAK;CACxD;AACF;;;;;AAiCA,SAAgB,kBAAkB,GAAmB,GAA4B;CAC/E,IAAI,EAAE,aAAa,EAAE,UAAU,OAAO;CACtC,IAAI,KAAK,KAAK,EAAE,cAAc,MAAM,EAAE,cAAc,EAAE,IAAI,KAAM,OAAO;CACvE,IAAI,EAAE,cAAc,WAAW,EAAE,cAAc,QAAQ,OAAO;CAC9D,OAAO;AACT;;;;;AAMA,SAAgB,aACd,UACA,SACA,SAAqB,CAAC,GACR;CACd,MAAM,aAAa,OAAO,cAAc;CACxC,MAAM,WAAW,IAAI,IAAI,SAAS,KAAK,MAAM,CAAC,EAAE,YAAY,CAAC,CAAC,CAAC;CAC/D,MAAM,UAAU,IAAI,IAAI,QAAQ,KAAK,MAAM,CAAC,EAAE,YAAY,CAAC,CAAC,CAAC;CAE7D,MAAM,WAA+B,CAAC;CACtC,MAAM,cAAkC,CAAC;CACzC,MAAM,YAAgC,CAAC;CACvC,MAAM,UAAmC,CAAC;CAE1C,KAAK,MAAM,CAAC,IAAI,QAAQ,SAAS;EAC/B,MAAM,OAAO,SAAS,IAAI,EAAE;EAC5B,IAAI,CAAC,MAAM;GACT,SAAS,KAAK,GAAG;GACjB;EACF;EACA,IAAI,WAAW,MAAM,GAAG,GACtB,QAAQ,KAAK;GAAE,UAAU;GAAM,SAAS;EAAI,CAAC;OAE7C,UAAU,KAAK,GAAG;CAEtB;CACA,KAAK,MAAM,CAAC,IAAI,SAAS,UACvB,IAAI,CAAC,QAAQ,IAAI,EAAE,GAAG,YAAY,KAAK,IAAI;CAE7C,OAAO;EAAE;EAAU;EAAa;EAAW;CAAQ;AACrD;;;AC9BA,MAAa,6BAAgE;CAC3E,QAAQ;CACR,WAAW;CACX,SAAS;AACX;AAmCA,MAAa,iCAAiC;AAE9C,MAAM,qBAAqB;AAC3B,MAAM,mBAAmB;AACzB,MAAM,uBAAuB;AAC7B,MAAM,kBAAkB;AACxB,MAAM,qBAAqB;AAC3B,MAAM,gBAAgB;AAEtB,MAAM,kBAAkB;CACtB,MAAM;CACN,sBAAsB;CACtB,UAAU,CAAC,WAAW,UAAU;CAChC,YAAY;EACV,SAAS;GAAE,MAAM;GAAU,WAAW;GAAI,WAAW;EAAI;EACzD,UAAU;GACR,MAAM;GACN,UAAU;GACV,OAAO;IACL,MAAM;IACN,sBAAsB;IACtB,UAAU;KAAC;KAAW;KAAW;KAAS;KAAY;IAAU;IAChE,YAAY;KACV,SAAS;MAAE,MAAM;MAAU,WAAW;MAAG,WAAW;KAAI;KACxD,SAAS,EAAE,MAAM,UAAU;KAC3B,OAAO;MAAE,MAAM;MAAU,SAAS;MAAG,SAAS;KAAG;KACjD,UAAU;MAAE,MAAM;MAAU,WAAW;MAAG,WAAW;KAAI;KACzD,UAAU;MAAE,MAAM;MAAU,MAAM;OAAC;OAAY;OAAS;OAAS;MAAM;KAAE;IAC3E;GACF;EACF;CACF;AACF;AAEA,SAAS,SAAS,MAAc,KAAa,OAAuB;CAClE,IAAI,KAAK,UAAU,KAAK,OAAO;CAC/B,OAAO,GAAG,KAAK,MAAM,GAAG,GAAG,EAAE,iBAAiB,KAAK,SAAS,IAAI,YAAY,MAAM;AACpF;AAEA,SAAS,YACP,OACA,MACQ;CACR,MAAM,aAAa,MAAM,YACtB,QAAQ,MAAM,EAAE,QAAQ,UAAU,KAAK,eAAe,CAAC,CACvD,KAAK,MAAM,aAAa,EAAE,KAAK,QAAQ,EAAE,SAAS,CAAC,CACnD,KAAK,MAAM;CAEd,MAAM,OAAO,MAAM,cAAc;CAEjC,OAAO;;;;;;;;;;EAUP,MAAM,YAAY;;EAElB,MAAM,gBAAgB,+BAA+B,MAAM,cAAc,mBAAmB,MAAM,uBAAuB,GAAG,QAAQ,GAAG;EACvI,MAAM,iBACL,KACE,GAAG,MACF,KAAK,IAAI,EAAE,KAAK,EAAE,KAAK,GAAG,EAAE,UAAU,SAAS,cAAc,EAAE,SAAS,MAAM,GAAG,CAAC,CAAC,CAAC,KAAK,KAAK,EAAE,KAAK,IACzG,CAAC,CACA,KAAK,IAAI,EAAE;;EAEZ,OAAO,qDAAqD,SAAS,MAAM,KAAK,cAAc,MAAM,EAAE,QAAQ,GAAG;EACjH,SAAS,YAAY,KAAK,gBAAgB,QAAQ,EAAE;;;;;;;;;;;;;;;;;;AAkBtD;;;;;;AASA,eAAsB,wBACpB,OACA,SACqC;CACrC,MAAM,QAAQ,KAAK,IAAI;CACvB,MAAM,aAAa,MAAM,iBAAiB;CAE1C,IAAI,eAAe,GACjB,OAAO;EACL,MAAM;EACN,SAAS;EACT,OAAO;EACP,cAAc;EACd,YAAY;EACZ,UAAU,CAAC;EACX,SAAS;EACT,YAAY;EACZ,SAAS;EACT,WAAW;EACX,OAAO;CACT;CAGF,MAAM,OAAO;EACX,MAAM,QAAQ;EACd,OAAO,QAAQ,SAAS,QAAQ,KAAK,gBAAgB;EACrD,WAAW,QAAQ,aAAa;EAChC,WAAW,QAAQ,aAAa;EAChC,gBAAgB,QAAQ,kBAAkB;EAC1C,iBAAiB,QAAQ,mBAAmB;EAC5C,cAAc,QAAQ,gBAAgB;EACtC,GAAI,QAAQ,UAAU,EAAE,SAAS,QAAQ,QAAQ,IAAI,CAAC;EACtD,YAAY,QAAQ,cAAc,IAAI,WAAW;EACjD,WAAW,QAAQ,aAAa;EAChC,UAAU,QAAQ,YAAY,CAAC;EAC/B,QAAQ,QAAQ,UAAU,IAAI,gBAAgB,CAAC,CAAC;EAChD,gBAAgB,QAAQ,kBAAkB;EAC1C,mBAAmB;GAAE,GAAG;GAA4B,GAAI,QAAQ,qBAAqB,CAAC;EAAG;CAC3F;CAMA,MAAM,oBAAoB,SAA8B;EACtD,IAAI,KAAK,mBAAmB,QAAQ,OAAO;EAC3C,IAAI,KAAK,UAAU,MAAM,OAAO,KAAK;EACrC,IAAI,KAAK,mBAAmB,cAC1B,OAAO,KAAK,kBAAkB,KAAK,cAAc,aAAa;EAEhE,OAAO;CACT;CACA,MAAM,eAAe,IAAI,IACvB,MAAM,iBAAiB,KAAK,MAAM,CAAC,EAAE,MAAM,iBAAiB,CAAC,CAAC,CAAC,CACjE;CAEA,IAAI;CACJ,IAAI;EACF,MAAM,UAAU;GACd,OAAO,KAAK;GACZ,UAAU,CACR;IACE,MAAM;IACN,SACE;GACJ,GACA;IAAE,MAAM;IAAiB,SAAS,YAAY,OAAO,IAAI;GAAE,CAC7D;GACA,YAAY;IAAE,MAAM;IAA0B,QAAQ;GAAgB;GACtE,aAAa;GACb,WAAW,KAAK;GAChB,WAAW,KAAK;EAClB;EACA,MAAM,OAAO,MAAM,aAA8D;GAC/E,MAAM,KAAK;GACX;GACA,QAAQ,KAAK;GACb,SAAS;GACT,OAAO,KAAK;GACZ,OAAO;GACP,MAAM,KAAK;GACX,QAAQ,KAAK;GACb,GAAI,KAAK,UAAU,EAAE,SAAS,KAAK,QAAQ,IAAI,CAAC;EAClD,CAAC;EACD,UAAU,KAAK;EACf,IAAI,CAAC,KAAK,WAAW,MAAM,KAAK;EAChC,MAAM,EAAE,UAAU;EAElB,IAAI,CAAC,OAAO,YAAY,CAAC,MAAM,QAAQ,MAAM,QAAQ,GACnD,MAAM,IAAI,MAAM,uEAAqE;EAGvF,MAAM,WAA6B,MAAM,SAAS,KAAK,OAAO;GAC5D,SAAS,OAAO,EAAE,OAAO;GACzB,SAAS,QAAQ,EAAE,OAAO;GAC1B,OAAO,KAAK,IAAI,GAAG,KAAK,IAAI,IAAI,OAAO,EAAE,SAAS,CAAC,CAAC,CAAC;GACrD,UAAU,OAAO,EAAE,YAAY,EAAE;GACjC,UAAW;IAAC;IAAY;IAAS;IAAS;GAAM,CAAC,CAAW,SAAS,EAAE,QAAQ,IAC3E,EAAE,WACF;EACN,EAAE;EAEF,MAAM,eAAe,SAAS,QAAQ,MAAM,EAAE,WAAW,EAAE,SAAS,CAAC,CAAC,CAAC;EACvE,IAAI,YAAY;EAChB,IAAI,mBAAmB;EACvB,KAAK,MAAM,KAAK,UAAU;GACxB,MAAM,IAAI,aAAa,IAAI,EAAE,OAAO,KAAK;GACzC,aAAa;GACb,oBAAoB,IAAI,EAAE;EAC5B;EACA,MAAM,WACJ,YAAY,IACR,mBAAmB,YACnB,SAAS,QAAQ,GAAG,MAAM,IAAI,EAAE,OAAO,CAAC,IAAI,KAAK,IAAI,GAAG,SAAS,MAAM;EAE7E,OAAO;GACL,MAAM;GACN,SAAS;GACT,OAAO,QAAQ,WAAW,GAAA,CAAI,QAAQ,CAAC,CAAC;GACxC;GACA;GACA;GACA,SAAS,OAAO,MAAM,WAAW,EAAE;GACnC,YAAY,KAAK,IAAI,IAAI;GACzB,SAAS,KAAK,QAAQ,cAAc,OAAO,KAAK,QAAQ;GACxD,WAAW;EACb;CACF,SAAS,KAAK;EACZ,OAAO;GACL,MAAM;GACN,SAAS;GACT,OAAO;GACP,cAAc;GACd;GACA,UAAU,CAAC;GACX,SAAS;GACT,YAAY,KAAK,IAAI,IAAI;GACzB,SAAS,WAAW,CAAC,QAAQ,cAAc,QAAQ,UAAU;GAC7D,WAAW;GACX,OAAO,eAAe,QAAQ,IAAI,UAAU,OAAO,GAAG;EACxD;CACF;AACF"}
|
|
@@ -1 +0,0 @@
|
|
|
1
|
-
{"version":3,"file":"tool-groups-DjwlMBvW.d.ts","names":[],"sources":["../src/analyst/tool-groups.ts"],"mappings":";;;;KAmBY;;;;;;;;;;;;;;;;;;;;iBAgDI,wBACd,OAAO,oBACP,OAAO,qBACN"}
|
|
@@ -1 +0,0 @@
|
|
|
1
|
-
{"version":3,"file":"tool-waste-CwGHzBzX.js","names":["prev"],"sources":["../src/statistics/agreement-irr.ts","../src/failure-taxonomy.ts","../src/pipelines/budget-breach.ts","../src/pipelines/failure-cluster.ts","../src/pipelines/judge-agreement.ts","../src/tool-use-metrics.ts","../src/pipelines/tool-waste.ts"],"sourcesContent":["import { ValidationError } from '../errors'\nimport {\n type ContinuousAgreement,\n type ContinuousAgreementOptions,\n continuousAgreement,\n} from '../judge-calibration'\nimport type { JudgeScore } from '../types'\n\n/**\n * Inter-rater reliability — Krippendorff's α under the squared-difference\n * metric, pooled across dimensions.\n *\n * Each inner array is one judge's scores. Items are matched by position\n * WITHIN a dimension: the k-th score a judge supplies carrying dimension\n * `d` is item k of `d`, and the ratings compared against each other are\n * the ones different judges gave to the same item. Every judge that scores\n * a dimension at all must supply the same number of scores for it —\n * ragged input cannot be aligned into items and throws rather than\n * comparing mismatched items.\n *\n * α = 1 − D_observed / D_expected: D_observed averages the squared\n * difference over within-item judge pairs, D_expected over every pair of\n * ratings irrespective of item. α = 1 is perfect agreement, 0 is chance,\n * negative is systematic disagreement.\n */\nexport function interRaterReliability(judgeScores: JudgeScore[][]): number {\n if (judgeScores.length < 2) return 1\n\n // dimension → one score sequence per judge, in that judge's supplied order.\n const perDimension = new Map<string, number[][]>()\n for (let judgeIndex = 0; judgeIndex < judgeScores.length; judgeIndex++) {\n for (const s of judgeScores[judgeIndex]!) {\n let byJudge = perDimension.get(s.dimension)\n if (byJudge === undefined) {\n byJudge = Array.from({ length: judgeScores.length }, () => [] as number[])\n perDimension.set(s.dimension, byJudge)\n }\n byJudge[judgeIndex]!.push(s.score)\n }\n }\n\n const allValues: number[] = []\n const pairDiffs: number[] = []\n\n for (const [dimension, byJudge] of perDimension) {\n const scoring = byJudge.filter((scores) => scores.length > 0)\n if (scoring.length < 2) continue\n const itemCount = scoring[0]!.length\n if (scoring.some((scores) => scores.length !== itemCount)) {\n throw new ValidationError(\n `interRaterReliability: dimension '${dimension}' has judges supplying ` +\n `${scoring.map((scores) => scores.length).join('/')} scores — items cannot be aligned`,\n )\n }\n for (let item = 0; item < itemCount; item++) {\n const ratings = scoring.map((scores) => scores[item]!)\n for (const v of ratings) allValues.push(v)\n for (let i = 0; i < ratings.length; i++) {\n for (let j = i + 1; j < ratings.length; j++) {\n pairDiffs.push((ratings[i]! - ratings[j]!) ** 2)\n }\n }\n }\n }\n\n if (pairDiffs.length === 0 || allValues.length < 2) return 1\n\n const observedDisagreement = pairDiffs.reduce((a, b) => a + b, 0) / pairDiffs.length\n\n // Expected disagreement from all possible pairings of values\n let expectedDisagreement = 0\n let expectedCount = 0\n for (let i = 0; i < allValues.length; i++) {\n for (let j = i + 1; j < allValues.length; j++) {\n expectedDisagreement += (allValues[i]! - allValues[j]!) ** 2\n expectedCount++\n }\n }\n expectedDisagreement = expectedCount > 0 ? expectedDisagreement / expectedCount : 0\n\n if (expectedDisagreement === 0) return 1\n return 1 - observedDisagreement / expectedDisagreement\n}\n\n// ── Corpus-wide inter-rater agreement ──────────────────────────────\n//\n// `interRaterReliability(judgeScores)` computes a within-item\n// Krippendorff α — multiple judges score *the same item* and we ask\n// \"how much do their scores agree on that item?\" Useful for a single\n// scenario, but it cannot answer \"how reliable are these judges across\n// the whole evaluation corpus?\"\n//\n// `corpusInterRaterAgreement` does the corpus-wide question properly.\n// Inputs are flat per-(item, judge, dimension) score records. For each\n// dimension we pivot to a complete [n_items × n_judges] matrix and feed\n// it to the ICC(2,1) + κ_w machinery already validated in\n// `judge-calibration.ts`. An overall pooled metric averages the\n// per-dimension ICC/κ across dimensions.\n\nexport interface CorpusScoreRecord {\n /** Stable identifier for the rated item (scenario, span, turn, …). */\n itemId: string\n /** Identifier for the judge that produced this score. */\n judgeName: string\n /** Dimension name (matches `JudgeScore.dimension`). */\n dimension: string\n /** Numeric score; must be finite. */\n score: number\n}\n\nexport interface CorpusAgreementPerDimension extends ContinuousAgreement {\n dimension: string\n /** Item IDs that contributed to this dimension's matrix (every judge scored them). */\n itemIds: string[]\n /** Judge IDs that contributed to this dimension's matrix. */\n judgeIds: string[]\n}\n\nexport interface CorpusAgreementReport {\n /** Per-dimension ICC(2,1) + κ_w + Pearson + Spearman + bootstrap CIs. */\n perDimension: CorpusAgreementPerDimension[]\n /** Mean ICC across dimensions (NaN if no dimension yielded a finite ICC). */\n overallIcc: number\n /** Mean weighted κ across dimensions (NaN if none finite). */\n overallWeightedKappa: number\n /** Dimensions evaluated (sorted). */\n dimensions: string[]\n /** Judges seen across the corpus (sorted). */\n judgeIds: string[]\n}\n\nexport interface CorpusAgreementOptions extends ContinuousAgreementOptions {\n /**\n * Restrict the audit to these dimensions. Default = every dimension\n * that appears in the input. A dimension named here but absent from\n * the input throws — silent omission would corrupt the overall metric.\n */\n dimensions?: string[]\n /**\n * Restrict the audit to these judges. Default = every judge that\n * appears in the input. A judge named here but absent from a\n * dimension throws (see \"fail loud\" below).\n */\n judges?: string[]\n}\n\n/**\n * Corpus-wide inter-rater agreement across N items × M judges × D dimensions.\n *\n * For each dimension, builds the [n_items][n_judges] matrix of scores\n * (keeping only items every judge rated on that dimension), then runs\n * `continuousAgreement` to get ICC(2,1), κ_w, Pearson, Spearman, and\n * bootstrap CIs. Reports a pooled mean across dimensions as a single\n * \"is this judge panel reliable on this corpus?\" number.\n *\n * Fail-loud contract:\n * - Empty input throws.\n * - Fewer than 2 judges or fewer than 2 items per dimension throws.\n * - A judge present in some dimensions but with zero scored items on\n * another dimension throws (would silently shrink the matrix).\n * - Duplicate (itemId, judgeName, dimension) records throw.\n */\nexport function corpusInterRaterAgreement(\n records: CorpusScoreRecord[],\n opts: CorpusAgreementOptions = {},\n): CorpusAgreementReport {\n if (records.length === 0) {\n throw new ValidationError('corpusInterRaterAgreement: no score records supplied')\n }\n\n const judgesSeen = new Set<string>()\n const dimsSeen = new Set<string>()\n // dimension → judge → itemId → score\n const grid = new Map<string, Map<string, Map<string, number>>>()\n\n for (const r of records) {\n if (!Number.isFinite(r.score)) {\n throw new ValidationError(\n `corpusInterRaterAgreement: non-finite score for (item=${r.itemId}, judge=${r.judgeName}, dim=${r.dimension})`,\n )\n }\n judgesSeen.add(r.judgeName)\n dimsSeen.add(r.dimension)\n const byJudge = grid.get(r.dimension) ?? new Map<string, Map<string, number>>()\n const byItem = byJudge.get(r.judgeName) ?? new Map<string, number>()\n if (byItem.has(r.itemId)) {\n throw new ValidationError(\n `corpusInterRaterAgreement: duplicate record for (item=${r.itemId}, judge=${r.judgeName}, dim=${r.dimension})`,\n )\n }\n byItem.set(r.itemId, r.score)\n byJudge.set(r.judgeName, byItem)\n grid.set(r.dimension, byJudge)\n }\n\n const targetDims = opts.dimensions ?? [...dimsSeen].sort()\n for (const d of targetDims) {\n if (!dimsSeen.has(d)) {\n throw new ValidationError(\n `corpusInterRaterAgreement: dimension '${d}' was requested but no records carry it`,\n )\n }\n }\n const targetJudges = opts.judges ? [...opts.judges] : [...judgesSeen].sort()\n for (const j of targetJudges) {\n if (!judgesSeen.has(j)) {\n throw new ValidationError(\n `corpusInterRaterAgreement: judge '${j}' was requested but produced no records`,\n )\n }\n }\n if (targetJudges.length < 2) {\n throw new ValidationError(\n `corpusInterRaterAgreement: need ≥2 judges, got ${targetJudges.length}`,\n )\n }\n\n const perDimension: CorpusAgreementPerDimension[] = []\n const iccs: number[] = []\n const kappas: number[] = []\n\n for (const dim of targetDims) {\n const byJudge = grid.get(dim)!\n // Fail loud: every requested judge must have scored ≥1 item on this dim.\n const judgeItemCounts: Record<string, number> = {}\n for (const j of targetJudges) {\n const m = byJudge.get(j)\n judgeItemCounts[j] = m?.size ?? 0\n }\n const emptyJudges = targetJudges.filter((j) => judgeItemCounts[j] === 0)\n if (emptyJudges.length > 0) {\n throw new ValidationError(\n `corpusInterRaterAgreement: dimension '${dim}' has no scores from judge(s) ${emptyJudges.join(', ')} (counts: ${JSON.stringify(judgeItemCounts)})`,\n )\n }\n\n // Items rated by *every* requested judge on this dim.\n let commonItems: Set<string> | null = null\n for (const j of targetJudges) {\n const ids = new Set(byJudge.get(j)!.keys())\n if (commonItems === null) {\n commonItems = ids\n } else {\n const prev: Set<string> = commonItems\n commonItems = new Set([...prev].filter((x) => ids.has(x)))\n }\n }\n const sortedItems = [...(commonItems ?? new Set<string>())].sort()\n if (sortedItems.length < 2) {\n throw new ValidationError(\n `corpusInterRaterAgreement: dimension '${dim}' has ${sortedItems.length} item(s) rated by all ${targetJudges.length} judges (need ≥2)`,\n )\n }\n\n const matrix: number[][] = sortedItems.map((itemId) =>\n targetJudges.map((j) => byJudge.get(j)!.get(itemId)!),\n )\n const agreement = continuousAgreement(matrix, opts)\n perDimension.push({\n ...agreement,\n dimension: dim,\n itemIds: sortedItems,\n judgeIds: [...targetJudges],\n })\n if (Number.isFinite(agreement.icc)) iccs.push(agreement.icc)\n if (Number.isFinite(agreement.weightedKappa)) kappas.push(agreement.weightedKappa)\n }\n\n const mean = (xs: number[]) =>\n xs.length === 0 ? Number.NaN : xs.reduce((a, b) => a + b, 0) / xs.length\n return {\n perDimension,\n overallIcc: mean(iccs),\n overallWeightedKappa: mean(kappas),\n dimensions: targetDims,\n judgeIds: targetJudges,\n }\n}\n\n/**\n * Convenience adapter for `JudgeScore[]` data keyed externally by item.\n *\n * Use when you have per-item arrays of `JudgeScore[]` (e.g. one\n * `ScenarioResult.judgeScores` per scenario) and want corpus-wide\n * agreement without manually flattening. `itemId` must be unique per\n * row of `itemsScores`.\n */\nexport function corpusInterRaterAgreementFromJudgeScores(\n itemsScores: Array<{ itemId: string; scores: JudgeScore[] }>,\n opts: CorpusAgreementOptions = {},\n): CorpusAgreementReport {\n const records: CorpusScoreRecord[] = []\n const seen = new Set<string>()\n for (const { itemId, scores } of itemsScores) {\n if (seen.has(itemId)) {\n throw new ValidationError(\n `corpusInterRaterAgreementFromJudgeScores: duplicate itemId '${itemId}'`,\n )\n }\n seen.add(itemId)\n for (const s of scores) {\n records.push({\n itemId,\n judgeName: s.judgeName,\n dimension: s.dimension,\n score: s.score,\n })\n }\n }\n return corpusInterRaterAgreement(records, opts)\n}\n","/**\n * Failure taxonomy — canonical classes + a default classifier.\n *\n * Every failed run should end up in a named class. The classifier here\n * is rule-based (fast, deterministic); an LLM fallback can be added by\n * the consumer for novel cases and trained into the rule base over time.\n *\n * Consumers call `classifyFailure(run, spans, events)` and persist the\n * returned class as `Run.outcome.failureClass`.\n */\n\nimport type { FailureClass, Run, Span, TraceEvent } from './trace/schema'\nimport { FAILURE_CLASSES } from './trace/schema'\n\nexport { FAILURE_CLASSES, type FailureClass }\n\nexport interface FailureContext {\n run: Run\n spans: Span[]\n events: TraceEvent[]\n}\n\nexport interface FailureClassification {\n failureClass: FailureClass\n reason: string\n triggerSpanId?: string\n triggerEventId?: string\n}\n\n/** Ordered rules — first match wins. */\nexport interface FailureRule {\n id: string\n match: (ctx: FailureContext) => {\n failureClass: FailureClass\n reason: string\n triggerSpanId?: string\n triggerEventId?: string\n } | null\n}\n\nexport const DEFAULT_RULES: FailureRule[] = [\n // Outcome already named? Respect it.\n {\n id: 'explicit-outcome',\n match: ({ run }) => {\n const fc = run.outcome?.failureClass\n if (fc && fc !== 'unknown')\n return { failureClass: fc, reason: 'outcome.failureClass set explicitly' }\n return null\n },\n },\n {\n id: 'knowledge-readiness-blocked',\n match: ({ events }) => {\n const event = events.find(\n (e) =>\n e.kind === 'custom' &&\n e.payload.kind === 'readiness_scored' &&\n e.payload.passed === false,\n )\n return event\n ? {\n failureClass: 'knowledge_readiness_blocked',\n reason: 'knowledge readiness report blocked execution',\n triggerEventId: event.eventId,\n }\n : null\n },\n },\n {\n id: 'bad-integration-manifest',\n match: ({ events }) => {\n const event = events.find(\n (e) =>\n e.kind === 'custom' &&\n ((e.payload.kind === 'integration_manifest_validated' && e.payload.valid === false) ||\n (e.payload.kind === 'integration_invoke_failed' &&\n e.payload.code === 'manifest_invalid')),\n )\n return event\n ? {\n failureClass: 'bad_integration_manifest',\n reason: 'integration manifest validation failed before launch',\n triggerEventId: event.eventId,\n }\n : null\n },\n },\n {\n id: 'missing-integration-connection',\n match: ({ events }) => {\n const event = events.find(\n (e) =>\n e.kind === 'custom' &&\n e.payload.kind === 'integration_manifest_resolved' &&\n hasResolutionStatus(e.payload, 'missing_connection'),\n )\n return event\n ? {\n failureClass: 'missing_integration_connection',\n reason: 'required integration connection was missing',\n triggerEventId: event.eventId,\n }\n : null\n },\n },\n {\n id: 'missing-integration-scope',\n match: ({ events }) => {\n const event = events.find(\n (e) =>\n e.kind === 'custom' &&\n ((e.payload.kind === 'integration_manifest_resolved' && hasMissingScopes(e.payload)) ||\n (e.payload.kind === 'integration_invoke_failed' && e.payload.code === 'scope_denied')),\n )\n return event\n ? {\n failureClass: 'missing_integration_scope',\n reason: 'integration grant or connection lacks required scopes',\n triggerEventId: event.eventId,\n }\n : null\n },\n },\n {\n id: 'integration-approval-required',\n match: ({ events }) => {\n const event = events.find(\n (e) =>\n e.kind === 'custom' &&\n ((e.payload.kind === 'integration_invoke' && e.payload.status === 'approval_required') ||\n (e.payload.kind === 'integration_invoke_failed' &&\n e.payload.code === 'approval_required') ||\n e.payload.kind === 'integration_approval_required'),\n )\n return event\n ? {\n failureClass: 'integration_approval_required',\n reason: 'integration write paused for user approval',\n triggerEventId: event.eventId,\n }\n : null\n },\n },\n {\n id: 'integration-auth-expired',\n match: ({ events }) => {\n const event = events.find(\n (e) =>\n e.kind === 'custom' &&\n e.payload.kind === 'integration_invoke_failed' &&\n (e.payload.code === 'auth_expired' ||\n e.payload.code === 'connection_not_active' ||\n e.payload.code === 'capability_expired' ||\n e.payload.status === 'expired'),\n )\n return event\n ? {\n failureClass: 'integration_auth_expired',\n reason: 'integration connection or capability expired',\n triggerEventId: event.eventId,\n }\n : null\n },\n },\n {\n id: 'unsafe-integration-write-denied',\n match: ({ events }) => {\n const event = events.find(\n (e) =>\n e.kind === 'custom' &&\n e.payload.kind === 'integration_invoke_failed' &&\n (e.payload.code === 'unsafe_write_denied' ||\n e.payload.code === 'policy_denied' ||\n e.payload.code === 'action_denied'),\n )\n return event\n ? {\n failureClass: 'unsafe_integration_write_denied',\n reason: 'integration write was denied by policy or capability scope',\n triggerEventId: event.eventId,\n }\n : null\n },\n },\n {\n id: 'integration-provider-failure',\n match: ({ events }) => {\n const event = events.find(\n (e) =>\n e.kind === 'custom' &&\n e.payload.kind === 'integration_invoke_failed' &&\n ![\n 'scope_denied',\n 'approval_required',\n 'auth_expired',\n 'connection_not_active',\n 'capability_expired',\n 'unsafe_write_denied',\n 'policy_denied',\n 'action_denied',\n 'manifest_invalid',\n ].includes(String(e.payload.code)),\n )\n return event\n ? {\n failureClass: 'integration_provider_failure',\n reason: 'integration provider invocation failed',\n triggerEventId: event.eventId,\n }\n : null\n },\n },\n {\n id: 'missing-credentials',\n match: ({ events }) => {\n const event = events.find(\n (e) =>\n e.kind === 'custom' &&\n e.payload.kind === 'knowledge_gap' &&\n e.payload.category === 'credential_or_secret',\n )\n return event\n ? {\n failureClass: 'missing_credentials',\n reason: 'required credential or secret was missing',\n triggerEventId: event.eventId,\n }\n : null\n },\n },\n {\n id: 'bad-retrieval',\n match: ({ run, spans }) => {\n if (run.outcome?.pass !== false) return null\n const retrieval = spans.find(\n (s) =>\n s.kind === 'retrieval' && (s.hits.length === 0 || s.hits.every((hit) => hit.score <= 0)),\n )\n return retrieval\n ? {\n failureClass: 'bad_retrieval',\n reason: 'retrieval returned no useful hits for a failed run',\n triggerSpanId: retrieval.spanId,\n }\n : null\n },\n },\n {\n id: 'insufficient-evidence',\n match: ({ events }) => {\n const event = events.find(\n (e) =>\n e.kind === 'custom' &&\n e.payload.kind === 'knowledge_gap' &&\n e.payload.reason === 'insufficient_evidence',\n )\n return event\n ? {\n failureClass: 'insufficient_evidence',\n reason: 'task proceeded with insufficient supporting evidence',\n triggerEventId: event.eventId,\n }\n : null\n },\n },\n {\n id: 'contradictory-evidence',\n match: ({ events }) => {\n const event = events.find(\n (e) =>\n e.kind === 'custom' &&\n e.payload.kind === 'knowledge_gap' &&\n e.payload.reason === 'contradictory_evidence',\n )\n return event\n ? {\n failureClass: 'contradictory_evidence',\n reason: 'supporting evidence contradicted itself',\n triggerEventId: event.eventId,\n }\n : null\n },\n },\n // Budget breach events\n {\n id: 'budget-breach',\n match: ({ events }) => {\n const breach = events.find((e) => e.kind === 'budget_breach')\n return breach\n ? {\n failureClass: 'budget_exceeded',\n reason: `budget breached on ${breach.payload.dimension ?? 'unknown dimension'}`,\n triggerEventId: breach.eventId,\n }\n : null\n },\n },\n // Policy violations\n {\n id: 'policy-violation',\n match: ({ events }) => {\n const e = events.find((x) => x.kind === 'policy_violation')\n return e\n ? {\n failureClass: 'policy_violation',\n reason: 'policy_violation event emitted',\n triggerEventId: e.eventId,\n }\n : null\n },\n },\n // Sandbox non-zero exit code\n {\n id: 'sandbox-failure',\n match: ({ spans }) => {\n const s = spans.find(\n (x) => x.kind === 'sandbox' && typeof x.exitCode === 'number' && x.exitCode !== 0,\n )\n if (!s) return null\n return {\n failureClass: 'sandbox_failure',\n reason: `sandbox exited ${(s as Extract<Span, { kind: 'sandbox' }>).exitCode}`,\n triggerSpanId: s.spanId,\n }\n },\n },\n // Timeout: run aborted by external signal\n {\n id: 'timeout',\n match: ({ run, events }) => {\n if (run.status !== 'aborted') return null\n const hasTimeout = events.some(\n (e) =>\n e.kind === 'error' &&\n String(e.payload.reason ?? '')\n .toLowerCase()\n .includes('timeout'),\n )\n const note = (run.outcome?.notes ?? '').toLowerCase()\n if (hasTimeout || note.includes('timeout') || note.includes('deadline')) {\n return { failureClass: 'timeout', reason: 'timeout signal observed' }\n }\n return null\n },\n },\n // Tool recovery failure: many consecutive tool errors on the same tool\n {\n id: 'tool-recovery-failure',\n match: ({ spans }) => {\n const tools = spans.filter((s) => s.kind === 'tool')\n const byTool = new Map<string, Span[]>()\n for (const t of tools) {\n const name = (t as Extract<Span, { kind: 'tool' }>).toolName\n const arr = byTool.get(name) ?? []\n arr.push(t)\n byTool.set(name, arr)\n }\n for (const [name, arr] of byTool) {\n const errs = arr.filter((s) => s.status === 'error')\n if (errs.length >= 3 && errs.length === arr.length) {\n return {\n failureClass: 'tool_recovery_failure',\n reason: `${errs.length} consecutive errors on tool \"${name}\"`,\n triggerSpanId: errs[errs.length - 1]!.spanId,\n }\n }\n }\n return null\n },\n },\n // Tool selection error: the run failed and agent called zero tools despite having them\n {\n id: 'tool-selection-error',\n match: ({ run, spans }) => {\n if (run.outcome?.pass !== false) return null\n const hasToolsAvailable = spans.some(\n (s) =>\n s.kind === 'agent' &&\n (s.attributes?.toolsAvailable as number | undefined) !== undefined &&\n (s.attributes?.toolsAvailable as number) > 0,\n )\n const tools = spans.filter((s) => s.kind === 'tool')\n if (hasToolsAvailable && tools.length === 0) {\n return {\n failureClass: 'tool_selection_error',\n reason: 'tools were available but none were called',\n }\n }\n return null\n },\n },\n // Format drift: scored by a judge with dimension='format' below threshold\n {\n id: 'format-drift',\n match: ({ spans }) => {\n const judge = spans.find(\n (s) =>\n s.kind === 'judge' &&\n (s as Extract<Span, { kind: 'judge' }>).dimension === 'format' &&\n (s as Extract<Span, { kind: 'judge' }>).score < 0.5,\n )\n return judge\n ? {\n failureClass: 'format_drift',\n reason: 'format judge scored below 0.5',\n triggerSpanId: judge.spanId,\n }\n : null\n },\n },\n]\n\nfunction hasResolutionStatus(payload: Record<string, unknown>, status: string): boolean {\n if (status === 'missing_connection' && stringArray(payload.missingConnections).length > 0)\n return true\n return resolutionItems(payload).some((item) => item.status === status)\n}\n\nfunction hasMissingScopes(payload: Record<string, unknown>): boolean {\n if (stringArray(payload.missingScopes).length > 0) return true\n return resolutionItems(payload).some(\n (item) => Array.isArray(item.missingScopes) && item.missingScopes.length > 0,\n )\n}\n\nfunction resolutionItems(payload: Record<string, unknown>): Array<Record<string, unknown>> {\n return [\n ...records(payload.missing),\n ...records(payload.optionalMissing),\n ...records(payload.ready),\n ]\n}\n\nfunction records(value: unknown): Array<Record<string, unknown>> {\n if (!Array.isArray(value)) return []\n return value.filter(\n (item): item is Record<string, unknown> =>\n Boolean(item) && typeof item === 'object' && !Array.isArray(item),\n )\n}\n\nfunction stringArray(value: unknown): string[] {\n return Array.isArray(value)\n ? value.filter((item): item is string => typeof item === 'string')\n : []\n}\n\n/** Classify the failure mode of a run using an ordered rule list. */\nexport function classifyFailure(\n ctx: FailureContext,\n rules: FailureRule[] = DEFAULT_RULES,\n): FailureClassification {\n if (ctx.run.outcome?.pass !== false && ctx.run.status === 'completed') {\n return { failureClass: 'success', reason: 'run completed with pass=true (or no explicit fail)' }\n }\n for (const rule of rules) {\n const hit = rule.match(ctx)\n if (hit) return hit\n }\n return { failureClass: 'unknown', reason: 'no rule matched; run failed for unclassified reason' }\n}\n","/**\n * BudgetBreachView — aggregates breach events across the corpus.\n *\n * Answers: which dimensions get hit most often? Which scenarios are\n * underbudgeted? Which variants trigger the most breaches?\n */\n\nimport type { BudgetSpec } from '../trace/schema'\nimport type { TraceStore } from '../trace/store'\n\nexport interface BudgetBreachFinding {\n runId: string\n scenarioId: string\n variantId?: string\n dimension: keyof BudgetSpec\n limit: number\n consumed: number\n excessRatio: number\n timestamp: number\n}\n\nexport interface BudgetBreachReport {\n findings: BudgetBreachFinding[]\n byDimension: Record<string, number>\n byScenario: Record<string, number>\n byVariant: Record<string, number>\n totalRuns: number\n breachedRunRatio: number\n}\n\nexport async function budgetBreachView(\n store: TraceStore,\n options: { scenarioId?: string; variantId?: string } = {},\n): Promise<BudgetBreachReport> {\n const runs = await store.listRuns({\n scenarioId: options.scenarioId,\n variantId: options.variantId,\n })\n const findings: BudgetBreachFinding[] = []\n const byDimension: Record<string, number> = {}\n const byScenario: Record<string, number> = {}\n const byVariant: Record<string, number> = {}\n\n for (const run of runs) {\n const entries = await store.budget(run.runId)\n for (const e of entries) {\n if (!e.breached) continue\n const excessRatio = e.limit > 0 ? e.consumed / e.limit : Infinity\n findings.push({\n runId: run.runId,\n scenarioId: run.scenarioId,\n variantId: run.variantId,\n dimension: e.dimension,\n limit: e.limit,\n consumed: e.consumed,\n excessRatio,\n timestamp: e.timestamp,\n })\n byDimension[e.dimension] = (byDimension[e.dimension] ?? 0) + 1\n byScenario[run.scenarioId] = (byScenario[run.scenarioId] ?? 0) + 1\n if (run.variantId) byVariant[run.variantId] = (byVariant[run.variantId] ?? 0) + 1\n }\n }\n\n const breachedRuns = new Set(findings.map((f) => f.runId))\n return {\n findings,\n byDimension,\n byScenario,\n byVariant,\n totalRuns: runs.length,\n breachedRunRatio: runs.length > 0 ? breachedRuns.size / runs.length : 0,\n }\n}\n","/**\n * FailureClusterView — groups failed runs by (failureClass, triggerTool,\n * argHash-prefix) so weekly reviews can prioritize the top-N clusters.\n *\n * Each cluster includes: N runs, scenarios affected, representative\n * error message, a proposed mitigation hint (rule → action table).\n */\n\nimport { classifyFailure, DEFAULT_RULES, type FailureRule } from '../failure-taxonomy'\nimport { argHash, hasCapturedToolArgs, toolSpans } from '../trace/query'\nimport type { FailureClass, Span } from '../trace/schema'\nimport type { TraceStore } from '../trace/store'\n\nexport interface FailureCluster {\n failureClass: FailureClass\n /** Tool name when the trigger was a tool span, else undefined. */\n toolName?: string\n /** First 16 chars of argHash — clusters similar args. */\n argPrefix?: string\n /**\n * Source dimension when the trigger was a judge span (e.g. `'format'`,\n * `'safety'`, `'correctness'`). Lets cross-template aggregators\n * group failures by the dimension that fired without overloading\n * `argPrefix`. Optional — clusters without this field deserialize cleanly.\n */\n dimension?: string\n runCount: number\n scenarioIds: string[]\n exampleError?: string\n exampleRunId: string\n}\n\nexport interface FailureClusterReport {\n clusters: FailureCluster[]\n totalFailures: number\n totalRuns: number\n}\n\nexport async function failureClusterView(\n store: TraceStore,\n options: { rules?: FailureRule[]; minClusterSize?: number } = {},\n): Promise<FailureClusterReport> {\n const rules = options.rules ?? DEFAULT_RULES\n const minSize = options.minClusterSize ?? 1\n const runs = await store.listRuns()\n\n type Key = string\n const clusters = new Map<Key, FailureCluster>()\n let totalFailures = 0\n\n for (const run of runs) {\n if (run.status === 'completed' && run.outcome?.pass !== false) continue\n totalFailures++\n const spans = await store.spans({ runId: run.runId })\n const events = await store.events({ runId: run.runId })\n const cls = classifyFailure({ run, spans, events }, rules)\n\n let toolName: string | undefined\n let argPrefix: string | undefined\n let dimension: string | undefined\n if (cls.triggerSpanId) {\n const trig = spans.find((s) => s.spanId === cls.triggerSpanId)\n if (trig?.kind === 'tool') {\n toolName = trig.toolName\n if (hasCapturedToolArgs(trig)) argPrefix = argHash(trig.args).slice(0, 16)\n } else if (trig?.kind === 'judge') {\n dimension = trig.dimension\n }\n }\n // Fallback: look at the last errored tool span\n if (!toolName) {\n const ts = await toolSpans(store, run.runId)\n const errored = ts.filter((t) => t.status === 'error').pop()\n if (errored) {\n toolName = errored.toolName\n if (hasCapturedToolArgs(errored)) argPrefix = argHash(errored.args).slice(0, 16)\n }\n }\n // Secondary signal: any judge span on the failed run carries a\n // dimension. Useful when the rule classified by judge score but\n // didn't surface the trigger span (or surfaced a non-judge span).\n if (!dimension) {\n const judge = spans.find((s) => s.kind === 'judge' && typeof s.dimension === 'string')\n if (judge?.kind === 'judge') dimension = judge.dimension\n }\n\n const key = `${cls.failureClass}|${toolName ?? ''}|${argPrefix ?? ''}|${dimension ?? ''}`\n let cluster = clusters.get(key)\n if (!cluster) {\n cluster = {\n failureClass: cls.failureClass,\n toolName,\n argPrefix,\n dimension,\n runCount: 0,\n scenarioIds: [],\n exampleRunId: run.runId,\n exampleError: firstErrorMessage(spans) ?? cls.reason,\n }\n clusters.set(key, cluster)\n }\n cluster.runCount++\n if (!cluster.scenarioIds.includes(run.scenarioId)) cluster.scenarioIds.push(run.scenarioId)\n }\n\n const arr = [...clusters.values()]\n .filter((c) => c.runCount >= minSize)\n .sort((a, b) => b.runCount - a.runCount)\n\n return { clusters: arr, totalFailures, totalRuns: runs.length }\n}\n\nfunction firstErrorMessage(spans: Span[]): string | undefined {\n const errored = spans.find((s) => s.status === 'error')\n return errored?.error\n}\n","/**\n * JudgeAgreementView — pairwise agreement between judges across the\n * corpus, grouped by dimension.\n *\n * Output drives two workflows:\n * - Judge robustness audit: \"does Claude agree with GPT at κ ≥ 0.6?\"\n * - Calibration tracking: κ vs golden human labels over time (by\n * providing a `humanGoldenJudgeId`).\n */\n\nimport { interRaterReliability, pearsonR } from '../statistics'\nimport type { JudgeSpan } from '../trace/schema'\nimport type { TraceStore } from '../trace/store'\n\nexport interface JudgePair {\n judgeA: string\n judgeB: string\n dimension: string\n /** Number of (targetSpanId, dimension) tuples both judges scored. */\n commonItems: number\n pearson: number\n krippendorff: number\n}\n\nexport interface JudgeAgreementReport {\n pairs: JudgePair[]\n dimensions: string[]\n judgeIds: string[]\n}\n\nexport async function judgeAgreementView(store: TraceStore): Promise<JudgeAgreementReport> {\n const all = (await store.spans({ kind: 'judge' })).filter(\n (s): s is JudgeSpan => s.kind === 'judge',\n )\n if (all.length === 0) return { pairs: [], dimensions: [], judgeIds: [] }\n\n const byDimension = new Map<string, JudgeSpan[]>()\n for (const s of all) {\n const arr = byDimension.get(s.dimension) ?? []\n arr.push(s)\n byDimension.set(s.dimension, arr)\n }\n\n const judgeIds = [...new Set(all.map((s) => s.judgeId))].sort()\n const pairs: JudgePair[] = []\n for (const [dim, spans] of byDimension) {\n const byJudge = new Map<string, Map<string, number>>()\n for (const s of spans) {\n const m = byJudge.get(s.judgeId) ?? new Map<string, number>()\n m.set(s.targetSpanId, s.score)\n byJudge.set(s.judgeId, m)\n }\n const judgesHere = [...byJudge.keys()]\n for (let i = 0; i < judgesHere.length; i++) {\n for (let j = i + 1; j < judgesHere.length; j++) {\n const judgeI = judgesHere[i]!\n const judgeJ = judgesHere[j]!\n const a = byJudge.get(judgeI)!\n const b = byJudge.get(judgeJ)!\n const common: Array<[number, number]> = []\n for (const [target, scoreA] of a) {\n const scoreB = b.get(target)\n if (scoreB !== undefined) common.push([scoreA, scoreB])\n }\n if (common.length < 2) continue\n const judgeScores = common.map(\n ([scoreA, scoreB]) =>\n [\n { judgeName: judgeI, dimension: dim, score: scoreA, reasoning: '' },\n { judgeName: judgeJ, dimension: dim, score: scoreB, reasoning: '' },\n ] as const,\n )\n const k = interRaterReliability(\n judgeScores[0]!.map((_, k2) => judgeScores.map((pair) => pair[k2]!)),\n )\n pairs.push({\n judgeA: judgeI,\n judgeB: judgeJ,\n dimension: dim,\n commonItems: common.length,\n pearson: pearsonR(\n common.map((c) => c[0]),\n common.map((c) => c[1]),\n ),\n krippendorff: k,\n })\n }\n }\n }\n\n return {\n pairs: pairs.sort((a, b) => b.commonItems - a.commonItems),\n dimensions: [...byDimension.keys()].sort(),\n judgeIds,\n }\n}\n","/**\n * Tool-use metrics — derived purely from trace data.\n *\n * No scoring assumptions: consumers supply optional ground-truth tool\n * selections per turn + optional \"information used downstream\" signals.\n * Without those, we still compute descriptive metrics (error rate,\n * retry rate, duplicate-call rate) that are useful on their own.\n */\n\nimport { argHash, groupBy, hasCapturedToolArgs, toolSpans } from './trace/query'\nimport type { Span } from './trace/schema'\nimport type { TraceStore } from './trace/store'\n\nexport interface ToolUseMetrics {\n runId: string\n totalCalls: number\n /** Calls whose arguments were captured and can be compared for duplication. */\n callsWithCapturedArgs: number\n byTool: Record<string, ToolStats>\n errorRate: number\n /** Ratio of captured-argument calls already seen with the same tool name and arguments. */\n duplicateRate: number\n /** Ratio of error calls followed by ≥1 retry on same tool. */\n retryRate: number\n /** Optional: of the calls agent made, fraction the evaluator marked as \"correct selection\". */\n selectionAccuracy?: number\n}\n\nexport interface ToolStats {\n calls: number\n callsWithCapturedArgs: number\n errors: number\n avgLatencyMs: number\n duplicates: number\n}\n\nexport interface ToolUseOptions {\n /** Map of spanId → whether the evaluator judged the tool selection correct. Optional. */\n selectionLabels?: Record<string, boolean>\n}\n\nexport async function computeToolUseMetrics(\n store: TraceStore,\n runId: string,\n options: ToolUseOptions = {},\n): Promise<ToolUseMetrics> {\n const tools = await toolSpans(store, runId)\n if (tools.length === 0) {\n return {\n runId,\n totalCalls: 0,\n callsWithCapturedArgs: 0,\n byTool: {},\n errorRate: 0,\n duplicateRate: 0,\n retryRate: 0,\n }\n }\n\n const byTool: Record<string, ToolStats> = {}\n let totalErrors = 0\n let totalDuplicates = 0\n let callsWithCapturedArgs = 0\n const sortedTools = [...tools].sort((a, b) => a.startedAt - b.startedAt)\n const seenSignatures = new Set<string>()\n\n // duplicate detection + per-tool aggregation\n for (const t of sortedTools) {\n byTool[t.toolName] ??= {\n calls: 0,\n callsWithCapturedArgs: 0,\n errors: 0,\n avgLatencyMs: 0,\n duplicates: 0,\n }\n const stat = byTool[t.toolName]!\n stat.calls += 1\n if (t.status === 'error') {\n stat.errors += 1\n totalErrors += 1\n }\n if (typeof t.latencyMs === 'number') stat.avgLatencyMs += t.latencyMs\n if (hasCapturedToolArgs(t)) {\n callsWithCapturedArgs += 1\n stat.callsWithCapturedArgs += 1\n const sig = `${t.toolName}|${argHash(t.args)}`\n if (seenSignatures.has(sig)) {\n stat.duplicates += 1\n totalDuplicates += 1\n }\n seenSignatures.add(sig)\n }\n }\n\n for (const stat of Object.values(byTool)) {\n stat.avgLatencyMs = stat.calls > 0 ? stat.avgLatencyMs / stat.calls : 0\n }\n\n // retry detection: per-tool chronological adjacency where error → next same-tool call\n let retryOpportunities = 0\n let retriesFollowed = 0\n for (const [, arr] of groupBy(sortedTools, (t) => t.toolName)) {\n for (let i = 0; i < arr.length; i++) {\n if (arr[i]!.status !== 'error') continue\n retryOpportunities += 1\n if (arr[i + 1]) retriesFollowed += 1\n }\n }\n const retryRate = retryOpportunities > 0 ? retriesFollowed / retryOpportunities : 0\n\n let selectionAccuracy: number | undefined\n if (options.selectionLabels) {\n const labeled = sortedTools.filter((t) => t.spanId in options.selectionLabels!)\n if (labeled.length > 0) {\n selectionAccuracy =\n labeled.filter((t) => options.selectionLabels![t.spanId]).length / labeled.length\n }\n }\n\n return {\n runId,\n totalCalls: sortedTools.length,\n callsWithCapturedArgs,\n byTool,\n errorRate: totalErrors / sortedTools.length,\n duplicateRate: callsWithCapturedArgs > 0 ? totalDuplicates / callsWithCapturedArgs : 0,\n retryRate,\n selectionAccuracy,\n }\n}\n\nexport type { Span }\n","/**\n * ToolWasteView — fraction of tool calls whose results weren't used\n * downstream. Without a \"used\" signal we fall back to structural\n * proxies: error calls, duplicate calls, and tool calls followed by\n * zero subsequent LLM spans are all considered waste.\n *\n * Consumers can pass a `usageOracle` that inspects a tool span and\n * returns true iff the tool's result appears in a later LLM message,\n * artifact, or state mutation — that's the canonical definition; the\n * default heuristic is a reasonable fallback.\n */\n\nimport { computeToolUseMetrics } from '../tool-use-metrics'\nimport { llmSpans, toolSpans } from '../trace/query'\nimport type { LlmSpan, ToolSpan } from '../trace/schema'\nimport type { TraceStore } from '../trace/store'\n\nexport interface ToolWasteFinding {\n runId: string\n wastedCalls: number\n totalCalls: number\n wasteRate: number\n}\n\nexport interface ToolWasteReport {\n byRun: ToolWasteFinding[]\n overallWasteRate: number\n}\n\nexport interface ToolWasteOptions {\n runId?: string\n usageOracle?: (tool: ToolSpan, later: { llm: Awaited<ReturnType<typeof llmSpans>> }) => boolean\n}\n\nexport async function toolWasteView(\n store: TraceStore,\n options: ToolWasteOptions = {},\n): Promise<ToolWasteReport> {\n const runs = options.runId ? [options.runId] : (await store.listRuns()).map((r) => r.runId)\n\n const byRun: ToolWasteFinding[] = []\n let totalCalls = 0\n let totalWasted = 0\n for (const runId of runs) {\n const tools = await toolSpans(store, runId)\n if (tools.length === 0) {\n byRun.push({ runId, wastedCalls: 0, totalCalls: 0, wasteRate: 0 })\n continue\n }\n const llms = await llmSpans(store, runId)\n // Sort LLM spans once by start time, then build a suffix index of the\n // concatenated message text. `suffixText[i]` is the haystack of every\n // string message content in spans[i..]. Per tool we binary-search the\n // first span started strictly after the tool, then test that one suffix\n // — turning the per-tool O(llms × messages × content) scan into a single\n // O(log llms) lookup over precomputed text.\n const sortedLlm = [...llms].sort((a, b) => a.startedAt - b.startedAt)\n const startTimes = sortedLlm.map((l) => l.startedAt)\n const suffixText = buildSuffixText(sortedLlm)\n let wasted = 0\n for (const t of tools) {\n if (t.status === 'error') {\n wasted++\n continue\n }\n // First LLM span started strictly after this tool (upper-bound search).\n const cutoff = upperBound(startTimes, t.startedAt)\n if (options.usageOracle) {\n if (!options.usageOracle(t, { llm: sortedLlm.slice(cutoff) })) wasted++\n } else {\n // Default heuristic: a tool whose result is NOT mentioned in any\n // later LLM input message is likely wasted. An empty/null result has\n // no payload to propagate downstream — there is nothing to find in a\n // later message, so it is not evidence of waste; skip it.\n const resultStr = stringify(t.result)\n if (resultStr === '') continue\n const haystack = suffixText[cutoff] ?? ''\n const used = haystack.includes(resultStr.slice(0, 120))\n if (!used) wasted++\n }\n }\n const wasteRate = wasted / tools.length\n byRun.push({ runId, wastedCalls: wasted, totalCalls: tools.length, wasteRate })\n totalCalls += tools.length\n totalWasted += wasted\n }\n return { byRun, overallWasteRate: totalCalls > 0 ? totalWasted / totalCalls : 0 }\n}\n\n/**\n * Build per-position suffix haystacks: result[i] is the concatenation of every\n * string message content in spans[i..end]. Built back-to-front so each entry\n * reuses the next one — O(total message text) rather than O(spans²).\n */\nfunction buildSuffixText(spans: LlmSpan[]): string[] {\n const result = new Array<string>(spans.length + 1)\n result[spans.length] = ''\n for (let i = spans.length - 1; i >= 0; i--) {\n const own = spans[i]!.messages.map((m) =>\n typeof m.content === 'string' ? m.content : '',\n ).join('\\n')\n result[i] = `${own}\\n${result[i + 1]}`\n }\n return result\n}\n\n/** Index of the first element strictly greater than `target` in a sorted array. */\nfunction upperBound(sorted: number[], target: number): number {\n let lo = 0\n let hi = sorted.length\n while (lo < hi) {\n const mid = (lo + hi) >>> 1\n if (sorted[mid]! <= target) lo = mid + 1\n else hi = mid\n }\n return lo\n}\n\nfunction stringify(v: unknown): string {\n if (v === null || v === undefined) return ''\n if (typeof v === 'string') return v\n try {\n return JSON.stringify(v)\n } catch {\n return String(v)\n }\n}\n\n// Re-export for convenience in consumers that want both descriptive and usage metrics.\nexport { computeToolUseMetrics }\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;AAyBA,SAAgB,sBAAsB,aAAqC;CACzE,IAAI,YAAY,SAAS,GAAG,OAAO;CAGnC,MAAM,+BAAe,IAAI,IAAwB;CACjD,KAAK,IAAI,aAAa,GAAG,aAAa,YAAY,QAAQ,cACxD,KAAK,MAAM,KAAK,YAAY,aAAc;EACxC,IAAI,UAAU,aAAa,IAAI,EAAE,SAAS;EAC1C,IAAI,YAAY,KAAA,GAAW;GACzB,UAAU,MAAM,KAAK,EAAE,QAAQ,YAAY,OAAO,SAAS,CAAC,CAAa;GACzE,aAAa,IAAI,EAAE,WAAW,OAAO;EACvC;EACA,QAAQ,WAAW,CAAE,KAAK,EAAE,KAAK;CACnC;CAGF,MAAM,YAAsB,CAAC;CAC7B,MAAM,YAAsB,CAAC;CAE7B,KAAK,MAAM,CAAC,WAAW,YAAY,cAAc;EAC/C,MAAM,UAAU,QAAQ,QAAQ,WAAW,OAAO,SAAS,CAAC;EAC5D,IAAI,QAAQ,SAAS,GAAG;EACxB,MAAM,YAAY,QAAQ,EAAE,CAAE;EAC9B,IAAI,QAAQ,MAAM,WAAW,OAAO,WAAW,SAAS,GACtD,MAAM,IAAI,gBACR,qCAAqC,UAAU,yBAC1C,QAAQ,KAAK,WAAW,OAAO,MAAM,CAAC,CAAC,KAAK,GAAG,EAAE,kCACxD;EAEF,KAAK,IAAI,OAAO,GAAG,OAAO,WAAW,QAAQ;GAC3C,MAAM,UAAU,QAAQ,KAAK,WAAW,OAAO,KAAM;GACrD,KAAK,MAAM,KAAK,SAAS,UAAU,KAAK,CAAC;GACzC,KAAK,IAAI,IAAI,GAAG,IAAI,QAAQ,QAAQ,KAClC,KAAK,IAAI,IAAI,IAAI,GAAG,IAAI,QAAQ,QAAQ,KACtC,UAAU,MAAM,QAAQ,KAAM,QAAQ,OAAQ,CAAC;EAGrD;CACF;CAEA,IAAI,UAAU,WAAW,KAAK,UAAU,SAAS,GAAG,OAAO;CAE3D,MAAM,uBAAuB,UAAU,QAAQ,GAAG,MAAM,IAAI,GAAG,CAAC,IAAI,UAAU;CAG9E,IAAI,uBAAuB;CAC3B,IAAI,gBAAgB;CACpB,KAAK,IAAI,IAAI,GAAG,IAAI,UAAU,QAAQ,KACpC,KAAK,IAAI,IAAI,IAAI,GAAG,IAAI,UAAU,QAAQ,KAAK;EAC7C,yBAAyB,UAAU,KAAM,UAAU,OAAQ;EAC3D;CACF;CAEF,uBAAuB,gBAAgB,IAAI,uBAAuB,gBAAgB;CAElF,IAAI,yBAAyB,GAAG,OAAO;CACvC,OAAO,IAAI,uBAAuB;AACpC;;;;;;;;;;;;;;;;;AAgFA,SAAgB,0BACd,SACA,OAA+B,CAAC,GACT;CACvB,IAAI,QAAQ,WAAW,GACrB,MAAM,IAAI,gBAAgB,sDAAsD;CAGlF,MAAM,6BAAa,IAAI,IAAY;CACnC,MAAM,2BAAW,IAAI,IAAY;CAEjC,MAAM,uBAAO,IAAI,IAA8C;CAE/D,KAAK,MAAM,KAAK,SAAS;EACvB,IAAI,CAAC,OAAO,SAAS,EAAE,KAAK,GAC1B,MAAM,IAAI,gBACR,yDAAyD,EAAE,OAAO,UAAU,EAAE,UAAU,QAAQ,EAAE,UAAU,EAC9G;EAEF,WAAW,IAAI,EAAE,SAAS;EAC1B,SAAS,IAAI,EAAE,SAAS;EACxB,MAAM,UAAU,KAAK,IAAI,EAAE,SAAS,qBAAK,IAAI,IAAiC;EAC9E,MAAM,SAAS,QAAQ,IAAI,EAAE,SAAS,qBAAK,IAAI,IAAoB;EACnE,IAAI,OAAO,IAAI,EAAE,MAAM,GACrB,MAAM,IAAI,gBACR,yDAAyD,EAAE,OAAO,UAAU,EAAE,UAAU,QAAQ,EAAE,UAAU,EAC9G;EAEF,OAAO,IAAI,EAAE,QAAQ,EAAE,KAAK;EAC5B,QAAQ,IAAI,EAAE,WAAW,MAAM;EAC/B,KAAK,IAAI,EAAE,WAAW,OAAO;CAC/B;CAEA,MAAM,aAAa,KAAK,cAAc,CAAC,GAAG,QAAQ,CAAC,CAAC,KAAK;CACzD,KAAK,MAAM,KAAK,YACd,IAAI,CAAC,SAAS,IAAI,CAAC,GACjB,MAAM,IAAI,gBACR,yCAAyC,EAAE,wCAC7C;CAGJ,MAAM,eAAe,KAAK,SAAS,CAAC,GAAG,KAAK,MAAM,IAAI,CAAC,GAAG,UAAU,CAAC,CAAC,KAAK;CAC3E,KAAK,MAAM,KAAK,cACd,IAAI,CAAC,WAAW,IAAI,CAAC,GACnB,MAAM,IAAI,gBACR,qCAAqC,EAAE,wCACzC;CAGJ,IAAI,aAAa,SAAS,GACxB,MAAM,IAAI,gBACR,kDAAkD,aAAa,QACjE;CAGF,MAAM,eAA8C,CAAC;CACrD,MAAM,OAAiB,CAAC;CACxB,MAAM,SAAmB,CAAC;CAE1B,KAAK,MAAM,OAAO,YAAY;EAC5B,MAAM,UAAU,KAAK,IAAI,GAAG;EAE5B,MAAM,kBAA0C,CAAC;EACjD,KAAK,MAAM,KAAK,cAEd,gBAAgB,KADN,QAAQ,IAAI,CACD,CAAC,EAAE,QAAQ;EAElC,MAAM,cAAc,aAAa,QAAQ,MAAM,gBAAgB,OAAO,CAAC;EACvE,IAAI,YAAY,SAAS,GACvB,MAAM,IAAI,gBACR,yCAAyC,IAAI,gCAAgC,YAAY,KAAK,IAAI,EAAE,YAAY,KAAK,UAAU,eAAe,EAAE,EAClJ;EAIF,IAAI,cAAkC;EACtC,KAAK,MAAM,KAAK,cAAc;GAC5B,MAAM,MAAM,IAAI,IAAI,QAAQ,IAAI,CAAC,CAAC,CAAE,KAAK,CAAC;GAC1C,IAAI,gBAAgB,MAClB,cAAc;QAGd,cAAc,IAAI,IAAI,CAAC,GAAGA,WAAI,CAAC,CAAC,QAAQ,MAAM,IAAI,IAAI,CAAC,CAAC,CAAC;EAE7D;EACA,MAAM,cAAc,CAAC,GAAI,+BAAe,IAAI,IAAY,CAAE,CAAC,CAAC,KAAK;EACjE,IAAI,YAAY,SAAS,GACvB,MAAM,IAAI,gBACR,yCAAyC,IAAI,QAAQ,YAAY,OAAO,wBAAwB,aAAa,OAAO,kBACtH;EAMF,MAAM,YAAY,oBAHS,YAAY,KAAK,WAC1C,aAAa,KAAK,MAAM,QAAQ,IAAI,CAAC,CAAC,CAAE,IAAI,MAAM,CAAE,CAEX,GAAG,IAAI;EAClD,aAAa,KAAK;GAChB,GAAG;GACH,WAAW;GACX,SAAS;GACT,UAAU,CAAC,GAAG,YAAY;EAC5B,CAAC;EACD,IAAI,OAAO,SAAS,UAAU,GAAG,GAAG,KAAK,KAAK,UAAU,GAAG;EAC3D,IAAI,OAAO,SAAS,UAAU,aAAa,GAAG,OAAO,KAAK,UAAU,aAAa;CACnF;CAEA,MAAM,QAAQ,OACZ,GAAG,WAAW,IAAI,MAAa,GAAG,QAAQ,GAAG,MAAM,IAAI,GAAG,CAAC,IAAI,GAAG;CACpE,OAAO;EACL;EACA,YAAY,KAAK,IAAI;EACrB,sBAAsB,KAAK,MAAM;EACjC,YAAY;EACZ,UAAU;CACZ;AACF;;;;;;;;;AAUA,SAAgB,yCACd,aACA,OAA+B,CAAC,GACT;CACvB,MAAM,UAA+B,CAAC;CACtC,MAAM,uBAAO,IAAI,IAAY;CAC7B,KAAK,MAAM,EAAE,QAAQ,YAAY,aAAa;EAC5C,IAAI,KAAK,IAAI,MAAM,GACjB,MAAM,IAAI,gBACR,+DAA+D,OAAO,EACxE;EAEF,KAAK,IAAI,MAAM;EACf,KAAK,MAAM,KAAK,QACd,QAAQ,KAAK;GACX;GACA,WAAW,EAAE;GACb,WAAW,EAAE;GACb,OAAO,EAAE;EACX,CAAC;CAEL;CACA,OAAO,0BAA0B,SAAS,IAAI;AAChD;;;AC9QA,MAAa,gBAA+B;CAE1C;EACE,IAAI;EACJ,QAAQ,EAAE,UAAU;GAClB,MAAM,KAAK,IAAI,SAAS;GACxB,IAAI,MAAM,OAAO,WACf,OAAO;IAAE,cAAc;IAAI,QAAQ;GAAsC;GAC3E,OAAO;EACT;CACF;CACA;EACE,IAAI;EACJ,QAAQ,EAAE,aAAa;GACrB,MAAM,QAAQ,OAAO,MAClB,MACC,EAAE,SAAS,YACX,EAAE,QAAQ,SAAS,sBACnB,EAAE,QAAQ,WAAW,KACzB;GACA,OAAO,QACH;IACE,cAAc;IACd,QAAQ;IACR,gBAAgB,MAAM;GACxB,IACA;EACN;CACF;CACA;EACE,IAAI;EACJ,QAAQ,EAAE,aAAa;GACrB,MAAM,QAAQ,OAAO,MAClB,MACC,EAAE,SAAS,aACT,EAAE,QAAQ,SAAS,oCAAoC,EAAE,QAAQ,UAAU,SAC1E,EAAE,QAAQ,SAAS,+BAClB,EAAE,QAAQ,SAAS,mBAC3B;GACA,OAAO,QACH;IACE,cAAc;IACd,QAAQ;IACR,gBAAgB,MAAM;GACxB,IACA;EACN;CACF;CACA;EACE,IAAI;EACJ,QAAQ,EAAE,aAAa;GACrB,MAAM,QAAQ,OAAO,MAClB,MACC,EAAE,SAAS,YACX,EAAE,QAAQ,SAAS,mCACnB,oBAAoB,EAAE,SAAS,oBAAoB,CACvD;GACA,OAAO,QACH;IACE,cAAc;IACd,QAAQ;IACR,gBAAgB,MAAM;GACxB,IACA;EACN;CACF;CACA;EACE,IAAI;EACJ,QAAQ,EAAE,aAAa;GACrB,MAAM,QAAQ,OAAO,MAClB,MACC,EAAE,SAAS,aACT,EAAE,QAAQ,SAAS,mCAAmC,iBAAiB,EAAE,OAAO,KAC/E,EAAE,QAAQ,SAAS,+BAA+B,EAAE,QAAQ,SAAS,eAC5E;GACA,OAAO,QACH;IACE,cAAc;IACd,QAAQ;IACR,gBAAgB,MAAM;GACxB,IACA;EACN;CACF;CACA;EACE,IAAI;EACJ,QAAQ,EAAE,aAAa;GACrB,MAAM,QAAQ,OAAO,MAClB,MACC,EAAE,SAAS,aACT,EAAE,QAAQ,SAAS,wBAAwB,EAAE,QAAQ,WAAW,uBAC/D,EAAE,QAAQ,SAAS,+BAClB,EAAE,QAAQ,SAAS,uBACrB,EAAE,QAAQ,SAAS,gCACzB;GACA,OAAO,QACH;IACE,cAAc;IACd,QAAQ;IACR,gBAAgB,MAAM;GACxB,IACA;EACN;CACF;CACA;EACE,IAAI;EACJ,QAAQ,EAAE,aAAa;GACrB,MAAM,QAAQ,OAAO,MAClB,MACC,EAAE,SAAS,YACX,EAAE,QAAQ,SAAS,gCAClB,EAAE,QAAQ,SAAS,kBAClB,EAAE,QAAQ,SAAS,2BACnB,EAAE,QAAQ,SAAS,wBACnB,EAAE,QAAQ,WAAW,UAC3B;GACA,OAAO,QACH;IACE,cAAc;IACd,QAAQ;IACR,gBAAgB,MAAM;GACxB,IACA;EACN;CACF;CACA;EACE,IAAI;EACJ,QAAQ,EAAE,aAAa;GACrB,MAAM,QAAQ,OAAO,MAClB,MACC,EAAE,SAAS,YACX,EAAE,QAAQ,SAAS,gCAClB,EAAE,QAAQ,SAAS,yBAClB,EAAE,QAAQ,SAAS,mBACnB,EAAE,QAAQ,SAAS,gBACzB;GACA,OAAO,QACH;IACE,cAAc;IACd,QAAQ;IACR,gBAAgB,MAAM;GACxB,IACA;EACN;CACF;CACA;EACE,IAAI;EACJ,QAAQ,EAAE,aAAa;GACrB,MAAM,QAAQ,OAAO,MAClB,MACC,EAAE,SAAS,YACX,EAAE,QAAQ,SAAS,+BACnB,CAAC;IACC;IACA;IACA;IACA;IACA;IACA;IACA;IACA;IACA;GACF,CAAC,CAAC,SAAS,OAAO,EAAE,QAAQ,IAAI,CAAC,CACrC;GACA,OAAO,QACH;IACE,cAAc;IACd,QAAQ;IACR,gBAAgB,MAAM;GACxB,IACA;EACN;CACF;CACA;EACE,IAAI;EACJ,QAAQ,EAAE,aAAa;GACrB,MAAM,QAAQ,OAAO,MAClB,MACC,EAAE,SAAS,YACX,EAAE,QAAQ,SAAS,mBACnB,EAAE,QAAQ,aAAa,sBAC3B;GACA,OAAO,QACH;IACE,cAAc;IACd,QAAQ;IACR,gBAAgB,MAAM;GACxB,IACA;EACN;CACF;CACA;EACE,IAAI;EACJ,QAAQ,EAAE,KAAK,YAAY;GACzB,IAAI,IAAI,SAAS,SAAS,OAAO,OAAO;GACxC,MAAM,YAAY,MAAM,MACrB,MACC,EAAE,SAAS,gBAAgB,EAAE,KAAK,WAAW,KAAK,EAAE,KAAK,OAAO,QAAQ,IAAI,SAAS,CAAC,EAC1F;GACA,OAAO,YACH;IACE,cAAc;IACd,QAAQ;IACR,eAAe,UAAU;GAC3B,IACA;EACN;CACF;CACA;EACE,IAAI;EACJ,QAAQ,EAAE,aAAa;GACrB,MAAM,QAAQ,OAAO,MAClB,MACC,EAAE,SAAS,YACX,EAAE,QAAQ,SAAS,mBACnB,EAAE,QAAQ,WAAW,uBACzB;GACA,OAAO,QACH;IACE,cAAc;IACd,QAAQ;IACR,gBAAgB,MAAM;GACxB,IACA;EACN;CACF;CACA;EACE,IAAI;EACJ,QAAQ,EAAE,aAAa;GACrB,MAAM,QAAQ,OAAO,MAClB,MACC,EAAE,SAAS,YACX,EAAE,QAAQ,SAAS,mBACnB,EAAE,QAAQ,WAAW,wBACzB;GACA,OAAO,QACH;IACE,cAAc;IACd,QAAQ;IACR,gBAAgB,MAAM;GACxB,IACA;EACN;CACF;CAEA;EACE,IAAI;EACJ,QAAQ,EAAE,aAAa;GACrB,MAAM,SAAS,OAAO,MAAM,MAAM,EAAE,SAAS,eAAe;GAC5D,OAAO,SACH;IACE,cAAc;IACd,QAAQ,sBAAsB,OAAO,QAAQ,aAAa;IAC1D,gBAAgB,OAAO;GACzB,IACA;EACN;CACF;CAEA;EACE,IAAI;EACJ,QAAQ,EAAE,aAAa;GACrB,MAAM,IAAI,OAAO,MAAM,MAAM,EAAE,SAAS,kBAAkB;GAC1D,OAAO,IACH;IACE,cAAc;IACd,QAAQ;IACR,gBAAgB,EAAE;GACpB,IACA;EACN;CACF;CAEA;EACE,IAAI;EACJ,QAAQ,EAAE,YAAY;GACpB,MAAM,IAAI,MAAM,MACb,MAAM,EAAE,SAAS,aAAa,OAAO,EAAE,aAAa,YAAY,EAAE,aAAa,CAClF;GACA,IAAI,CAAC,GAAG,OAAO;GACf,OAAO;IACL,cAAc;IACd,QAAQ,kBAAmB,EAAyC;IACpE,eAAe,EAAE;GACnB;EACF;CACF;CAEA;EACE,IAAI;EACJ,QAAQ,EAAE,KAAK,aAAa;GAC1B,IAAI,IAAI,WAAW,WAAW,OAAO;GACrC,MAAM,aAAa,OAAO,MACvB,MACC,EAAE,SAAS,WACX,OAAO,EAAE,QAAQ,UAAU,EAAE,CAAC,CAC3B,YAAY,CAAC,CACb,SAAS,SAAS,CACzB;GACA,MAAM,QAAQ,IAAI,SAAS,SAAS,GAAA,CAAI,YAAY;GACpD,IAAI,cAAc,KAAK,SAAS,SAAS,KAAK,KAAK,SAAS,UAAU,GACpE,OAAO;IAAE,cAAc;IAAW,QAAQ;GAA0B;GAEtE,OAAO;EACT;CACF;CAEA;EACE,IAAI;EACJ,QAAQ,EAAE,YAAY;GACpB,MAAM,QAAQ,MAAM,QAAQ,MAAM,EAAE,SAAS,MAAM;GACnD,MAAM,yBAAS,IAAI,IAAoB;GACvC,KAAK,MAAM,KAAK,OAAO;IACrB,MAAM,OAAQ,EAAsC;IACpD,MAAM,MAAM,OAAO,IAAI,IAAI,KAAK,CAAC;IACjC,IAAI,KAAK,CAAC;IACV,OAAO,IAAI,MAAM,GAAG;GACtB;GACA,KAAK,MAAM,CAAC,MAAM,QAAQ,QAAQ;IAChC,MAAM,OAAO,IAAI,QAAQ,MAAM,EAAE,WAAW,OAAO;IACnD,IAAI,KAAK,UAAU,KAAK,KAAK,WAAW,IAAI,QAC1C,OAAO;KACL,cAAc;KACd,QAAQ,GAAG,KAAK,OAAO,+BAA+B,KAAK;KAC3D,eAAe,KAAK,KAAK,SAAS,EAAE,CAAE;IACxC;GAEJ;GACA,OAAO;EACT;CACF;CAEA;EACE,IAAI;EACJ,QAAQ,EAAE,KAAK,YAAY;GACzB,IAAI,IAAI,SAAS,SAAS,OAAO,OAAO;GACxC,MAAM,oBAAoB,MAAM,MAC7B,MACC,EAAE,SAAS,WACV,EAAE,YAAY,mBAA0C,KAAA,KACxD,EAAE,YAAY,iBAA4B,CAC/C;GACA,MAAM,QAAQ,MAAM,QAAQ,MAAM,EAAE,SAAS,MAAM;GACnD,IAAI,qBAAqB,MAAM,WAAW,GACxC,OAAO;IACL,cAAc;IACd,QAAQ;GACV;GAEF,OAAO;EACT;CACF;CAEA;EACE,IAAI;EACJ,QAAQ,EAAE,YAAY;GACpB,MAAM,QAAQ,MAAM,MACjB,MACC,EAAE,SAAS,WACV,EAAuC,cAAc,YACrD,EAAuC,QAAQ,EACpD;GACA,OAAO,QACH;IACE,cAAc;IACd,QAAQ;IACR,eAAe,MAAM;GACvB,IACA;EACN;CACF;AACF;AAEA,SAAS,oBAAoB,SAAkC,QAAyB;CACtF,IAAI,WAAW,wBAAwB,YAAY,QAAQ,kBAAkB,CAAC,CAAC,SAAS,GACtF,OAAO;CACT,OAAO,gBAAgB,OAAO,CAAC,CAAC,MAAM,SAAS,KAAK,WAAW,MAAM;AACvE;AAEA,SAAS,iBAAiB,SAA2C;CACnE,IAAI,YAAY,QAAQ,aAAa,CAAC,CAAC,SAAS,GAAG,OAAO;CAC1D,OAAO,gBAAgB,OAAO,CAAC,CAAC,MAC7B,SAAS,MAAM,QAAQ,KAAK,aAAa,KAAK,KAAK,cAAc,SAAS,CAC7E;AACF;AAEA,SAAS,gBAAgB,SAAkE;CACzF,OAAO;EACL,GAAG,QAAQ,QAAQ,OAAO;EAC1B,GAAG,QAAQ,QAAQ,eAAe;EAClC,GAAG,QAAQ,QAAQ,KAAK;CAC1B;AACF;AAEA,SAAS,QAAQ,OAAgD;CAC/D,IAAI,CAAC,MAAM,QAAQ,KAAK,GAAG,OAAO,CAAC;CACnC,OAAO,MAAM,QACV,SACC,QAAQ,IAAI,KAAK,OAAO,SAAS,YAAY,CAAC,MAAM,QAAQ,IAAI,CACpE;AACF;AAEA,SAAS,YAAY,OAA0B;CAC7C,OAAO,MAAM,QAAQ,KAAK,IACtB,MAAM,QAAQ,SAAyB,OAAO,SAAS,QAAQ,IAC/D,CAAC;AACP;;AAGA,SAAgB,gBACd,KACA,QAAuB,eACA;CACvB,IAAI,IAAI,IAAI,SAAS,SAAS,SAAS,IAAI,IAAI,WAAW,aACxD,OAAO;EAAE,cAAc;EAAW,QAAQ;CAAqD;CAEjG,KAAK,MAAM,QAAQ,OAAO;EACxB,MAAM,MAAM,KAAK,MAAM,GAAG;EAC1B,IAAI,KAAK,OAAO;CAClB;CACA,OAAO;EAAE,cAAc;EAAW,QAAQ;CAAsD;AAClG;;;AC/aA,eAAsB,iBACpB,OACA,UAAuD,CAAC,GAC3B;CAC7B,MAAM,OAAO,MAAM,MAAM,SAAS;EAChC,YAAY,QAAQ;EACpB,WAAW,QAAQ;CACrB,CAAC;CACD,MAAM,WAAkC,CAAC;CACzC,MAAM,cAAsC,CAAC;CAC7C,MAAM,aAAqC,CAAC;CAC5C,MAAM,YAAoC,CAAC;CAE3C,KAAK,MAAM,OAAO,MAAM;EACtB,MAAM,UAAU,MAAM,MAAM,OAAO,IAAI,KAAK;EAC5C,KAAK,MAAM,KAAK,SAAS;GACvB,IAAI,CAAC,EAAE,UAAU;GACjB,MAAM,cAAc,EAAE,QAAQ,IAAI,EAAE,WAAW,EAAE,QAAQ;GACzD,SAAS,KAAK;IACZ,OAAO,IAAI;IACX,YAAY,IAAI;IAChB,WAAW,IAAI;IACf,WAAW,EAAE;IACb,OAAO,EAAE;IACT,UAAU,EAAE;IACZ;IACA,WAAW,EAAE;GACf,CAAC;GACD,YAAY,EAAE,cAAc,YAAY,EAAE,cAAc,KAAK;GAC7D,WAAW,IAAI,eAAe,WAAW,IAAI,eAAe,KAAK;GACjE,IAAI,IAAI,WAAW,UAAU,IAAI,cAAc,UAAU,IAAI,cAAc,KAAK;EAClF;CACF;CAEA,MAAM,eAAe,IAAI,IAAI,SAAS,KAAK,MAAM,EAAE,KAAK,CAAC;CACzD,OAAO;EACL;EACA;EACA;EACA;EACA,WAAW,KAAK;EAChB,kBAAkB,KAAK,SAAS,IAAI,aAAa,OAAO,KAAK,SAAS;CACxE;AACF;;;;;;;;;;ACnCA,eAAsB,mBACpB,OACA,UAA8D,CAAC,GAChC;CAC/B,MAAM,QAAQ,QAAQ,SAAS;CAC/B,MAAM,UAAU,QAAQ,kBAAkB;CAC1C,MAAM,OAAO,MAAM,MAAM,SAAS;CAGlC,MAAM,2BAAW,IAAI,IAAyB;CAC9C,IAAI,gBAAgB;CAEpB,KAAK,MAAM,OAAO,MAAM;EACtB,IAAI,IAAI,WAAW,eAAe,IAAI,SAAS,SAAS,OAAO;EAC/D;EACA,MAAM,QAAQ,MAAM,MAAM,MAAM,EAAE,OAAO,IAAI,MAAM,CAAC;EAEpD,MAAM,MAAM,gBAAgB;GAAE;GAAK;GAAO,QAAA,MADrB,MAAM,OAAO,EAAE,OAAO,IAAI,MAAM,CAAC;EACL,GAAG,KAAK;EAEzD,IAAI;EACJ,IAAI;EACJ,IAAI;EACJ,IAAI,IAAI,eAAe;GACrB,MAAM,OAAO,MAAM,MAAM,MAAM,EAAE,WAAW,IAAI,aAAa;GAC7D,IAAI,MAAM,SAAS,QAAQ;IACzB,WAAW,KAAK;IAChB,IAAI,oBAAoB,IAAI,GAAG,YAAY,QAAQ,KAAK,IAAI,CAAC,CAAC,MAAM,GAAG,EAAE;GAC3E,OAAO,IAAI,MAAM,SAAS,SACxB,YAAY,KAAK;EAErB;EAEA,IAAI,CAAC,UAAU;GAEb,MAAM,WAAU,MADC,UAAU,OAAO,IAAI,KAAK,EAAA,CACxB,QAAQ,MAAM,EAAE,WAAW,OAAO,CAAC,CAAC,IAAI;GAC3D,IAAI,SAAS;IACX,WAAW,QAAQ;IACnB,IAAI,oBAAoB,OAAO,GAAG,YAAY,QAAQ,QAAQ,IAAI,CAAC,CAAC,MAAM,GAAG,EAAE;GACjF;EACF;EAIA,IAAI,CAAC,WAAW;GACd,MAAM,QAAQ,MAAM,MAAM,MAAM,EAAE,SAAS,WAAW,OAAO,EAAE,cAAc,QAAQ;GACrF,IAAI,OAAO,SAAS,SAAS,YAAY,MAAM;EACjD;EAEA,MAAM,MAAM,GAAG,IAAI,aAAa,GAAG,YAAY,GAAG,GAAG,aAAa,GAAG,GAAG,aAAa;EACrF,IAAI,UAAU,SAAS,IAAI,GAAG;EAC9B,IAAI,CAAC,SAAS;GACZ,UAAU;IACR,cAAc,IAAI;IAClB;IACA;IACA;IACA,UAAU;IACV,aAAa,CAAC;IACd,cAAc,IAAI;IAClB,cAAc,kBAAkB,KAAK,KAAK,IAAI;GAChD;GACA,SAAS,IAAI,KAAK,OAAO;EAC3B;EACA,QAAQ;EACR,IAAI,CAAC,QAAQ,YAAY,SAAS,IAAI,UAAU,GAAG,QAAQ,YAAY,KAAK,IAAI,UAAU;CAC5F;CAMA,OAAO;EAAE,UAJG,CAAC,GAAG,SAAS,OAAO,CAAC,CAAC,CAC/B,QAAQ,MAAM,EAAE,YAAY,OAAO,CAAC,CACpC,MAAM,GAAG,MAAM,EAAE,WAAW,EAAE,QAEZ;EAAG;EAAe,WAAW,KAAK;CAAO;AAChE;AAEA,SAAS,kBAAkB,OAAmC;CAE5D,OADgB,MAAM,MAAM,MAAM,EAAE,WAAW,OAClC,CAAC,EAAE;AAClB;;;;;;;;;;;;ACrFA,eAAsB,mBAAmB,OAAkD;CACzF,MAAM,OAAO,MAAM,MAAM,MAAM,EAAE,MAAM,QAAQ,CAAC,EAAA,CAAG,QAChD,MAAsB,EAAE,SAAS,OACpC;CACA,IAAI,IAAI,WAAW,GAAG,OAAO;EAAE,OAAO,CAAC;EAAG,YAAY,CAAC;EAAG,UAAU,CAAC;CAAE;CAEvE,MAAM,8BAAc,IAAI,IAAyB;CACjD,KAAK,MAAM,KAAK,KAAK;EACnB,MAAM,MAAM,YAAY,IAAI,EAAE,SAAS,KAAK,CAAC;EAC7C,IAAI,KAAK,CAAC;EACV,YAAY,IAAI,EAAE,WAAW,GAAG;CAClC;CAEA,MAAM,WAAW,CAAC,GAAG,IAAI,IAAI,IAAI,KAAK,MAAM,EAAE,OAAO,CAAC,CAAC,CAAC,CAAC,KAAK;CAC9D,MAAM,QAAqB,CAAC;CAC5B,KAAK,MAAM,CAAC,KAAK,UAAU,aAAa;EACtC,MAAM,0BAAU,IAAI,IAAiC;EACrD,KAAK,MAAM,KAAK,OAAO;GACrB,MAAM,IAAI,QAAQ,IAAI,EAAE,OAAO,qBAAK,IAAI,IAAoB;GAC5D,EAAE,IAAI,EAAE,cAAc,EAAE,KAAK;GAC7B,QAAQ,IAAI,EAAE,SAAS,CAAC;EAC1B;EACA,MAAM,aAAa,CAAC,GAAG,QAAQ,KAAK,CAAC;EACrC,KAAK,IAAI,IAAI,GAAG,IAAI,WAAW,QAAQ,KACrC,KAAK,IAAI,IAAI,IAAI,GAAG,IAAI,WAAW,QAAQ,KAAK;GAC9C,MAAM,SAAS,WAAW;GAC1B,MAAM,SAAS,WAAW;GAC1B,MAAM,IAAI,QAAQ,IAAI,MAAM;GAC5B,MAAM,IAAI,QAAQ,IAAI,MAAM;GAC5B,MAAM,SAAkC,CAAC;GACzC,KAAK,MAAM,CAAC,QAAQ,WAAW,GAAG;IAChC,MAAM,SAAS,EAAE,IAAI,MAAM;IAC3B,IAAI,WAAW,KAAA,GAAW,OAAO,KAAK,CAAC,QAAQ,MAAM,CAAC;GACxD;GACA,IAAI,OAAO,SAAS,GAAG;GACvB,MAAM,cAAc,OAAO,KACxB,CAAC,QAAQ,YACR,CACE;IAAE,WAAW;IAAQ,WAAW;IAAK,OAAO;IAAQ,WAAW;GAAG,GAClE;IAAE,WAAW;IAAQ,WAAW;IAAK,OAAO;IAAQ,WAAW;GAAG,CACpE,CACJ;GACA,MAAM,IAAI,sBACR,YAAY,EAAE,CAAE,KAAK,GAAG,OAAO,YAAY,KAAK,SAAS,KAAK,GAAI,CAAC,CACrE;GACA,MAAM,KAAK;IACT,QAAQ;IACR,QAAQ;IACR,WAAW;IACX,aAAa,OAAO;IACpB,SAAS,SACP,OAAO,KAAK,MAAM,EAAE,EAAE,GACtB,OAAO,KAAK,MAAM,EAAE,EAAE,CACxB;IACA,cAAc;GAChB,CAAC;EACH;CAEJ;CAEA,OAAO;EACL,OAAO,MAAM,MAAM,GAAG,MAAM,EAAE,cAAc,EAAE,WAAW;EACzD,YAAY,CAAC,GAAG,YAAY,KAAK,CAAC,CAAC,CAAC,KAAK;EACzC;CACF;AACF;;;;;;;;;;;ACtDA,eAAsB,sBACpB,OACA,OACA,UAA0B,CAAC,GACF;CACzB,MAAM,QAAQ,MAAM,UAAU,OAAO,KAAK;CAC1C,IAAI,MAAM,WAAW,GACnB,OAAO;EACL;EACA,YAAY;EACZ,uBAAuB;EACvB,QAAQ,CAAC;EACT,WAAW;EACX,eAAe;EACf,WAAW;CACb;CAGF,MAAM,SAAoC,CAAC;CAC3C,IAAI,cAAc;CAClB,IAAI,kBAAkB;CACtB,IAAI,wBAAwB;CAC5B,MAAM,cAAc,CAAC,GAAG,KAAK,CAAC,CAAC,MAAM,GAAG,MAAM,EAAE,YAAY,EAAE,SAAS;CACvE,MAAM,iCAAiB,IAAI,IAAY;CAGvC,KAAK,MAAM,KAAK,aAAa;EAC3B,OAAO,EAAE,cAAc;GACrB,OAAO;GACP,uBAAuB;GACvB,QAAQ;GACR,cAAc;GACd,YAAY;EACd;EACA,MAAM,OAAO,OAAO,EAAE;EACtB,KAAK,SAAS;EACd,IAAI,EAAE,WAAW,SAAS;GACxB,KAAK,UAAU;GACf,eAAe;EACjB;EACA,IAAI,OAAO,EAAE,cAAc,UAAU,KAAK,gBAAgB,EAAE;EAC5D,IAAI,oBAAoB,CAAC,GAAG;GAC1B,yBAAyB;GACzB,KAAK,yBAAyB;GAC9B,MAAM,MAAM,GAAG,EAAE,SAAS,GAAG,QAAQ,EAAE,IAAI;GAC3C,IAAI,eAAe,IAAI,GAAG,GAAG;IAC3B,KAAK,cAAc;IACnB,mBAAmB;GACrB;GACA,eAAe,IAAI,GAAG;EACxB;CACF;CAEA,KAAK,MAAM,QAAQ,OAAO,OAAO,MAAM,GACrC,KAAK,eAAe,KAAK,QAAQ,IAAI,KAAK,eAAe,KAAK,QAAQ;CAIxE,IAAI,qBAAqB;CACzB,IAAI,kBAAkB;CACtB,KAAK,MAAM,GAAG,QAAQ,QAAQ,cAAc,MAAM,EAAE,QAAQ,GAC1D,KAAK,IAAI,IAAI,GAAG,IAAI,IAAI,QAAQ,KAAK;EACnC,IAAI,IAAI,EAAE,CAAE,WAAW,SAAS;EAChC,sBAAsB;EACtB,IAAI,IAAI,IAAI,IAAI,mBAAmB;CACrC;CAEF,MAAM,YAAY,qBAAqB,IAAI,kBAAkB,qBAAqB;CAElF,IAAI;CACJ,IAAI,QAAQ,iBAAiB;EAC3B,MAAM,UAAU,YAAY,QAAQ,MAAM,EAAE,UAAU,QAAQ,eAAgB;EAC9E,IAAI,QAAQ,SAAS,GACnB,oBACE,QAAQ,QAAQ,MAAM,QAAQ,gBAAiB,EAAE,OAAO,CAAC,CAAC,SAAS,QAAQ;CAEjF;CAEA,OAAO;EACL;EACA,YAAY,YAAY;EACxB;EACA;EACA,WAAW,cAAc,YAAY;EACrC,eAAe,wBAAwB,IAAI,kBAAkB,wBAAwB;EACrF;EACA;CACF;AACF;;;;;;;;;;;;;;AC/FA,eAAsB,cACpB,OACA,UAA4B,CAAC,GACH;CAC1B,MAAM,OAAO,QAAQ,QAAQ,CAAC,QAAQ,KAAK,KAAK,MAAM,MAAM,SAAS,EAAA,CAAG,KAAK,MAAM,EAAE,KAAK;CAE1F,MAAM,QAA4B,CAAC;CACnC,IAAI,aAAa;CACjB,IAAI,cAAc;CAClB,KAAK,MAAM,SAAS,MAAM;EACxB,MAAM,QAAQ,MAAM,UAAU,OAAO,KAAK;EAC1C,IAAI,MAAM,WAAW,GAAG;GACtB,MAAM,KAAK;IAAE;IAAO,aAAa;IAAG,YAAY;IAAG,WAAW;GAAE,CAAC;GACjE;EACF;EAQA,MAAM,YAAY,CAAC,GAAG,MAPH,SAAS,OAAO,KAAK,CAOd,CAAC,CAAC,MAAM,GAAG,MAAM,EAAE,YAAY,EAAE,SAAS;EACpE,MAAM,aAAa,UAAU,KAAK,MAAM,EAAE,SAAS;EACnD,MAAM,aAAa,gBAAgB,SAAS;EAC5C,IAAI,SAAS;EACb,KAAK,MAAM,KAAK,OAAO;GACrB,IAAI,EAAE,WAAW,SAAS;IACxB;IACA;GACF;GAEA,MAAM,SAAS,WAAW,YAAY,EAAE,SAAS;GACjD,IAAI,QAAQ,aACN;QAAA,CAAC,QAAQ,YAAY,GAAG,EAAE,KAAK,UAAU,MAAM,MAAM,EAAE,CAAC,GAAG;GAAA,OAC1D;IAKL,MAAM,YAAY,UAAU,EAAE,MAAM;IACpC,IAAI,cAAc,IAAI;IAGtB,IAAI,EAFa,WAAW,WAAW,GAAA,CACjB,SAAS,UAAU,MAAM,GAAG,GAAG,CAC7C,GAAG;GACb;EACF;EACA,MAAM,YAAY,SAAS,MAAM;EACjC,MAAM,KAAK;GAAE;GAAO,aAAa;GAAQ,YAAY,MAAM;GAAQ;EAAU,CAAC;EAC9E,cAAc,MAAM;EACpB,eAAe;CACjB;CACA,OAAO;EAAE;EAAO,kBAAkB,aAAa,IAAI,cAAc,aAAa;CAAE;AAClF;;;;;;AAOA,SAAS,gBAAgB,OAA4B;CACnD,MAAM,SAAS,IAAI,MAAc,MAAM,SAAS,CAAC;CACjD,OAAO,MAAM,UAAU;CACvB,KAAK,IAAI,IAAI,MAAM,SAAS,GAAG,KAAK,GAAG,KAIrC,OAAO,KAAK,GAHA,MAAM,EAAE,CAAE,SAAS,KAAK,MAClC,OAAO,EAAE,YAAY,WAAW,EAAE,UAAU,EAC9C,CAAC,CAAC,KAAK,IACU,EAAE,IAAI,OAAO,IAAI;CAEpC,OAAO;AACT;;AAGA,SAAS,WAAW,QAAkB,QAAwB;CAC5D,IAAI,KAAK;CACT,IAAI,KAAK,OAAO;CAChB,OAAO,KAAK,IAAI;EACd,MAAM,MAAO,KAAK,OAAQ;EAC1B,IAAI,OAAO,QAAS,QAAQ,KAAK,MAAM;OAClC,KAAK;CACZ;CACA,OAAO;AACT;AAEA,SAAS,UAAU,GAAoB;CACrC,IAAI,MAAM,QAAQ,MAAM,KAAA,GAAW,OAAO;CAC1C,IAAI,OAAO,MAAM,UAAU,OAAO;CAClC,IAAI;EACF,OAAO,KAAK,UAAU,CAAC;CACzB,QAAQ;EACN,OAAO,OAAO,CAAC;CACjB;AACF"}
|