@tangle-network/agent-eval 0.172.0 → 0.173.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +49 -0
- package/README.md +17 -2
- package/dist/adapters/http.d.ts +2 -2
- package/dist/analyst/index.d.ts +13 -13
- package/dist/analyst/index.js +7 -6
- package/dist/analyst/index.js.map +1 -1
- package/dist/{attestation-CJBGmMVh.d.ts → attestation-c1QvaBdX.d.ts} +2 -2
- package/dist/{attestation-CJBGmMVh.d.ts.map → attestation-c1QvaBdX.d.ts.map} +1 -1
- package/dist/{backend-integrity-e79K3UPD.d.ts → backend-integrity-CeuTgqsd.d.ts} +3 -4
- package/dist/backend-integrity-CeuTgqsd.d.ts.map +1 -0
- package/dist/{benchmark-h-h4bfqj.d.ts → benchmark-BjLGkfnN.d.ts} +3 -3
- package/dist/{benchmark-h-h4bfqj.d.ts.map → benchmark-BjLGkfnN.d.ts.map} +1 -1
- package/dist/{benchmark-command-DoFcisuM.js → benchmark-command-9S20PRel.js} +9 -9
- package/dist/{benchmark-command-DoFcisuM.js.map → benchmark-command-9S20PRel.js.map} +1 -1
- package/dist/benchmarks/index.d.ts +5 -5
- package/dist/benchmarks/index.js +3 -3
- package/dist/{bounded-process-CVOC_D3H.js → bounded-process-BBZob7vl.js} +64 -10
- package/dist/bounded-process-BBZob7vl.js.map +1 -0
- package/dist/builder-eval/index.d.ts +3 -3
- package/dist/builder-eval/index.js +2 -2
- package/dist/campaign/index.d.ts +9 -9
- package/dist/campaign/index.js +6 -6
- package/dist/{campaign-Dp35pBbS.js → campaign-pxS0wmo4.js} +8 -8
- package/dist/{campaign-Dp35pBbS.js.map → campaign-pxS0wmo4.js.map} +1 -1
- package/dist/chat-client-Db4bqYfA.js +115 -0
- package/dist/chat-client-Db4bqYfA.js.map +1 -0
- package/dist/{chat-json-call-5Jxna-aV.js → chat-json-call-C26igCih.js} +16 -6
- package/dist/chat-json-call-C26igCih.js.map +1 -0
- package/dist/cli.js +31 -17
- package/dist/cli.js.map +1 -1
- package/dist/{client-BvwNkIRN.js → client-BlLY6o2w.js} +2 -2
- package/dist/{client-BvwNkIRN.js.map → client-BlLY6o2w.js.map} +1 -1
- package/dist/{client-Df7wdslk.d.ts → client-DlqdbM7n.d.ts} +4 -4
- package/dist/{client-Df7wdslk.d.ts.map → client-DlqdbM7n.d.ts.map} +1 -1
- package/dist/contract/index.d.ts +13 -13
- package/dist/contract/index.js +202 -10
- package/dist/contract/index.js.map +1 -1
- package/dist/{counterfactual-Bee5_BIn.d.ts → counterfactual-CLgrwhkY.d.ts} +4 -4
- package/dist/{counterfactual-Bee5_BIn.d.ts.map → counterfactual-CLgrwhkY.d.ts.map} +1 -1
- package/dist/{chat-client-DI79OPye.js → default-registry-B0bKikCb.js} +3 -39
- package/dist/default-registry-B0bKikCb.js.map +1 -0
- package/dist/{default-registry-XxedTLwu.d.ts → default-registry-BKwc8bN5.d.ts} +6 -6
- package/dist/{default-registry-XxedTLwu.d.ts.map → default-registry-BKwc8bN5.d.ts.map} +1 -1
- package/dist/{define-agent-eval-0wW7gFhr.d.ts → define-agent-eval-CY6qdlGV.d.ts} +6 -6
- package/dist/{define-agent-eval-0wW7gFhr.d.ts.map → define-agent-eval-CY6qdlGV.d.ts.map} +1 -1
- package/dist/{define-agent-eval-jS8xj_Q_.js → define-agent-eval-D_i_s69h.js} +7 -7
- package/dist/{define-agent-eval-jS8xj_Q_.js.map → define-agent-eval-D_i_s69h.js.map} +1 -1
- package/dist/{dspy-rlm-engine-CS3qcCEk.js → dspy-rlm-engine-D5byiHn9.js} +5 -6
- package/dist/dspy-rlm-engine-D5byiHn9.js.map +1 -0
- package/dist/{emitter-Bvnu0VzL.d.ts → emitter-Cs0egaFd.d.ts} +3 -3
- package/dist/{emitter-Bvnu0VzL.d.ts.map → emitter-Cs0egaFd.d.ts.map} +1 -1
- package/dist/{engine-BfRay1qD.d.ts → engine-DhFir3Ys.d.ts} +23 -8
- package/dist/{engine-BfRay1qD.d.ts.map → engine-DhFir3Ys.d.ts.map} +1 -1
- package/dist/{eval-campaign-JDTeE6Pl.js → eval-campaign-BeAjdhzC.js} +2 -2
- package/dist/{eval-campaign-JDTeE6Pl.js.map → eval-campaign-BeAjdhzC.js.map} +1 -1
- package/dist/{exact-types-BEecmnWm.d.ts → exact-types-BKOEILRP.d.ts} +2 -2
- package/dist/{exact-types-BEecmnWm.d.ts.map → exact-types-BKOEILRP.d.ts.map} +1 -1
- package/dist/experiment/index.d.ts +4 -4
- package/dist/{external-optimizer-contracts-szBJ_1vh.d.ts → external-optimizer-contracts-CQCpyrIL.d.ts} +2 -2
- package/dist/{external-optimizer-contracts-szBJ_1vh.d.ts.map → external-optimizer-contracts-CQCpyrIL.d.ts.map} +1 -1
- package/dist/{external-optimizer-process-CQxylYeG.js → external-optimizer-process-Cq_Pg15r.js} +2 -2
- package/dist/{external-optimizer-process-CQxylYeG.js.map → external-optimizer-process-Cq_Pg15r.js.map} +1 -1
- package/dist/{external-optimizer-subprocess-Cex8Da2i.js → external-optimizer-subprocess-DgNebftP.js} +2 -2
- package/dist/{external-optimizer-subprocess-Cex8Da2i.js.map → external-optimizer-subprocess-DgNebftP.js.map} +1 -1
- package/dist/failure-cluster-OldNRoAt.d.ts +154 -0
- package/dist/failure-cluster-OldNRoAt.d.ts.map +1 -0
- package/dist/{feedback-trajectory-DIqpCyF0.d.ts → feedback-trajectory-CMnv_uYs.d.ts} +6 -6
- package/dist/{feedback-trajectory-DIqpCyF0.d.ts.map → feedback-trajectory-CMnv_uYs.d.ts.map} +1 -1
- package/dist/{heldout-gate-JgNRDZwZ.d.ts → heldout-gate-Df5hsqmm.d.ts} +7 -7
- package/dist/{heldout-gate-JgNRDZwZ.d.ts.map → heldout-gate-Df5hsqmm.d.ts.map} +1 -1
- package/dist/hosted/index.d.ts +2 -2
- package/dist/hosted/index.js +1 -1
- package/dist/{index-DnglhM0A.d.ts → index-BQqOjerE.d.ts} +94 -13
- package/dist/index-BQqOjerE.d.ts.map +1 -0
- package/dist/{index-_vPrVMRX.d.ts → index-CFDffsKz.d.ts} +11 -11
- package/dist/{index-_vPrVMRX.d.ts.map → index-CFDffsKz.d.ts.map} +1 -1
- package/dist/{index-DDAPhUJJ.d.ts → index-D0Db5X-4.d.ts} +45 -9
- package/dist/index-D0Db5X-4.d.ts.map +1 -0
- package/dist/{index-DMoxLG8P.d.ts → index-e7LXeRVa.d.ts} +3 -3
- package/dist/{index-DMoxLG8P.d.ts.map → index-e7LXeRVa.d.ts.map} +1 -1
- package/dist/index.d.ts +172 -46
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +64 -41
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-08F022xN.d.ts → insight-report-DETqPc_A.d.ts} +4 -4
- package/dist/{insight-report-08F022xN.d.ts.map → insight-report-DETqPc_A.d.ts.map} +1 -1
- package/dist/{integrity-B_EDELom.d.ts → integrity-BKTcA-HP.d.ts} +3 -3
- package/dist/{integrity-B_EDELom.d.ts.map → integrity-BKTcA-HP.d.ts.map} +1 -1
- package/dist/{kind-factory-DMeEoMQZ.js → kind-factory-gP6lDySe.js} +164 -149
- package/dist/kind-factory-gP6lDySe.js.map +1 -0
- package/dist/{llm-client-BFMRpmqb.js → llm-client-CxQtdtd6.js} +12 -5
- package/dist/llm-client-CxQtdtd6.js.map +1 -0
- package/dist/{llm-judge-aQHIk5_-.js → llm-judge-B2YxbAJb.js} +81 -7
- package/dist/{llm-judge-aQHIk5_-.js.map → llm-judge-B2YxbAJb.js.map} +1 -1
- package/dist/{matrix-Ch8JO1pG.d.ts → matrix-DGu8KhSs.d.ts} +2 -2
- package/dist/{matrix-Ch8JO1pG.d.ts.map → matrix-DGu8KhSs.d.ts.map} +1 -1
- package/dist/meta-eval/index.d.ts +100 -5
- package/dist/meta-eval/index.d.ts.map +1 -1
- package/dist/meta-eval/index.js +200 -2
- package/dist/meta-eval/index.js.map +1 -1
- package/dist/{mint-DjfDUMHr.js → mint-vWOdD8Ae.js} +2 -2
- package/dist/{mint-DjfDUMHr.js.map → mint-vWOdD8Ae.js.map} +1 -1
- package/dist/multishot/golden/index.d.ts +1 -1
- package/dist/multishot/index.d.ts +2 -2
- package/dist/openapi.json +1 -1
- package/dist/pipelines/index.d.ts +5 -5
- package/dist/pipelines/index.js +3 -3
- package/dist/{pre-registration-DHz6P_6f.d.ts → pre-registration-BoI4ucR3.d.ts} +2 -2
- package/dist/{pre-registration-DHz6P_6f.d.ts.map → pre-registration-BoI4ucR3.d.ts.map} +1 -1
- package/dist/{produced-state-CxmbFxFd.js → produced-state-7VYDwtkk.js} +3 -3
- package/dist/{produced-state-CxmbFxFd.js.map → produced-state-7VYDwtkk.js.map} +1 -1
- package/dist/{promotion-policy-BBBcz5_3.d.ts → promotion-policy-CvMda3kU.d.ts} +2 -2
- package/dist/{promotion-policy-BBBcz5_3.d.ts.map → promotion-policy-CvMda3kU.d.ts.map} +1 -1
- package/dist/{provenance-Dp-vvyrU.d.ts → provenance-CRY67X50.d.ts} +39 -165
- package/dist/provenance-CRY67X50.d.ts.map +1 -0
- package/dist/{query-BPGMVlbM.js → query-D1nLIKt7.js} +2 -2
- package/dist/{query-BPGMVlbM.js.map → query-D1nLIKt7.js.map} +1 -1
- package/dist/{query-Na5gEIGd.d.ts → query-D6W6MaGx.d.ts} +3 -3
- package/dist/{query-Na5gEIGd.d.ts.map → query-D6W6MaGx.d.ts.map} +1 -1
- package/dist/{registry-xEb_xfns.d.ts → registry-7pOUBrtX.d.ts} +4 -4
- package/dist/{registry-xEb_xfns.d.ts.map → registry-7pOUBrtX.d.ts.map} +1 -1
- package/dist/{release-confidence-D6lQw_o7.d.ts → release-confidence-BAcNYOf1.d.ts} +4 -4
- package/dist/{release-confidence-D6lQw_o7.d.ts.map → release-confidence-BAcNYOf1.d.ts.map} +1 -1
- package/dist/{release-confidence-CzUHc4z4.js → release-confidence-BsGEg_xg.js} +3 -3
- package/dist/{release-confidence-CzUHc4z4.js.map → release-confidence-BsGEg_xg.js.map} +1 -1
- package/dist/reporting.d.ts +3 -3
- package/dist/reporting.js +1 -1
- package/dist/{researcher-CMUTQXD7.d.ts → researcher-jsW1X94L.d.ts} +7 -8
- package/dist/researcher-jsW1X94L.d.ts.map +1 -0
- package/dist/{reward-hacking-SkxYgT0x.js → reward-hacking-CKW4teig.js} +2 -2
- package/dist/{reward-hacking-SkxYgT0x.js.map → reward-hacking-CKW4teig.js.map} +1 -1
- package/dist/{reward-hacking-CgPRUesA.d.ts → reward-hacking-ZXEi9VCq.d.ts} +2 -2
- package/dist/{reward-hacking-CgPRUesA.d.ts.map → reward-hacking-ZXEi9VCq.d.ts.map} +1 -1
- package/dist/rl.d.ts +7 -7
- package/dist/rl.js +4 -4
- package/dist/rollout/index.d.ts +1 -1
- package/dist/rollout/index.js +2 -2
- package/dist/{rollout-Crypdx8s.js → rollout-C-znbbYg.js} +2 -2
- package/dist/{rollout-Crypdx8s.js.map → rollout-C-znbbYg.js.map} +1 -1
- package/dist/{rubric-predictive-validity-DluJLCKQ.d.ts → rubric-predictive-validity-Dl1dvKCv.d.ts} +2 -2
- package/dist/{rubric-predictive-validity-DluJLCKQ.d.ts.map → rubric-predictive-validity-Dl1dvKCv.d.ts.map} +1 -1
- package/dist/{run-record-DQjRcYwA.d.ts → run-record-DTv1MdjK.d.ts} +2 -2
- package/dist/{run-record-DQjRcYwA.d.ts.map → run-record-DTv1MdjK.d.ts.map} +1 -1
- package/dist/{run-record-DLORoL7t.js → run-record-ZIsR9Fif.js} +2 -2
- package/dist/{run-record-DLORoL7t.js.map → run-record-ZIsR9Fif.js.map} +1 -1
- package/dist/{schema-DID1Cqct.d.ts → schema-CR5cpjQ3.d.ts} +57 -2
- package/dist/{schema-DID1Cqct.d.ts.map → schema-CR5cpjQ3.d.ts.map} +1 -1
- package/dist/{schema-CdIX2aHu.js → schema-CSf6qWgZ.js} +35 -1
- package/dist/{schema-CdIX2aHu.js.map → schema-CSf6qWgZ.js.map} +1 -1
- package/dist/semantic-concept-judge-Ct3QU7t5.js +780 -0
- package/dist/semantic-concept-judge-Ct3QU7t5.js.map +1 -0
- package/dist/{series-convergence-D9WgpXGi.d.ts → series-convergence-DeG33RpC.d.ts} +2 -2
- package/dist/{series-convergence-D9WgpXGi.d.ts.map → series-convergence-DeG33RpC.d.ts.map} +1 -1
- package/dist/{server-CCEnywOR.js → server-BR6onwZB.js} +2 -2
- package/dist/{server-CCEnywOR.js.map → server-BR6onwZB.js.map} +1 -1
- package/dist/{skillopt-optimization-method-LHi02MzH.js → skillopt-optimization-method-BzdphODy.js} +6 -6
- package/dist/{skillopt-optimization-method-LHi02MzH.js.map → skillopt-optimization-method-BzdphODy.js.map} +1 -1
- package/dist/{statistical-heldout-Yldkntvy.d.ts → statistical-heldout-DTyB_6-1.d.ts} +3 -3
- package/dist/{statistical-heldout-Yldkntvy.d.ts.map → statistical-heldout-DTyB_6-1.d.ts.map} +1 -1
- package/dist/{store-Cq9oOrI1.d.ts → store-BErPvYBr.d.ts} +2 -2
- package/dist/{store-Cq9oOrI1.d.ts.map → store-BErPvYBr.d.ts.map} +1 -1
- package/dist/{store-otlp-CHjBvWQY.js → store-otlp-Dow0pk_5.js} +2 -2
- package/dist/{store-otlp-CHjBvWQY.js.map → store-otlp-Dow0pk_5.js.map} +1 -1
- package/dist/{store-tool-spans-BvdUbeOB.d.ts → store-tool-spans-CCZNsihA.d.ts} +8 -8
- package/dist/{store-tool-spans-BvdUbeOB.d.ts.map → store-tool-spans-CCZNsihA.d.ts.map} +1 -1
- package/dist/{store-tool-spans-B9o6tU8f.js → store-tool-spans-CeNj_m2L.js} +3 -3
- package/dist/{store-tool-spans-B9o6tU8f.js.map → store-tool-spans-CeNj_m2L.js.map} +1 -1
- package/dist/storyboard/index.d.ts +1 -1
- package/dist/{summary-report-DRstQNBX.d.ts → summary-report-gMrbYawB.d.ts} +3 -3
- package/dist/{summary-report-DRstQNBX.d.ts.map → summary-report-gMrbYawB.d.ts.map} +1 -1
- package/dist/{task-failure-attributes-CBGtLS_H.js → task-failure-attributes-CZjZeBsY.js} +3 -3
- package/dist/{task-failure-attributes-CBGtLS_H.js.map → task-failure-attributes-CZjZeBsY.js.map} +1 -1
- package/dist/{tool-groups-DjwlMBvW.d.ts → tool-groups-Cp4Xdzrp.d.ts} +3 -3
- package/dist/tool-groups-Cp4Xdzrp.d.ts.map +1 -0
- package/dist/{tool-waste-CwGHzBzX.js → tool-waste-B9tdWV6g.js} +253 -6
- package/dist/tool-waste-B9tdWV6g.js.map +1 -0
- package/dist/{tool-waste-BrmLKxMw.d.ts → tool-waste-D23I0zWm.d.ts} +4 -4
- package/dist/{tool-waste-BrmLKxMw.d.ts.map → tool-waste-D23I0zWm.d.ts.map} +1 -1
- package/dist/trace-repair/index.d.ts +2 -2
- package/dist/traces.d.ts +10 -10
- package/dist/traces.js +7 -7
- package/dist/{trajectory-r1bQqvBQ.d.ts → trajectory-D7qrNvaN.d.ts} +3 -3
- package/dist/{trajectory-r1bQqvBQ.d.ts.map → trajectory-D7qrNvaN.d.ts.map} +1 -1
- package/dist/trajectory-replay/index.d.ts +3 -3
- package/dist/{types-CCZ34qmV.d.ts → types-BDV4PiMR.d.ts} +3 -3
- package/dist/{types-CCZ34qmV.d.ts.map → types-BDV4PiMR.d.ts.map} +1 -1
- package/dist/{types-nokrtr7M.d.ts → types-Ba5UQyVD.d.ts} +4 -4
- package/dist/{types-nokrtr7M.d.ts.map → types-Ba5UQyVD.d.ts.map} +1 -1
- package/dist/{types-DMoNFDWi.d.ts → types-DN2WdT5S.d.ts} +3 -3
- package/dist/{types-DMoNFDWi.d.ts.map → types-DN2WdT5S.d.ts.map} +1 -1
- package/dist/types-gvRsyJLh.d.ts +831 -0
- package/dist/types-gvRsyJLh.d.ts.map +1 -0
- package/dist/wire/index.d.ts +3 -3
- package/dist/wire/index.js +1 -1
- package/docs/plants.md +69 -1
- package/docs/public-api.md +6 -4
- package/docs/trace-analysis.md +55 -8
- package/package.json +1 -1
- package/dist/backend-integrity-e79K3UPD.d.ts.map +0 -1
- package/dist/bounded-process-CVOC_D3H.js.map +0 -1
- package/dist/chat-client-DI79OPye.js.map +0 -1
- package/dist/chat-json-call-5Jxna-aV.js.map +0 -1
- package/dist/dspy-rlm-engine-CS3qcCEk.js.map +0 -1
- package/dist/failure-cluster-6YSvsKlp.d.ts +0 -58
- package/dist/failure-cluster-6YSvsKlp.d.ts.map +0 -1
- package/dist/index-DDAPhUJJ.d.ts.map +0 -1
- package/dist/index-DnglhM0A.d.ts.map +0 -1
- package/dist/kind-factory-DMeEoMQZ.js.map +0 -1
- package/dist/llm-client-BFMRpmqb.js.map +0 -1
- package/dist/provenance-Dp-vvyrU.d.ts.map +0 -1
- package/dist/raw-provider-sink-BU29Sh8h.d.ts +0 -134
- package/dist/raw-provider-sink-BU29Sh8h.d.ts.map +0 -1
- package/dist/researcher-CMUTQXD7.d.ts.map +0 -1
- package/dist/semantic-concept-judge-I36eejJx.js +0 -382
- package/dist/semantic-concept-judge-I36eejJx.js.map +0 -1
- package/dist/tool-groups-DjwlMBvW.d.ts.map +0 -1
- package/dist/tool-waste-CwGHzBzX.js.map +0 -1
- package/dist/types-Bfk0uxRj.d.ts +0 -443
- package/dist/types-Bfk0uxRj.d.ts.map +0 -1
package/CHANGELOG.md
CHANGED
|
@@ -4,6 +4,55 @@ All notable changes to `@tangle-network/agent-eval` and its sibling `agent-eval-
|
|
|
4
4
|
|
|
5
5
|
---
|
|
6
6
|
|
|
7
|
+
## [0.173.0] — 2026-09-01
|
|
8
|
+
|
|
9
|
+
### Added
|
|
10
|
+
|
|
11
|
+
- `createChatClient({ transport: 'openai-compatible' })` exposes the OpenAI-compatible HTTP transport the CLI already used, with `baseUrl` and one credential form as required arguments and no env fallback (#723). A consumer such as discovery-lab's two-vendor judge panel no longer hand-rolls a transport.
|
|
12
|
+
- `createChatTraceEngine`: a `TraceAnalysisEngine` that runs in Node over a caller-owned `ChatClient`, so `buildDefaultAnalystRegistry({ engine })` constructs and runs every default analyst without Python (#719).
|
|
13
|
+
- Failure taxonomy: `FailureBlame`, `FAILURE_BLAME`, `INFRA_FAILURE_BLAMES`, `classifyFailureReason` and `DEFAULT_FAILURE_REASON_RULES`; `FAILURE_CLASSES` widens from 35 to 52 and every `FailureClassification` carries a `blame` (#720).
|
|
14
|
+
- `runBoundedProcess` accepts `stdin`; `BoundedProcessResult` reports `spawned` and `stdinDelivered`; `localCommandRunner` runs on the same bounded runner (#721).
|
|
15
|
+
- `quotaExhaustedUntil`: a dated provider quota refusal is terminal for its seat until the stated instant, never transient (#725).
|
|
16
|
+
- Plants: `plantByPerturbation` and `perturbEvidence` author a reject-direction plant from a verified claim by changing one value (#726).
|
|
17
|
+
- `observeCodeAgentStore(scope)`: a run-scoped finder over a harness session store that attributes a session only by its own recorded place, and reports `unavailable` with no counts when nothing is attributable (#727).
|
|
18
|
+
|
|
19
|
+
### Fixed
|
|
20
|
+
|
|
21
|
+
- Output of a bounded process is decoded as one stream, so a multi-byte character straddling a chunk boundary no longer becomes U+FFFD (#721).
|
|
22
|
+
- The durable `selfImprove` provenance test is bounded by its measured cost, so `pnpm test` is green on `main` again (#722).
|
|
23
|
+
|
|
24
|
+
---
|
|
25
|
+
|
|
26
|
+
## [0.172.1] — 2026-09-01
|
|
27
|
+
|
|
28
|
+
### Added
|
|
29
|
+
|
|
30
|
+
- `runBoundedProcess` accepts `args`, an argument vector delivered to the program with no shell between.
|
|
31
|
+
The runner could only be given a command line for a shell to read, so `interpreter -flag <text>` was expressible only by quoting the text into that line.
|
|
32
|
+
A caller with text it did not author — a script body, a path, a pattern — therefore had to carry its own shell quoter, and a bug in that quoter is a command injection and not a wrong answer.
|
|
33
|
+
`{ command: 'bash', args: ['-n', '-c', body] }` now parses `body` whatever shell metacharacters it holds.
|
|
34
|
+
The argv form keeps the deadline, the detached process group, the group kill, the output cap, the `envMode` rule and every result flag.
|
|
35
|
+
`args` together with a truthy `shell` is contradictory, because a shell cannot interpret an argument vector: the call spawns nothing and reports `runnerError`, so the caller's bug reads as a failure and never as a pass.
|
|
36
|
+
The consumer this was added for is agent-knowledge's claim grader, which parses a claim's check under `bash -n -c` before it runs the check; it moves onto this runner in agent-knowledge 12.1.0 and holds no reference to it yet.
|
|
37
|
+
Two limits on `args` are the platform's, and are reported rather than thrown: a NUL byte in an entry cannot reach a process at all and comes back as a `runnerError`, and a non-string entry from an untypechecked caller is coerced with `String()`.
|
|
38
|
+
With `envMode: 'replace'` and no `PATH` named, the two forms do not search the same directories — the shell applies its own compiled-in default and the argv form falls back to `confstr(_CS_PATH)`, measured on Linux/glibc as `/usr/local/sbin:/usr/local/bin:/usr/sbin:/usr/bin:/sbin:/bin` against `/bin:/usr/bin`. Name `PATH` and both forms search what was named.
|
|
39
|
+
|
|
40
|
+
### Fixed
|
|
41
|
+
|
|
42
|
+
- `runBoundedProcess` no longer REJECTS when `spawn` refuses an argument.
|
|
43
|
+
`spawn` validates `command` and each `args` entry synchronously and throws — a NUL byte, or an `args` that is not an array — and a throw inside the promise executor rejected, against this module's standing guarantee that a call always resolves.
|
|
44
|
+
The throw is now caught and reported as a `runnerError` result, so a caller running text it did not author scores that text as failed instead of dying on it.
|
|
45
|
+
The same hole existed for `command` on the shell form and is closed with it.
|
|
46
|
+
|
|
47
|
+
### Changed
|
|
48
|
+
|
|
49
|
+
- This release is a patch. `args` is optional and purely additive, and the package is pre-1.0, so the compatibility boundary sits at the minor.
|
|
50
|
+
Every consumer still declares `>=0.171.0 <0.172.0` — agent-runtime, agent-knowledge and agent-app — so all three are already one cohort behind 0.172.0 and have to move to reach it at all.
|
|
51
|
+
A `0.173.0` would open a third cohort for a change none of them consume; a patch keeps it inside the minor they are moving to.
|
|
52
|
+
- Pin the four dependency manifests with digest `02565ef4ac091a2df7d995186bc2d952fc5f739d5ccad77be63e856d59b8fc07`. Three of the four carry the version, so the lock digest moves with every release.
|
|
53
|
+
|
|
54
|
+
---
|
|
55
|
+
|
|
7
56
|
## [0.172.0] — 2026-09-01
|
|
8
57
|
|
|
9
58
|
### Added
|
package/README.md
CHANGED
|
@@ -105,13 +105,15 @@ Every row is a function you call. Each links to a runnable example.
|
|
|
105
105
|
| [`deltaRepair()`](./docs/trace-repair-grader.md) — a finding must be graded by executing the repair it proposes | a trajectory, an analyst finding, a sandbox | the repair's measured effect against a no-fix control |
|
|
106
106
|
| [`replayVerify()`](./docs/trajectory-replay.md) — you must know whether a recorded failure still reproduces | a recorded shell trajectory and its pinned image | a re-execution verdict and the divergences found |
|
|
107
107
|
| [`analyzeSupervisorRun()`](./docs/adapters-observability.md) — a recursive or supervised run directory must be read | a run directory | counts that stay missing when a measurement is missing, never zero |
|
|
108
|
-
| [`seedPlants()` / `catchRate()`](./docs/plants.md) — you must know whether the grader catches a wrong answer, not only how the work scored | a grading set and items authored wrong by
|
|
108
|
+
| [`plantByPerturbation()` / `seedPlants()` / `catchRate()`](./docs/plants.md) — you must know whether the grader catches a wrong answer, not only how the work scored | a grading set, and claims the grader verified | items authored wrong by one value, a sealed manifest, then a catch rate that refuses rather than guessing |
|
|
109
109
|
| [`buildRlDataset()`](./examples/publish-rl-dataset/) — scored runs should become training data | run records and preferences | reward, preference, and supervised rows |
|
|
110
110
|
|
|
111
111
|
## Configure Model Calls
|
|
112
112
|
|
|
113
113
|
Benchmarks, user drivers, executors, built-in judges, completion checkers, and judge adapters all take the same `ChatClient`.
|
|
114
|
-
You own model execution
|
|
114
|
+
You own model execution, and Agent Eval never goes looking for a credential: it reads no environment variable to find one, and every transport is bound at the call site.
|
|
115
|
+
|
|
116
|
+
Bind a transport one of two ways. Pass a `chat` function you wrote:
|
|
115
117
|
|
|
116
118
|
```ts
|
|
117
119
|
import { createChatClient } from '@tangle-network/agent-eval'
|
|
@@ -124,6 +126,19 @@ const chat = createChatClient({
|
|
|
124
126
|
})
|
|
125
127
|
```
|
|
126
128
|
|
|
129
|
+
Or name an OpenAI-compatible endpoint and hand over a credential as values, and Agent Eval drives `POST {baseUrl}/chat/completions` for you:
|
|
130
|
+
|
|
131
|
+
```ts
|
|
132
|
+
const chat = createChatClient({
|
|
133
|
+
transport: 'openai-compatible',
|
|
134
|
+
baseUrl: 'https://router.example/v1', // ends at /v1; the path is ours to append
|
|
135
|
+
apiKey: process.env.MY_ROUTER_KEY, // or `bearer`, or `authHeader`
|
|
136
|
+
defaultModel: 'claude-sonnet-4-6',
|
|
137
|
+
})
|
|
138
|
+
```
|
|
139
|
+
|
|
140
|
+
Prefer the second over hand-rolling a fetch loop. It carries the retry, degrade, and — load-bearing — the `servedModel` echo that `assertServedModel` and `assertCrossFamilyServed` read; a transport that omits that field makes both checks report `unreported`, so the cross-vendor rules they enforce measure nothing. `baseUrl` and one credential form are required arguments with no default and no fallback: a half-configured client is refused at construction rather than reaching an endpoint you did not name.
|
|
141
|
+
|
|
127
142
|
On Agent Runtime, `profileChatClient({ profile, executor, context })` from `@tangle-network/agent-runtime/kernel` is that transport: every call runs one exact `AgentProfile` and reports its measured usage, retries, and served model identity.
|
|
128
143
|
Use `sandbox-sdk` for Sandbox and `mock` in tests.
|
|
129
144
|
A custom adapter must return a `ChatResponse` and declare `maximumAttempts` before a capped cost account can dispatch it.
|
package/dist/adapters/http.d.ts
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
|
-
import { R as Scenario, d as DispatchContext, f as DispatchFn } from "../types-
|
|
2
|
-
import "../index-
|
|
1
|
+
import { R as Scenario, d as DispatchContext, f as DispatchFn } from "../types-Ba5UQyVD.js";
|
|
2
|
+
import "../index-BQqOjerE.js";
|
|
3
3
|
//#region src/adapters/http.d.ts
|
|
4
4
|
interface HttpDispatchOptions<TScenario extends Scenario, _TArtifact> {
|
|
5
5
|
/** Static endpoint URL. Mutually exclusive with `resolveUrl`. */
|
package/dist/analyst/index.d.ts
CHANGED
|
@@ -1,14 +1,14 @@
|
|
|
1
1
|
import { b as CustomTokenPricing, c as CostLedgerHandle } from "../cost-ledger-DbQdN3nO.js";
|
|
2
|
-
import { I as TraceAnalystSpan, _ as ProposalFinding, a as AnalystInputKind, b as makeFinding, c as AnalystRunInputs, d as AnalystSeverity, f as AnalystUsageReceipt, g as ExecutionProbeRequest, h as ExecutionProbeOutcome, i as AnalystFinding, l as AnalystRunResult, m as ExecutionProbe, n as AnalystContext, o as AnalystRequirements, p as EvidenceRef, r as AnalystCost, s as AnalystRunEvent, t as Analyst, u as AnalystRunSummary, v as ProposalFindingOrigin, w as TraceAnalysisStore, x as makeProposalFinding, y as computeFindingId } from "../types-
|
|
3
|
-
import { _ as CreateChatClientOpts, b as
|
|
4
|
-
import { f as ExternalOptimizerModelExecutionObservation, h as ExternalOptimizerRunnerCommand, l as ExternalOptimizerModelCall } from "../external-optimizer-contracts-
|
|
5
|
-
import { a as createTraceAnalyst, c as BehavioralAnalystOptions, i as TraceAnalystDefinition, l as behavioralAnalyst, n as buildDefaultAnalystRegistry, o as renderPriorFindings, r as CreateTraceAnalystOptions, s as runTraceAnalyst, t as DefaultAnalystRegistryOptions, u as deriveEfficiencyFindings } from "../default-registry-
|
|
6
|
-
import { a as ExactAnalystRunPolicySnapshot, c as ExactAnalystSnapshot, d as ExactExecutionComponentSnapshot, i as ExactAnalystRunEvent, l as ExactCapableAnalyst, n as ExactAnalystExecutionPlanSnapshot, o as ExactAnalystRunResult, r as ExactAnalystRunCompletion, s as ExactAnalystRunSummary, t as ExactAnalystBudgetSnapshot, u as ExactExecutionComponentIdentity } from "../exact-types-
|
|
7
|
-
import { a as ExactAnalystBudgetPolicy, c as RegistryRunOpts, i as BudgetPolicy, n as AnalystRegistry, o as ExactAnalystRunExecutionError, r as AnalystRegistryOptions, s as ExactRegistryRunOpts, t as AnalystHooks } from "../registry-
|
|
8
|
-
import { a as
|
|
9
|
-
import { n as buildTraceToolsForGroup, t as TraceToolGroupName } from "../tool-groups-
|
|
10
|
-
import { C as DspyRlmTraceEngineOptions,
|
|
11
|
-
import { C as scoreAnalystFindings, S as traceStoreEvidenceResolver, _ as AnalystIssueExpectation, a as AnalystBenchmarkLabelState, b as registryBenchmarkRunner, c as AnalystBenchmarkProvenance, d as AnalystBenchmarkSummary, f as AnalystEvidenceExpectation, g as AnalystFindingScore, h as AnalystEvidenceResolver, i as AnalystBenchmarkError, l as AnalystBenchmarkResult, m as AnalystEvidenceResolutionError, n as AnalystBenchmarkDatasetRef, o as AnalystBenchmarkObservation, p as AnalystEvidenceResolution, r as AnalystBenchmarkDescriptor, s as AnalystBenchmarkOutput, t as AnalystBenchmarkCase, u as AnalystBenchmarkRunner, v as AnalystLatencyDistribution, x as runAnalystBenchmark, y as RunAnalystBenchmarkOptions } from "../benchmark-
|
|
2
|
+
import { I as TraceAnalystSpan, _ as ProposalFinding, a as AnalystInputKind, b as makeFinding, c as AnalystRunInputs, d as AnalystSeverity, f as AnalystUsageReceipt, g as ExecutionProbeRequest, h as ExecutionProbeOutcome, i as AnalystFinding, l as AnalystRunResult, m as ExecutionProbe, n as AnalystContext, o as AnalystRequirements, p as EvidenceRef, r as AnalystCost, s as AnalystRunEvent, t as Analyst, u as AnalystRunSummary, v as ProposalFindingOrigin, w as TraceAnalysisStore, x as makeProposalFinding, y as computeFindingId } from "../types-DN2WdT5S.js";
|
|
3
|
+
import { S as createChatClient, _ as CreateChatClientOpts, b as OpenAiCompatibleTransportOpts, f as ChatCallOpts, g as ChatTransport, h as ChatResponse, m as ChatRequest, p as ChatClient, v as CustomTransportOpts, x as SandboxSdkTransportOpts, y as MockTransportOpts } from "../types-gvRsyJLh.js";
|
|
4
|
+
import { f as ExternalOptimizerModelExecutionObservation, h as ExternalOptimizerRunnerCommand, l as ExternalOptimizerModelCall } from "../external-optimizer-contracts-CQCpyrIL.js";
|
|
5
|
+
import { a as createTraceAnalyst, c as BehavioralAnalystOptions, i as TraceAnalystDefinition, l as behavioralAnalyst, n as buildDefaultAnalystRegistry, o as renderPriorFindings, r as CreateTraceAnalystOptions, s as runTraceAnalyst, t as DefaultAnalystRegistryOptions, u as deriveEfficiencyFindings } from "../default-registry-BKwc8bN5.js";
|
|
6
|
+
import { a as ExactAnalystRunPolicySnapshot, c as ExactAnalystSnapshot, d as ExactExecutionComponentSnapshot, i as ExactAnalystRunEvent, l as ExactCapableAnalyst, n as ExactAnalystExecutionPlanSnapshot, o as ExactAnalystRunResult, r as ExactAnalystRunCompletion, s as ExactAnalystRunSummary, t as ExactAnalystBudgetSnapshot, u as ExactExecutionComponentIdentity } from "../exact-types-BKOEILRP.js";
|
|
7
|
+
import { a as ExactAnalystBudgetPolicy, c as RegistryRunOpts, i as BudgetPolicy, n as AnalystRegistry, o as ExactAnalystRunExecutionError, r as AnalystRegistryOptions, s as ExactRegistryRunOpts, t as AnalystHooks } from "../registry-7pOUBrtX.js";
|
|
8
|
+
import { a as TraceAnalystLimits, c as RawAnalystEvidence, d as evidenceRefsFromRawFinding, f as parseRawFinding, i as TraceAnalysisEngineResult, l as RawAnalystFinding, n as TraceAnalysisEngine, o as resolveTraceAnalystLimits, r as TraceAnalysisEngineRequest, s as RAW_FINDING_SCHEMA_PROMPT, t as DEFAULT_TRACE_ANALYST_OUTPUT_TOKENS, u as RawAnalystFindingSchema } from "../engine-DhFir3Ys.js";
|
|
9
|
+
import { n as buildTraceToolsForGroup, t as TraceToolGroupName } from "../tool-groups-Cp4Xdzrp.js";
|
|
10
|
+
import { A as SemanticConceptJudgeInput, C as DspyRlmTraceEngineOptions, E as createChatTraceEngine, S as renderFindingSubject, T as ChatTraceEngineOptions, _ as FindingSubject, a as FAILURE_MODE_KIND_SPEC, b as findingSubjectGrammarPromptFor, c as emitControlIntegrityFindings, d as FindingsStore, f as PersistedFinding, g as FINDING_SUBJECT_SYNTAX, h as FINDING_SUBJECT_KINDS, i as IMPROVEMENT_KIND_SPEC, j as SemanticConceptJudgeOptions, l as DiffPolicy, m as diffFindings, n as KNOWLEDGE_POISONING_KIND_SPEC, o as CONTROL_INTEGRITY_ANALYST, p as defaultIsMaterial, r as KNOWLEDGE_GAP_KIND_SPEC, s as ControlIntegrityAnalyst, t as DEFAULT_TRACE_ANALYST_KINDS, u as FindingsDiff, v as FindingSubjectKind, w as createDspyRlmTraceEngine, x as parseFindingSubject, y as KIND_EXPECTED_SUBJECTS } from "../index-D0Db5X-4.js";
|
|
11
|
+
import { C as scoreAnalystFindings, S as traceStoreEvidenceResolver, _ as AnalystIssueExpectation, a as AnalystBenchmarkLabelState, b as registryBenchmarkRunner, c as AnalystBenchmarkProvenance, d as AnalystBenchmarkSummary, f as AnalystEvidenceExpectation, g as AnalystFindingScore, h as AnalystEvidenceResolver, i as AnalystBenchmarkError, l as AnalystBenchmarkResult, m as AnalystEvidenceResolutionError, n as AnalystBenchmarkDatasetRef, o as AnalystBenchmarkObservation, p as AnalystEvidenceResolution, r as AnalystBenchmarkDescriptor, s as AnalystBenchmarkOutput, t as AnalystBenchmarkCase, u as AnalystBenchmarkRunner, v as AnalystLatencyDistribution, x as runAnalystBenchmark, y as RunAnalystBenchmarkOptions } from "../benchmark-BjLGkfnN.js";
|
|
12
12
|
import { t as AgentProfile } from "../agent-profile-B9_GGsG8.js";
|
|
13
13
|
import { i as nodeHttpPrimeBridgeTransport, n as PrimeBridgeTransportRequest, r as PrimeBridgeTransportResult, t as PrimeBridgeTransport } from "../prime-bridge-transport-6feEglLf.js";
|
|
14
14
|
import { z } from "zod";
|
|
@@ -1267,11 +1267,11 @@ declare function runAnalystBenchmarkCommand(argv: readonly string[], env?: NodeJ
|
|
|
1267
1267
|
declare const ANALYST_BENCHMARK_IMPLEMENTATION_DIGEST_ALGORITHM = "sha256-canonical-source-manifest";
|
|
1268
1268
|
declare const ANALYST_BENCHMARK_DEPENDENCY_LOCK_DIGEST_ALGORITHM = "sha256-canonical-file-manifest";
|
|
1269
1269
|
declare const ANALYST_BENCHMARK_DEPENDENCY_LOCK_FILES: readonly string[];
|
|
1270
|
-
declare const ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256 = "
|
|
1270
|
+
declare const ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256 = "8e4b431bbba1d1d485847fd21acea07ce78bde91f43b9f64750046d70b7373b1";
|
|
1271
1271
|
declare const ANALYST_BENCHMARK_EVIDENCE_DEPENDENCY_LOCK_SHA256 = "1e03f2daed356d60316aabefb407ec1e437ac94d408d61eea4ae096e9c6fbb5b";
|
|
1272
1272
|
declare const ANALYST_BENCHMARK_EVIDENCE_IMPLEMENTATION_SHA256 = "4dba263b6256a30d56c7fdb2d992d3a953c0035d731f359b704db806f68f75ac";
|
|
1273
1273
|
declare const ANALYST_BENCHMARK_IMPLEMENTATION_FILES: readonly string[];
|
|
1274
|
-
declare const ANALYST_BENCHMARK_IMPLEMENTATION_SHA256 = "
|
|
1274
|
+
declare const ANALYST_BENCHMARK_IMPLEMENTATION_SHA256 = "d099d13e08710074eeced469aea8220ef50acf41738372a5c17cb26155628cf7";
|
|
1275
1275
|
declare function analystBenchmarkImplementationDigest(): string;
|
|
1276
1276
|
declare function analystBenchmarkDependencyLockDigest(): string;
|
|
1277
1277
|
//#endregion
|
|
@@ -1682,5 +1682,5 @@ declare function isProposalFinding(finding: unknown): finding is ProposalFinding
|
|
|
1682
1682
|
*/
|
|
1683
1683
|
declare function assertProposalFindings(findings: unknown, context?: string): ReadonlyArray<ProposalFinding>;
|
|
1684
1684
|
//#endregion
|
|
1685
|
-
export { AGENT_RX_UPSTREAM_REVISION, ANALYST_BENCHMARK_COST_LEDGER_FILE, ANALYST_BENCHMARK_DEPENDENCY_LOCK_DIGEST_ALGORITHM, ANALYST_BENCHMARK_DEPENDENCY_LOCK_FILES, ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256, ANALYST_BENCHMARK_EVIDENCE_DEPENDENCY_LOCK_SHA256, ANALYST_BENCHMARK_EVIDENCE_IMPLEMENTATION_SHA256, ANALYST_BENCHMARK_IMPLEMENTATION_DIGEST_ALGORITHM, ANALYST_BENCHMARK_IMPLEMENTATION_FILES, ANALYST_BENCHMARK_IMPLEMENTATION_SHA256, ANALYST_BENCHMARK_LOCAL_RECEIPT_FILE, ANALYST_BENCHMARK_MANIFEST_FILE, ANALYST_BENCHMARK_OBSERVATIONS_FILE, type AgentRxBenchmarkCaseOptions, type AgentRxCalibrationRunnerSummary, type AgentRxCalibrationSummary, type AgentRxFailure, type AgentRxPrediction, type AgentRxPredictionReport, type AgentRxRow, type Analyst, type AnalystBenchmarkArtifact, type AnalystBenchmarkCase, type AnalystBenchmarkCommandConfig, type AnalystBenchmarkCommandDependencies, type AnalystBenchmarkDatasetRef, type AnalystBenchmarkDescriptor, type AnalystBenchmarkError, type AnalystBenchmarkLabelState, type AnalystBenchmarkLocalRunReceipt, type AnalystBenchmarkObservation, type AnalystBenchmarkOutput, type AnalystBenchmarkProgressRow, type AnalystBenchmarkProvenance, type AnalystBenchmarkResult, type AnalystBenchmarkRunIdentity, type AnalystBenchmarkRunManifest, type AnalystBenchmarkRunner, type AnalystBenchmarkSummary, type AnalystBudgetDeclaration, type AnalystComparisonMetric, type AnalystContext, type AnalystCost, type AnalystDefinition, type AnalystDefinitionAsymmetry, type AnalystDefinitionAsymmetryReport, type AnalystEvidenceBinding, type AnalystEvidenceExpectation, type AnalystEvidenceResolution, type AnalystEvidenceResolutionError, type AnalystEvidenceResolver, AnalystExpressivenessError, type AnalystFinding, type AnalystFindingScore, type AnalystHooks, type AnalystInputKind, type AnalystInstructionsOverride, type AnalystIssueExpectation, type AnalystLatencyDistribution, type AnalystMetricComparison, type AnalystProfileFragment, AnalystRegistry, type AnalystRegistryOptions, type AnalystRepairDeclaration, type AnalystRequirements, type AnalystRowExpansion, type AnalystRunEvent, type AnalystRunInputs, type AnalystRunResult, type AnalystRunSummary, type AnalystRunnerComparison, type AnalystSeverity, type AnalystTransportBinding, type AnalystUsageReceipt, type BehavioralAnalystOptions, type BudgetPolicy, CODE_TRACE_BENCH_ANALYST_PROMPT, CONTROL_INTEGRITY_ANALYST, type ChatCallOpts, type ChatClient, type ChatRequest, type ChatResponse, type ChatTransport, type ChunkedEvidenceBinding, type CodeTraceBenchCaseOptions, type CodeTraceBenchLabelOptions, type CodeTraceBenchLabelSet, type CodeTraceBenchRow, type CodeTraceBlockDiagnostics, type CodeTraceCalibrationRunnerSummary, type CodeTraceCalibrationSummary, type CodeTraceFailureBlock, type CodeTraceStageAnnotation, type CodeTracerLabelGroup, type CodeTracerPredictionAdapterOptions, type CodeTracerPredictions, type CodeTracerStepLabel, ControlIntegrityAnalyst, type CreateChatClientOpts, type CreateTraceAnalystOptions, type CustomTransportOpts, DEFAULT_MAX_VERIFICATION_ARTIFACT_BYTES, DEFAULT_TRACE_ANALYST_KINDS, type DecodedReply, type DefaultAnalystRegistryOptions, type DefineCustomAnalystOptions, type DefineExactCustomAnalystOptions, type DiffPolicy, type DspyRlmTraceEngineOptions, type EvidenceProjection, type EvidenceRef, type ExactAnalystBudgetPolicy, type ExactAnalystBudgetSnapshot, type ExactAnalystExecutionPlanSnapshot, type ExactAnalystRunCompletion, type ExactAnalystRunEvent, ExactAnalystRunExecutionError, type ExactAnalystRunPolicySnapshot, type ExactAnalystRunResult, type ExactAnalystRunSummary, type ExactAnalystSnapshot, type ExactCapableAnalyst, type ExactExecutionComponentIdentity, type ExactExecutionComponentSnapshot, type ExactRegistryRunOpts, type ExecutionProbe, type ExecutionProbeOutcome, type ExecutionProbeRequest, type ExpandRowsArgs, FAILURE_MODE_KIND_SPEC, FINDING_SUBJECT_KINDS, FINDING_SUBJECT_SYNTAX, type FindingSubject, type FindingSubjectKind, type FindingsDiff, FindingsStore, IMPROVEMENT_KIND_SPEC, type InlineEvidenceBinding, KIND_EXPECTED_SUBJECTS, KNOWLEDGE_GAP_KIND_SPEC, KNOWLEDGE_POISONING_KIND_SPEC, type LoadedVerificationArtifacts, MAX_INCORRECT_BLOCKS, MAX_INCORRECT_BLOCK_STEPS, type MockTransportOpts, type PersistedFinding, type PreparedPublicAnalystBenchmark, type PrimeBenchmarkRunnerOptions, type PrimeBridgeTransport, type PrimeBridgeTransportRequest, type PrimeBridgeTransportResult, type PrimeCodeTraceDefinitionArgs, type PrimeExchangeOptions, type PrimeExchangeOutcome, type PrimeFailure, type PrimeProjectionDelivery, type PrimeProjectionOutcome, type PrimeProjectionSource, type PrimePromptSpec, type PrimeProtocolIdentity, type PrimeRawUsage, type PrimeRejectedRow, type PrimeRepairPromptSpec, type PrimeRepairState, type PrimeReplyContract, type PrimeRowDecoded, type PrimeTurnRecord, type ProposalFinding, type ProposalFindingOrigin, type PublicAnalystBenchmarkDataset, type PublicAnalystBenchmarkModelConfig, type PublicAnalystBenchmarkModelOwner, type PublicAnalystBenchmarkModelSettings, type PublicBenchmarkDistributions, type PublicBenchmarkModelPrediction, type PublicBenchmarkSelectionReport, type PublicBenchmarkValueDistribution, type PublicDirectDefinitionArgs, type PublicRlmDefinitionArgs, RAW_FINDING_SCHEMA_PROMPT, type RawAnalystEvidence, type RawAnalystFinding, RawAnalystFindingSchema, type RegistryRunOpts, type ReplVariableConsensusPort, type ReplVariableEvidenceBinding, type ReplyContract, type ReplyEnvelope, type ReplyRowDecoded, type ReplyRowRejection, type RunAnalystBenchmarkOptions, SKILL_USAGE_ANALYST, type SandboxSdkTransportOpts, type SemanticConceptJudgeAdapterOpts, type SkillUsageRecord, type SkillUsageReport, type SkillUsageScanConfig, type StepLabelAdapterOptions, type TraceAnalysisEngine, type TraceAnalysisEngineRequest, type TraceAnalysisEngineResult, type TraceAnalystDefinition, type TraceAnalystLimits, type TraceToolGroupName, type UpstreamPredictionAdapterOptions, type VerificationArtifactFile, type VerificationArtifactManifest, type VerificationArtifactRole, type VerificationAvailabilitySummary, type VerificationOutcome, type VerificationOutcomeSource, type VerificationOutcomeStatus, type VerificationResultFile, adaptPublicBenchmarkFindings, agentRxBenchmarkCase, agentRxPredictionsToFindings, analystBenchmarkDependencyLockDigest, analystBenchmarkImplementationDigest, analystDefinitionAsymmetries, analystDefinitionProtocolSha256, analystInstructionsOverrideFromText, analystUsageReceiptFromPrimeUsage, appendVerificationArtifactsToOtlp, assertProposalFindings, behavioralAnalyst, bindAnalyst, buildDefaultAnalystRegistry, buildPrimePrompt, buildPrimeRepairPrompt, buildSkillUsageReport, buildTraceToolsForGroup, codeTraceBenchCase, codeTracerPredictionsToFindings, coerceJson, compareAnalystRunners, computeFindingId, createChatClient, createDspyRlmTraceEngine, createPrimeBenchmarkRunner, createPublicBenchmarkDirectRunner, createPublicBenchmarkRlmRunner, createSemanticConceptJudgeAdapter, createTraceAnalyst, decodeReplyRows, defaultIsMaterial, defineCustomAnalyst, defineTraceAnalyst, deriveEfficiencyFindings, diffFindings, effectiveAnalystProtocolSha256, emitControlIntegrityFindings, emitSkillUsageFindings, emptyPrimeRawUsage, emptyPublicBenchmarkRunner, evidenceRefsFromRawFinding, expandCodeTraceFailureBlocks, extractPrimeJsonObject, findingSubjectGrammarPromptFor, isProposalFinding, loadCodeTraceVerificationArtifacts, loadPublicBenchmarkRows, makeFinding, makeProposalFinding, mergePrimeRawUsage, nodeHttpPrimeBridgeTransport, normalizeAgentRxCategory, normalizeBenchmarkLabel, normalizePrimeUsage, parseFindingSubject, parseRawFinding, parseVerificationOutcome, preparePublicAnalystBenchmark, primeAnalystProtocolSha256, primeCodeTraceAnalystDefinition, primeProtocolSha256, primeReplyDefect, projectPrimeTrajectory, publicBenchmarkDistributions, publicBenchmarkProtocolSha256, publicBenchmarkRlmInstructions, publicBenchmarkSelectionReport, publicBenchmarkSystemPrompt, publicDirectAnalystDefinition, publicRlmAnalystDefinition, readAnalystBenchmarkArtifact, readAnalystInstructionsOverride, registryBenchmarkRunner, renderAgentRxCalibrationMarkdown, renderAnalystBenchmarkMarkdown, renderCodeTraceCalibrationMarkdown, renderFindingSubject, renderPriorFindings, resolveTraceAnalystLimits, rlmEngineLimits, roundAgentRxStep, runAnalystBenchmark, runAnalystBenchmarkCommand, runPrimeExchange, runTraceAnalyst, scoreAnalystFindings, selectPublicBenchmarkRows, stripCodeFences, summarizeAgentRxCalibration, summarizeAnalystBenchmarkRunner, summarizeCodeTraceCalibration, traceStoreEvidenceResolver };
|
|
1685
|
+
export { AGENT_RX_UPSTREAM_REVISION, ANALYST_BENCHMARK_COST_LEDGER_FILE, ANALYST_BENCHMARK_DEPENDENCY_LOCK_DIGEST_ALGORITHM, ANALYST_BENCHMARK_DEPENDENCY_LOCK_FILES, ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256, ANALYST_BENCHMARK_EVIDENCE_DEPENDENCY_LOCK_SHA256, ANALYST_BENCHMARK_EVIDENCE_IMPLEMENTATION_SHA256, ANALYST_BENCHMARK_IMPLEMENTATION_DIGEST_ALGORITHM, ANALYST_BENCHMARK_IMPLEMENTATION_FILES, ANALYST_BENCHMARK_IMPLEMENTATION_SHA256, ANALYST_BENCHMARK_LOCAL_RECEIPT_FILE, ANALYST_BENCHMARK_MANIFEST_FILE, ANALYST_BENCHMARK_OBSERVATIONS_FILE, type AgentRxBenchmarkCaseOptions, type AgentRxCalibrationRunnerSummary, type AgentRxCalibrationSummary, type AgentRxFailure, type AgentRxPrediction, type AgentRxPredictionReport, type AgentRxRow, type Analyst, type AnalystBenchmarkArtifact, type AnalystBenchmarkCase, type AnalystBenchmarkCommandConfig, type AnalystBenchmarkCommandDependencies, type AnalystBenchmarkDatasetRef, type AnalystBenchmarkDescriptor, type AnalystBenchmarkError, type AnalystBenchmarkLabelState, type AnalystBenchmarkLocalRunReceipt, type AnalystBenchmarkObservation, type AnalystBenchmarkOutput, type AnalystBenchmarkProgressRow, type AnalystBenchmarkProvenance, type AnalystBenchmarkResult, type AnalystBenchmarkRunIdentity, type AnalystBenchmarkRunManifest, type AnalystBenchmarkRunner, type AnalystBenchmarkSummary, type AnalystBudgetDeclaration, type AnalystComparisonMetric, type AnalystContext, type AnalystCost, type AnalystDefinition, type AnalystDefinitionAsymmetry, type AnalystDefinitionAsymmetryReport, type AnalystEvidenceBinding, type AnalystEvidenceExpectation, type AnalystEvidenceResolution, type AnalystEvidenceResolutionError, type AnalystEvidenceResolver, AnalystExpressivenessError, type AnalystFinding, type AnalystFindingScore, type AnalystHooks, type AnalystInputKind, type AnalystInstructionsOverride, type AnalystIssueExpectation, type AnalystLatencyDistribution, type AnalystMetricComparison, type AnalystProfileFragment, AnalystRegistry, type AnalystRegistryOptions, type AnalystRepairDeclaration, type AnalystRequirements, type AnalystRowExpansion, type AnalystRunEvent, type AnalystRunInputs, type AnalystRunResult, type AnalystRunSummary, type AnalystRunnerComparison, type AnalystSeverity, type AnalystTransportBinding, type AnalystUsageReceipt, type BehavioralAnalystOptions, type BudgetPolicy, CODE_TRACE_BENCH_ANALYST_PROMPT, CONTROL_INTEGRITY_ANALYST, type ChatCallOpts, type ChatClient, type ChatRequest, type ChatResponse, type ChatTraceEngineOptions, type ChatTransport, type ChunkedEvidenceBinding, type CodeTraceBenchCaseOptions, type CodeTraceBenchLabelOptions, type CodeTraceBenchLabelSet, type CodeTraceBenchRow, type CodeTraceBlockDiagnostics, type CodeTraceCalibrationRunnerSummary, type CodeTraceCalibrationSummary, type CodeTraceFailureBlock, type CodeTraceStageAnnotation, type CodeTracerLabelGroup, type CodeTracerPredictionAdapterOptions, type CodeTracerPredictions, type CodeTracerStepLabel, ControlIntegrityAnalyst, type CreateChatClientOpts, type CreateTraceAnalystOptions, type CustomTransportOpts, DEFAULT_MAX_VERIFICATION_ARTIFACT_BYTES, DEFAULT_TRACE_ANALYST_KINDS, DEFAULT_TRACE_ANALYST_OUTPUT_TOKENS, type DecodedReply, type DefaultAnalystRegistryOptions, type DefineCustomAnalystOptions, type DefineExactCustomAnalystOptions, type DiffPolicy, type DspyRlmTraceEngineOptions, type EvidenceProjection, type EvidenceRef, type ExactAnalystBudgetPolicy, type ExactAnalystBudgetSnapshot, type ExactAnalystExecutionPlanSnapshot, type ExactAnalystRunCompletion, type ExactAnalystRunEvent, ExactAnalystRunExecutionError, type ExactAnalystRunPolicySnapshot, type ExactAnalystRunResult, type ExactAnalystRunSummary, type ExactAnalystSnapshot, type ExactCapableAnalyst, type ExactExecutionComponentIdentity, type ExactExecutionComponentSnapshot, type ExactRegistryRunOpts, type ExecutionProbe, type ExecutionProbeOutcome, type ExecutionProbeRequest, type ExpandRowsArgs, FAILURE_MODE_KIND_SPEC, FINDING_SUBJECT_KINDS, FINDING_SUBJECT_SYNTAX, type FindingSubject, type FindingSubjectKind, type FindingsDiff, FindingsStore, IMPROVEMENT_KIND_SPEC, type InlineEvidenceBinding, KIND_EXPECTED_SUBJECTS, KNOWLEDGE_GAP_KIND_SPEC, KNOWLEDGE_POISONING_KIND_SPEC, type LoadedVerificationArtifacts, MAX_INCORRECT_BLOCKS, MAX_INCORRECT_BLOCK_STEPS, type MockTransportOpts, type OpenAiCompatibleTransportOpts, type PersistedFinding, type PreparedPublicAnalystBenchmark, type PrimeBenchmarkRunnerOptions, type PrimeBridgeTransport, type PrimeBridgeTransportRequest, type PrimeBridgeTransportResult, type PrimeCodeTraceDefinitionArgs, type PrimeExchangeOptions, type PrimeExchangeOutcome, type PrimeFailure, type PrimeProjectionDelivery, type PrimeProjectionOutcome, type PrimeProjectionSource, type PrimePromptSpec, type PrimeProtocolIdentity, type PrimeRawUsage, type PrimeRejectedRow, type PrimeRepairPromptSpec, type PrimeRepairState, type PrimeReplyContract, type PrimeRowDecoded, type PrimeTurnRecord, type ProposalFinding, type ProposalFindingOrigin, type PublicAnalystBenchmarkDataset, type PublicAnalystBenchmarkModelConfig, type PublicAnalystBenchmarkModelOwner, type PublicAnalystBenchmarkModelSettings, type PublicBenchmarkDistributions, type PublicBenchmarkModelPrediction, type PublicBenchmarkSelectionReport, type PublicBenchmarkValueDistribution, type PublicDirectDefinitionArgs, type PublicRlmDefinitionArgs, RAW_FINDING_SCHEMA_PROMPT, type RawAnalystEvidence, type RawAnalystFinding, RawAnalystFindingSchema, type RegistryRunOpts, type ReplVariableConsensusPort, type ReplVariableEvidenceBinding, type ReplyContract, type ReplyEnvelope, type ReplyRowDecoded, type ReplyRowRejection, type RunAnalystBenchmarkOptions, SKILL_USAGE_ANALYST, type SandboxSdkTransportOpts, type SemanticConceptJudgeAdapterOpts, type SkillUsageRecord, type SkillUsageReport, type SkillUsageScanConfig, type StepLabelAdapterOptions, type TraceAnalysisEngine, type TraceAnalysisEngineRequest, type TraceAnalysisEngineResult, type TraceAnalystDefinition, type TraceAnalystLimits, type TraceToolGroupName, type UpstreamPredictionAdapterOptions, type VerificationArtifactFile, type VerificationArtifactManifest, type VerificationArtifactRole, type VerificationAvailabilitySummary, type VerificationOutcome, type VerificationOutcomeSource, type VerificationOutcomeStatus, type VerificationResultFile, adaptPublicBenchmarkFindings, agentRxBenchmarkCase, agentRxPredictionsToFindings, analystBenchmarkDependencyLockDigest, analystBenchmarkImplementationDigest, analystDefinitionAsymmetries, analystDefinitionProtocolSha256, analystInstructionsOverrideFromText, analystUsageReceiptFromPrimeUsage, appendVerificationArtifactsToOtlp, assertProposalFindings, behavioralAnalyst, bindAnalyst, buildDefaultAnalystRegistry, buildPrimePrompt, buildPrimeRepairPrompt, buildSkillUsageReport, buildTraceToolsForGroup, codeTraceBenchCase, codeTracerPredictionsToFindings, coerceJson, compareAnalystRunners, computeFindingId, createChatClient, createChatTraceEngine, createDspyRlmTraceEngine, createPrimeBenchmarkRunner, createPublicBenchmarkDirectRunner, createPublicBenchmarkRlmRunner, createSemanticConceptJudgeAdapter, createTraceAnalyst, decodeReplyRows, defaultIsMaterial, defineCustomAnalyst, defineTraceAnalyst, deriveEfficiencyFindings, diffFindings, effectiveAnalystProtocolSha256, emitControlIntegrityFindings, emitSkillUsageFindings, emptyPrimeRawUsage, emptyPublicBenchmarkRunner, evidenceRefsFromRawFinding, expandCodeTraceFailureBlocks, extractPrimeJsonObject, findingSubjectGrammarPromptFor, isProposalFinding, loadCodeTraceVerificationArtifacts, loadPublicBenchmarkRows, makeFinding, makeProposalFinding, mergePrimeRawUsage, nodeHttpPrimeBridgeTransport, normalizeAgentRxCategory, normalizeBenchmarkLabel, normalizePrimeUsage, parseFindingSubject, parseRawFinding, parseVerificationOutcome, preparePublicAnalystBenchmark, primeAnalystProtocolSha256, primeCodeTraceAnalystDefinition, primeProtocolSha256, primeReplyDefect, projectPrimeTrajectory, publicBenchmarkDistributions, publicBenchmarkProtocolSha256, publicBenchmarkRlmInstructions, publicBenchmarkSelectionReport, publicBenchmarkSystemPrompt, publicDirectAnalystDefinition, publicRlmAnalystDefinition, readAnalystBenchmarkArtifact, readAnalystInstructionsOverride, registryBenchmarkRunner, renderAgentRxCalibrationMarkdown, renderAnalystBenchmarkMarkdown, renderCodeTraceCalibrationMarkdown, renderFindingSubject, renderPriorFindings, resolveTraceAnalystLimits, rlmEngineLimits, roundAgentRxStep, runAnalystBenchmark, runAnalystBenchmarkCommand, runPrimeExchange, runTraceAnalyst, scoreAnalystFindings, selectPublicBenchmarkRows, stripCodeFences, summarizeAgentRxCalibration, summarizeAnalystBenchmarkRunner, summarizeCodeTraceCalibration, traceStoreEvidenceResolver };
|
|
1686
1686
|
//# sourceMappingURL=index.d.ts.map
|
package/dist/analyst/index.js
CHANGED
|
@@ -1,12 +1,13 @@
|
|
|
1
1
|
import { i as CostLedger } from "../cost-ledger-B1qx30B4.js";
|
|
2
2
|
import { n as isProposalFinding, t as assertProposalFindings } from "../proposal-findings-bko3GGy-.js";
|
|
3
3
|
import { c as validateUsageSettlementTimeout, i as makeProposalFinding, n as computeFindingId, o as settleUsageReceiptFromCostLedger, r as makeFinding } from "../types-CiWITkGo.js";
|
|
4
|
-
import { A as
|
|
5
|
-
import { a as
|
|
6
|
-
import { t as createDspyRlmTraceEngine } from "../dspy-rlm-engine-
|
|
7
|
-
import { a as diffFindings, i as defaultIsMaterial, n as runSemanticConceptJudge, r as FindingsStore, t as SEMANTIC_CONCEPT_JUDGE_VERSION } from "../semantic-concept-judge-
|
|
4
|
+
import { A as RAW_FINDING_SCHEMA_PROMPT, B as parseFindingSubject, F as stripCodeFences, H as DEFAULT_TRACE_ANALYST_OUTPUT_TOKENS, I as FINDING_SUBJECT_KINDS, L as FINDING_SUBJECT_SYNTAX, M as evidenceRefsFromRawFinding, N as parseRawFinding, P as coerceJson, R as KIND_EXPECTED_SUBJECTS, U as resolveTraceAnalystLimits, V as renderFindingSubject, i as buildTraceToolsForGroup, j as RawAnalystFindingSchema, n as renderPriorFindings, r as runTraceAnalyst, t as createTraceAnalyst, z as findingSubjectGrammarPromptFor } from "../kind-factory-gP6lDySe.js";
|
|
5
|
+
import { a as KNOWLEDGE_POISONING_KIND_SPEC, c as FAILURE_MODE_KIND_SPEC, d as emitControlIntegrityFindings, f as behavioralAnalyst, i as DEFAULT_TRACE_ANALYST_KINDS, l as CONTROL_INTEGRITY_ANALYST, n as AnalystRegistry, o as KNOWLEDGE_GAP_KIND_SPEC, p as deriveEfficiencyFindings, r as ExactAnalystRunExecutionError, s as IMPROVEMENT_KIND_SPEC, t as buildDefaultAnalystRegistry, u as ControlIntegrityAnalyst } from "../default-registry-B0bKikCb.js";
|
|
6
|
+
import { t as createDspyRlmTraceEngine } from "../dspy-rlm-engine-D5byiHn9.js";
|
|
7
|
+
import { a as diffFindings, i as defaultIsMaterial, n as runSemanticConceptJudge, o as createChatTraceEngine, r as FindingsStore, t as SEMANTIC_CONCEPT_JUDGE_VERSION } from "../semantic-concept-judge-Ct3QU7t5.js";
|
|
8
|
+
import { t as createChatClient } from "../chat-client-Db4bqYfA.js";
|
|
8
9
|
import { a as scoreAnalystFindings, i as summarizeAnalystBenchmarkRunner, n as runAnalystBenchmark, r as traceStoreEvidenceResolver, t as registryBenchmarkRunner } from "../benchmark-C4wk_Sjr.js";
|
|
9
|
-
import { $ as ANALYST_BENCHMARK_LOCAL_RECEIPT_FILE, A as effectiveAnalystProtocolSha256, B as loadCodeTraceVerificationArtifacts, C as adaptPublicBenchmarkFindings, D as renderCodeTraceCalibrationMarkdown, E as readAnalystBenchmarkArtifact, F as publicBenchmarkProtocolSha256, G as ANALYST_BENCHMARK_EVIDENCE_DEPENDENCY_LOCK_SHA256, H as ANALYST_BENCHMARK_DEPENDENCY_LOCK_DIGEST_ALGORITHM, I as publicBenchmarkRlmInstructions, J as ANALYST_BENCHMARK_IMPLEMENTATION_FILES, K as ANALYST_BENCHMARK_EVIDENCE_IMPLEMENTATION_SHA256, L as publicBenchmarkSystemPrompt, M as CODE_TRACE_BENCH_ANALYST_PROMPT, N as MAX_INCORRECT_BLOCKS, O as summarizeCodeTraceCalibration, P as MAX_INCORRECT_BLOCK_STEPS, Q as ANALYST_BENCHMARK_COST_LEDGER_FILE, R as DEFAULT_MAX_VERIFICATION_ARTIFACT_BYTES, S as analystDefinitionProtocolSha256, T as expandCodeTraceFailureBlocks, U as ANALYST_BENCHMARK_DEPENDENCY_LOCK_FILES, V as parseVerificationOutcome, W as ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256, X as analystBenchmarkDependencyLockDigest, Y as ANALYST_BENCHMARK_IMPLEMENTATION_SHA256, Z as analystBenchmarkImplementationDigest, _ as createPublicBenchmarkDirectRunner, a as primeCodeTraceAnalystDefinition, at as summarizeAgentRxCalibration, b as AnalystExpressivenessError, c as loadPublicBenchmarkRows, ct as agentRxBenchmarkCase, d as publicBenchmarkSelectionReport, dt as roundAgentRxStep, et as ANALYST_BENCHMARK_MANIFEST_FILE, f as selectPublicBenchmarkRows, ft as normalizeBenchmarkLabel, g as runReplVariableAnalystDefinition, h as rlmEngineLimits, i as primeAnalystProtocolSha256, it as renderAgentRxCalibrationMarkdown, j as readAnalystInstructionsOverride, k as analystInstructionsOverrideFromText, l as preparePublicAnalystBenchmark, lt as agentRxPredictionsToFindings, m as publicRlmAnalystDefinition, n as renderAnalystBenchmarkMarkdown, nt as compareAnalystRunners, o as runInlineAnalystDefinition, ot as codeTraceBenchCase, p as createPublicBenchmarkRlmRunner, q as ANALYST_BENCHMARK_IMPLEMENTATION_DIGEST_ALGORITHM, r as createPrimeBenchmarkRunner, rt as AGENT_RX_UPSTREAM_REVISION, s as nodeHttpPrimeBridgeTransport, st as codeTracerPredictionsToFindings, t as runAnalystBenchmarkCommand, tt as ANALYST_BENCHMARK_OBSERVATIONS_FILE, u as publicBenchmarkDistributions, ut as normalizeAgentRxCategory, v as publicDirectAnalystDefinition, w as emptyPublicBenchmarkRunner, x as analystDefinitionAsymmetries, y as runChunkedAnalystDefinition, z as appendVerificationArtifactsToOtlp } from "../benchmark-command-
|
|
10
|
+
import { $ as ANALYST_BENCHMARK_LOCAL_RECEIPT_FILE, A as effectiveAnalystProtocolSha256, B as loadCodeTraceVerificationArtifacts, C as adaptPublicBenchmarkFindings, D as renderCodeTraceCalibrationMarkdown, E as readAnalystBenchmarkArtifact, F as publicBenchmarkProtocolSha256, G as ANALYST_BENCHMARK_EVIDENCE_DEPENDENCY_LOCK_SHA256, H as ANALYST_BENCHMARK_DEPENDENCY_LOCK_DIGEST_ALGORITHM, I as publicBenchmarkRlmInstructions, J as ANALYST_BENCHMARK_IMPLEMENTATION_FILES, K as ANALYST_BENCHMARK_EVIDENCE_IMPLEMENTATION_SHA256, L as publicBenchmarkSystemPrompt, M as CODE_TRACE_BENCH_ANALYST_PROMPT, N as MAX_INCORRECT_BLOCKS, O as summarizeCodeTraceCalibration, P as MAX_INCORRECT_BLOCK_STEPS, Q as ANALYST_BENCHMARK_COST_LEDGER_FILE, R as DEFAULT_MAX_VERIFICATION_ARTIFACT_BYTES, S as analystDefinitionProtocolSha256, T as expandCodeTraceFailureBlocks, U as ANALYST_BENCHMARK_DEPENDENCY_LOCK_FILES, V as parseVerificationOutcome, W as ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256, X as analystBenchmarkDependencyLockDigest, Y as ANALYST_BENCHMARK_IMPLEMENTATION_SHA256, Z as analystBenchmarkImplementationDigest, _ as createPublicBenchmarkDirectRunner, a as primeCodeTraceAnalystDefinition, at as summarizeAgentRxCalibration, b as AnalystExpressivenessError, c as loadPublicBenchmarkRows, ct as agentRxBenchmarkCase, d as publicBenchmarkSelectionReport, dt as roundAgentRxStep, et as ANALYST_BENCHMARK_MANIFEST_FILE, f as selectPublicBenchmarkRows, ft as normalizeBenchmarkLabel, g as runReplVariableAnalystDefinition, h as rlmEngineLimits, i as primeAnalystProtocolSha256, it as renderAgentRxCalibrationMarkdown, j as readAnalystInstructionsOverride, k as analystInstructionsOverrideFromText, l as preparePublicAnalystBenchmark, lt as agentRxPredictionsToFindings, m as publicRlmAnalystDefinition, n as renderAnalystBenchmarkMarkdown, nt as compareAnalystRunners, o as runInlineAnalystDefinition, ot as codeTraceBenchCase, p as createPublicBenchmarkRlmRunner, q as ANALYST_BENCHMARK_IMPLEMENTATION_DIGEST_ALGORITHM, r as createPrimeBenchmarkRunner, rt as AGENT_RX_UPSTREAM_REVISION, s as nodeHttpPrimeBridgeTransport, st as codeTracerPredictionsToFindings, t as runAnalystBenchmarkCommand, tt as ANALYST_BENCHMARK_OBSERVATIONS_FILE, u as publicBenchmarkDistributions, ut as normalizeAgentRxCategory, v as publicDirectAnalystDefinition, w as emptyPublicBenchmarkRunner, x as analystDefinitionAsymmetries, y as runChunkedAnalystDefinition, z as appendVerificationArtifactsToOtlp } from "../benchmark-command-9S20PRel.js";
|
|
10
11
|
import { a as extractPrimeJsonObject, c as primeProtocolSha256, d as runPrimeExchange, f as decodeReplyRows, i as emptyPrimeRawUsage, l as primeReplyDefect, n as buildPrimePrompt, o as mergePrimeRawUsage, r as buildPrimeRepairPrompt, s as normalizePrimeUsage, t as analystUsageReceiptFromPrimeUsage, u as projectPrimeTrajectory } from "../prime-protocol-6tZTVsWm.js";
|
|
11
12
|
import { existsSync, readFileSync, readdirSync, statSync } from "node:fs";
|
|
12
13
|
import { join } from "node:path";
|
|
@@ -374,6 +375,6 @@ var SkillUsageAnalyst = class {
|
|
|
374
375
|
};
|
|
375
376
|
const SKILL_USAGE_ANALYST = new SkillUsageAnalyst();
|
|
376
377
|
//#endregion
|
|
377
|
-
export { AGENT_RX_UPSTREAM_REVISION, ANALYST_BENCHMARK_COST_LEDGER_FILE, ANALYST_BENCHMARK_DEPENDENCY_LOCK_DIGEST_ALGORITHM, ANALYST_BENCHMARK_DEPENDENCY_LOCK_FILES, ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256, ANALYST_BENCHMARK_EVIDENCE_DEPENDENCY_LOCK_SHA256, ANALYST_BENCHMARK_EVIDENCE_IMPLEMENTATION_SHA256, ANALYST_BENCHMARK_IMPLEMENTATION_DIGEST_ALGORITHM, ANALYST_BENCHMARK_IMPLEMENTATION_FILES, ANALYST_BENCHMARK_IMPLEMENTATION_SHA256, ANALYST_BENCHMARK_LOCAL_RECEIPT_FILE, ANALYST_BENCHMARK_MANIFEST_FILE, ANALYST_BENCHMARK_OBSERVATIONS_FILE, AnalystExpressivenessError, AnalystRegistry, CODE_TRACE_BENCH_ANALYST_PROMPT, CONTROL_INTEGRITY_ANALYST, ControlIntegrityAnalyst, DEFAULT_MAX_VERIFICATION_ARTIFACT_BYTES, DEFAULT_TRACE_ANALYST_KINDS, ExactAnalystRunExecutionError, FAILURE_MODE_KIND_SPEC, FINDING_SUBJECT_KINDS, FINDING_SUBJECT_SYNTAX, FindingsStore, IMPROVEMENT_KIND_SPEC, KIND_EXPECTED_SUBJECTS, KNOWLEDGE_GAP_KIND_SPEC, KNOWLEDGE_POISONING_KIND_SPEC, MAX_INCORRECT_BLOCKS, MAX_INCORRECT_BLOCK_STEPS, RAW_FINDING_SCHEMA_PROMPT, RawAnalystFindingSchema, SKILL_USAGE_ANALYST, adaptPublicBenchmarkFindings, agentRxBenchmarkCase, agentRxPredictionsToFindings, analystBenchmarkDependencyLockDigest, analystBenchmarkImplementationDigest, analystDefinitionAsymmetries, analystDefinitionProtocolSha256, analystInstructionsOverrideFromText, analystUsageReceiptFromPrimeUsage, appendVerificationArtifactsToOtlp, assertProposalFindings, behavioralAnalyst, bindAnalyst, buildDefaultAnalystRegistry, buildPrimePrompt, buildPrimeRepairPrompt, buildSkillUsageReport, buildTraceToolsForGroup, codeTraceBenchCase, codeTracerPredictionsToFindings, coerceJson, compareAnalystRunners, computeFindingId, createChatClient, createDspyRlmTraceEngine, createPrimeBenchmarkRunner, createPublicBenchmarkDirectRunner, createPublicBenchmarkRlmRunner, createSemanticConceptJudgeAdapter, createTraceAnalyst, decodeReplyRows, defaultIsMaterial, defineCustomAnalyst, defineTraceAnalyst, deriveEfficiencyFindings, diffFindings, effectiveAnalystProtocolSha256, emitControlIntegrityFindings, emitSkillUsageFindings, emptyPrimeRawUsage, emptyPublicBenchmarkRunner, evidenceRefsFromRawFinding, expandCodeTraceFailureBlocks, extractPrimeJsonObject, findingSubjectGrammarPromptFor, isProposalFinding, loadCodeTraceVerificationArtifacts, loadPublicBenchmarkRows, makeFinding, makeProposalFinding, mergePrimeRawUsage, nodeHttpPrimeBridgeTransport, normalizeAgentRxCategory, normalizeBenchmarkLabel, normalizePrimeUsage, parseFindingSubject, parseRawFinding, parseVerificationOutcome, preparePublicAnalystBenchmark, primeAnalystProtocolSha256, primeCodeTraceAnalystDefinition, primeProtocolSha256, primeReplyDefect, projectPrimeTrajectory, publicBenchmarkDistributions, publicBenchmarkProtocolSha256, publicBenchmarkRlmInstructions, publicBenchmarkSelectionReport, publicBenchmarkSystemPrompt, publicDirectAnalystDefinition, publicRlmAnalystDefinition, readAnalystBenchmarkArtifact, readAnalystInstructionsOverride, registryBenchmarkRunner, renderAgentRxCalibrationMarkdown, renderAnalystBenchmarkMarkdown, renderCodeTraceCalibrationMarkdown, renderFindingSubject, renderPriorFindings, resolveTraceAnalystLimits, rlmEngineLimits, roundAgentRxStep, runAnalystBenchmark, runAnalystBenchmarkCommand, runPrimeExchange, runTraceAnalyst, scoreAnalystFindings, selectPublicBenchmarkRows, stripCodeFences, summarizeAgentRxCalibration, summarizeAnalystBenchmarkRunner, summarizeCodeTraceCalibration, traceStoreEvidenceResolver };
|
|
378
|
+
export { AGENT_RX_UPSTREAM_REVISION, ANALYST_BENCHMARK_COST_LEDGER_FILE, ANALYST_BENCHMARK_DEPENDENCY_LOCK_DIGEST_ALGORITHM, ANALYST_BENCHMARK_DEPENDENCY_LOCK_FILES, ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256, ANALYST_BENCHMARK_EVIDENCE_DEPENDENCY_LOCK_SHA256, ANALYST_BENCHMARK_EVIDENCE_IMPLEMENTATION_SHA256, ANALYST_BENCHMARK_IMPLEMENTATION_DIGEST_ALGORITHM, ANALYST_BENCHMARK_IMPLEMENTATION_FILES, ANALYST_BENCHMARK_IMPLEMENTATION_SHA256, ANALYST_BENCHMARK_LOCAL_RECEIPT_FILE, ANALYST_BENCHMARK_MANIFEST_FILE, ANALYST_BENCHMARK_OBSERVATIONS_FILE, AnalystExpressivenessError, AnalystRegistry, CODE_TRACE_BENCH_ANALYST_PROMPT, CONTROL_INTEGRITY_ANALYST, ControlIntegrityAnalyst, DEFAULT_MAX_VERIFICATION_ARTIFACT_BYTES, DEFAULT_TRACE_ANALYST_KINDS, DEFAULT_TRACE_ANALYST_OUTPUT_TOKENS, ExactAnalystRunExecutionError, FAILURE_MODE_KIND_SPEC, FINDING_SUBJECT_KINDS, FINDING_SUBJECT_SYNTAX, FindingsStore, IMPROVEMENT_KIND_SPEC, KIND_EXPECTED_SUBJECTS, KNOWLEDGE_GAP_KIND_SPEC, KNOWLEDGE_POISONING_KIND_SPEC, MAX_INCORRECT_BLOCKS, MAX_INCORRECT_BLOCK_STEPS, RAW_FINDING_SCHEMA_PROMPT, RawAnalystFindingSchema, SKILL_USAGE_ANALYST, adaptPublicBenchmarkFindings, agentRxBenchmarkCase, agentRxPredictionsToFindings, analystBenchmarkDependencyLockDigest, analystBenchmarkImplementationDigest, analystDefinitionAsymmetries, analystDefinitionProtocolSha256, analystInstructionsOverrideFromText, analystUsageReceiptFromPrimeUsage, appendVerificationArtifactsToOtlp, assertProposalFindings, behavioralAnalyst, bindAnalyst, buildDefaultAnalystRegistry, buildPrimePrompt, buildPrimeRepairPrompt, buildSkillUsageReport, buildTraceToolsForGroup, codeTraceBenchCase, codeTracerPredictionsToFindings, coerceJson, compareAnalystRunners, computeFindingId, createChatClient, createChatTraceEngine, createDspyRlmTraceEngine, createPrimeBenchmarkRunner, createPublicBenchmarkDirectRunner, createPublicBenchmarkRlmRunner, createSemanticConceptJudgeAdapter, createTraceAnalyst, decodeReplyRows, defaultIsMaterial, defineCustomAnalyst, defineTraceAnalyst, deriveEfficiencyFindings, diffFindings, effectiveAnalystProtocolSha256, emitControlIntegrityFindings, emitSkillUsageFindings, emptyPrimeRawUsage, emptyPublicBenchmarkRunner, evidenceRefsFromRawFinding, expandCodeTraceFailureBlocks, extractPrimeJsonObject, findingSubjectGrammarPromptFor, isProposalFinding, loadCodeTraceVerificationArtifacts, loadPublicBenchmarkRows, makeFinding, makeProposalFinding, mergePrimeRawUsage, nodeHttpPrimeBridgeTransport, normalizeAgentRxCategory, normalizeBenchmarkLabel, normalizePrimeUsage, parseFindingSubject, parseRawFinding, parseVerificationOutcome, preparePublicAnalystBenchmark, primeAnalystProtocolSha256, primeCodeTraceAnalystDefinition, primeProtocolSha256, primeReplyDefect, projectPrimeTrajectory, publicBenchmarkDistributions, publicBenchmarkProtocolSha256, publicBenchmarkRlmInstructions, publicBenchmarkSelectionReport, publicBenchmarkSystemPrompt, publicDirectAnalystDefinition, publicRlmAnalystDefinition, readAnalystBenchmarkArtifact, readAnalystInstructionsOverride, registryBenchmarkRunner, renderAgentRxCalibrationMarkdown, renderAnalystBenchmarkMarkdown, renderCodeTraceCalibrationMarkdown, renderFindingSubject, renderPriorFindings, resolveTraceAnalystLimits, rlmEngineLimits, roundAgentRxStep, runAnalystBenchmark, runAnalystBenchmarkCommand, runPrimeExchange, runTraceAnalyst, scoreAnalystFindings, selectPublicBenchmarkRows, stripCodeFences, summarizeAgentRxCalibration, summarizeAnalystBenchmarkRunner, summarizeCodeTraceCalibration, traceStoreEvidenceResolver };
|
|
378
379
|
|
|
379
380
|
//# sourceMappingURL=index.js.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"index.js","names":[],"sources":["../../src/analyst/adapters.ts","../../src/analyst/bind.ts","../../src/analyst/define.ts","../../src/analyst/kinds/skill-usage.ts"],"sourcesContent":["/**\n * Adapter factory — lifts the semantic-concept judge into the Analyst\n * contract without re-implementing it.\n *\n * It builds an Analyst with a stable id (caller chooses; a default is given),\n * a version derived from the judge's version plus an adapter revision, and an\n * `analyze()` that calls the judge and lifts its output to `AnalystFinding[]`\n * through `makeFinding()`. The judge's `Severity` ('critical' | 'major' |\n * 'minor' | 'info') projects onto `AnalystSeverity`: 'major' becomes 'high'\n * and 'minor' becomes 'medium'.\n *\n * The adapter owns no state. Calling the factory twice is safe.\n */\n\nimport { CostLedger } from '../cost-ledger'\nimport type { Severity as LayerSeverity } from '../multi-layer-verifier'\nimport {\n runSemanticConceptJudge,\n SEMANTIC_CONCEPT_JUDGE_VERSION,\n type SemanticConceptJudgeInput,\n type SemanticConceptJudgeOptions,\n type SemanticConceptJudgeResult,\n} from '../semantic-concept-judge'\nimport type { Analyst, AnalystFinding, AnalystSeverity } from './types'\nimport { makeFinding } from './types'\nimport { settleUsageReceiptFromCostLedger, validateUsageSettlementTimeout } from './usage-receipt'\n\nconst ADAPTER_REV = '1'\n\n// ── Severity bridges ───────────────────────────────────────────────\n\nfunction liftSeverity(s: LayerSeverity): AnalystSeverity {\n switch (s) {\n case 'critical':\n return 'critical'\n case 'major':\n return 'high'\n case 'minor':\n return 'medium'\n case 'info':\n return 'info'\n }\n}\n\n// ── SemanticConceptJudge → Analyst ─────────────────────────────────\n\nexport interface SemanticConceptJudgeAdapterOpts {\n id?: string\n area?: string\n /** Registry context owns cancellation and the per-analyst cost ledger. */\n options: Omit<SemanticConceptJudgeOptions, 'costLedger' | 'signal'>\n /** Maximum post-cancellation wait for a provider receipt. Default 5 seconds. */\n settlementTimeoutMs?: number\n}\n\nexport function createSemanticConceptJudgeAdapter(\n opts: SemanticConceptJudgeAdapterOpts,\n): Analyst<SemanticConceptJudgeInput> {\n const id = opts.id ?? 'semantic-concept-judge'\n const area = opts.area ?? 'concept-coverage'\n const settlementTimeoutMs = validateUsageSettlementTimeout(opts.settlementTimeoutMs)\n return {\n id,\n description:\n 'Runs the semantic-concept judge and surfaces missing / weak concepts as findings.',\n inputKind: 'custom',\n cost: {\n kind: 'llm',\n models: opts.options.model ? [opts.options.model] : undefined,\n settlement_timeout_ms: settlementTimeoutMs,\n },\n version: `${SEMANTIC_CONCEPT_JUDGE_VERSION}-adapter-${ADAPTER_REV}`,\n async analyze(input, ctx) {\n const costLedger = new CostLedger(ctx.budgetUsd)\n let result: SemanticConceptJudgeResult\n try {\n result = await runSemanticConceptJudge(input, {\n ...opts.options,\n costLedger,\n signal: ctx.signal,\n })\n } finally {\n const usage = await settleUsageReceiptFromCostLedger(costLedger, {\n channel: 'judge',\n timeoutMs: settlementTimeoutMs,\n })\n if (!usage.settled) {\n ctx.log?.('semantic-concept judge provider settlement timed out', {\n pending_calls: usage.pendingCalls,\n timeout_ms: settlementTimeoutMs,\n })\n }\n ctx.recordUsage?.(usage.receipt)\n }\n if (!result.available) {\n return [\n makeFinding({\n analyst_id: id,\n area,\n claim: 'semantic-concept judge unavailable',\n rationale: result.error,\n severity: 'info',\n confidence: 1,\n evidence_refs: [],\n metadata: { reason: result.error },\n }),\n ]\n }\n const out: AnalystFinding[] = []\n for (const f of result.findings) {\n // Only surface gaps: missing concepts or low scores. Concepts at\n // 7+/10 with present=true are not findings — they're successes.\n if (f.present && f.score >= 7) continue\n out.push(\n makeFinding({\n analyst_id: id,\n area,\n subject: f.concept,\n claim: f.present\n ? `concept \"${f.concept}\" is weak (${f.score}/10)`\n : `concept \"${f.concept}\" is missing`,\n rationale: f.evidence,\n severity: liftSeverity(f.severity),\n confidence: 0.85,\n evidence_refs: [{ kind: 'artifact', uri: 'inline:evidence', excerpt: f.evidence }],\n metadata: {\n concept: f.concept,\n present: f.present,\n score_10: f.score,\n },\n }),\n )\n }\n return out\n },\n }\n}\n","/**\n * bindAnalyst — compile an `AnalystDefinition` into a runnable arm.\n *\n * The definition declares protocol (question, task text, reply grammar,\n * evidence projection, budget, repair turns); the transport binding supplies\n * execution (a bridge endpoint, or the caller-owned model path). Each\n * projection mode dispatches to the same strategy its arm's entry point runs,\n * so a compiled definition and the historical creator send byte-identical\n * requests — the definition parity suite asserts this per arm.\n *\n * A projection × transport pair with no strategy fails loud with\n * `AnalystExpressivenessError` naming the pair: an arm shape this layer cannot\n * express must be reported, never approximated.\n */\n\nimport type { CustomTokenPricing } from '../cost-ledger'\nimport type { AnalystBenchmarkRunner } from './benchmark'\nimport { runChunkedAnalystDefinition } from './benchmark-public-model'\nimport { runReplVariableAnalystDefinition } from './benchmark-public-rlm'\nimport type { PublicAnalystBenchmarkModelConfig } from './benchmark-public-types'\nimport { runInlineAnalystDefinition } from './benchmark-runner-prime'\nimport { type AnalystDefinition, AnalystExpressivenessError } from './definition'\nimport type { PrimeBridgeTransport } from './prime-bridge-transport'\nimport type { AnalystRunInputs } from './types'\n\n/** Execution half of a binding: who actually reaches the model. */\nexport type AnalystTransportBinding =\n | {\n /** An OpenAI-compatible cli-bridge endpoint (inline projections). */\n readonly kind: 'prime-bridge'\n readonly baseUrl: string\n readonly model: string\n readonly transport?: PrimeBridgeTransport\n readonly pricing?: CustomTokenPricing\n }\n | {\n /** The caller-owned model path (chunked and repl-variable projections). */\n readonly kind: 'model-owner'\n readonly config: PublicAnalystBenchmarkModelConfig\n }\n\n/**\n * Compile a definition plus a transport binding into a runnable arm. Dispatch\n * is total over the expressible pairs; everything else is a loud refusal.\n */\nexport function bindAnalyst<TRow, TAssignment, TBlock>(\n definition: AnalystDefinition<TRow, TAssignment, TBlock>,\n transports: AnalystTransportBinding,\n): AnalystBenchmarkRunner<AnalystRunInputs> {\n const mode = definition.projection.mode\n if (mode === 'inline' && transports.kind === 'prime-bridge') {\n return runInlineAnalystDefinition(definition, {\n baseUrl: transports.baseUrl,\n model: transports.model,\n ...(transports.transport ? { transport: transports.transport } : {}),\n ...(transports.pricing ? { pricing: transports.pricing } : {}),\n })\n }\n if (mode === 'chunked' && transports.kind === 'model-owner') {\n return runChunkedAnalystDefinition(definition, transports.config)\n }\n if (mode === 'repl-variable' && transports.kind === 'model-owner') {\n return runReplVariableAnalystDefinition(\n // The repl-variable strategy is written against the engine's raw-finding\n // row type; the definition's own row declaration carries it.\n definition as Parameters<typeof runReplVariableAnalystDefinition>[0],\n transports.config,\n )\n }\n throw new AnalystExpressivenessError(\n `no strategy compiles a '${mode}' projection over a '${transports.kind}' transport; ` +\n `definition '${definition.id}' cannot be expressed by this layer yet`,\n )\n}\n","import type { TraceAnalysisStore } from '../trace-analyst/store'\nimport type { ExactCapableAnalyst } from './exact-types'\nimport type { TraceAnalystDefinition } from './kind-factory'\nimport type { Analyst, AnalystCost } from './types'\n\n/**\n * Define a reusable trace-research question.\n *\n * The returned value contains no model, credentials, or execution state. Bind\n * it to any TraceAnalysisEngine with `runTraceAnalyst` or\n * `createTraceAnalyst`.\n */\nexport function defineTraceAnalyst(definition: TraceAnalystDefinition): TraceAnalystDefinition {\n for (const [name, value] of [\n ['id', definition.id],\n ['description', definition.description],\n ['area', definition.area],\n ['version', definition.version],\n ['instructions', definition.instructions],\n ] as const) {\n if (typeof value !== 'string' || !value.trim()) {\n throw new TypeError(`defineTraceAnalyst: ${name} must be a non-empty string`)\n }\n }\n return {\n ...definition,\n limits: definition.limits ? { ...definition.limits } : undefined,\n }\n}\n\nexport interface DefineCustomAnalystOptions {\n id: string\n description: string\n version?: string\n cost: AnalystCost\n analyze: Analyst<TraceAnalysisStore>['analyze']\n}\n\nexport interface DefineExactCustomAnalystOptions extends DefineCustomAnalystOptions {\n /** Canonical JSON for behavior knobs not already bound by `version`. */\n executionConfig: Readonly<Record<string, unknown>>\n}\n\n/** Construct a registrable analyst from a hand-written analyze function. */\nexport function defineCustomAnalyst(\n options: DefineExactCustomAnalystOptions,\n): ExactCapableAnalyst<TraceAnalysisStore>\nexport function defineCustomAnalyst(\n options: DefineCustomAnalystOptions,\n): Analyst<TraceAnalysisStore>\nexport function defineCustomAnalyst(\n options: DefineCustomAnalystOptions | DefineExactCustomAnalystOptions,\n): Analyst<TraceAnalysisStore> | ExactCapableAnalyst<TraceAnalysisStore> {\n if (!options.id.trim()) throw new TypeError('defineCustomAnalyst: id must not be empty')\n if (!options.description.trim()) {\n throw new TypeError('defineCustomAnalyst: description must not be empty')\n }\n if (options.cost === undefined) {\n throw new TypeError('defineCustomAnalyst: cost must be declared')\n }\n if (\n 'executionConfig' in options &&\n (!options.executionConfig ||\n typeof options.executionConfig !== 'object' ||\n Array.isArray(options.executionConfig))\n ) {\n throw new TypeError('defineCustomAnalyst: executionConfig must be an object')\n }\n return {\n id: options.id,\n description: options.description,\n version: options.version ?? '1.0.0',\n inputKind: 'trace-store',\n cost: options.cost,\n analyze: options.analyze,\n ...('executionConfig' in options ? { executionConfig: options.executionConfig } : {}),\n }\n}\n","/**\n * Skill-usage analyst — a DETERMINISTIC `Analyst` over a Claude/Codex skill\n * library + its trace corpus. Unlike the trace-store kinds (failure-mode,\n * improvement, ...) this kind calls no LLM: it mines real usage and skill\n * structure and emits findings by rule.\n *\n * It exists because the naive \"Skill-tool invocation count\" lies low — it\n * misses orchestrated sub-dispatch (a leaf skill run BY /pursue or /governor\n * logs under the parent), slash-command entry, local-script bypass, and\n * on-disk artifacts. The 2026-05-30 skill audit found 39/53 skills at zero\n * direct invocations, yet only one was a genuine cut: the rest were\n * measurement-invisible or discovery-limited. This analyst encodes that\n * lesson as a multi-signal usage model so a cheap repeatable pass can keep\n * the library honest, and so the expensive audit workflow's verdicts can\n * GEPA-distill it toward agreement (see `gold/skill-verdicts.gold.jsonl`).\n *\n * Report-building (`buildSkillUsageReport`, an fs scan) is separated from\n * finding emission (`SkillUsageAnalyst.analyze`, pure) so the slow scan runs\n * once at the registry boundary and the rule logic stays unit-testable.\n */\n\nimport { type Dirent, existsSync, readdirSync, readFileSync, statSync } from 'node:fs'\nimport { join } from 'node:path'\nimport type { ExactCapableAnalyst } from '../exact-types'\nimport type { AnalystContext, AnalystFinding, AnalystSeverity } from '../types'\nimport { computeFindingId } from '../types'\n\n// ── Input model ──────────────────────────────────────────────────────\n\nexport type SkillKind = 'public' | 'private'\n\n/** One skill's multi-signal usage + structure. All counts are deterministic. */\nexport interface SkillUsageRecord {\n name: string\n kind: SkillKind\n /** Absolute path to the skill's SKILL.md. */\n path: string\n lines: number\n /** `\"skill\":\"<name>\"` Skill-tool invocations across the trace corpus. */\n directInvocations: number\n /** `<command-name>/<name>` slash invocations across the trace corpus. */\n slashInvocations: number\n /** Sibling skills whose SKILL.md dispatches to this one (`/<name>`). Proxy\n * for orchestrated sub-dispatch the per-skill counter cannot see. */\n inboundRefs: number\n /** On-disk artifacts attributable to the skill (e.g. `.evolve/<name>/**`). */\n artifactCount: number\n /** Tangle-private reference count in the body (leak signal for public skills). */\n tanglePrivateRefs: number\n hasReferencesDir: boolean\n hasEvalsDir: boolean\n /** Body mentions `skill-runs.jsonl` (visible to /reflect + /governor). */\n logsRuns: boolean\n /** Description carries an explicit `Triggers:` clause / trigger phrases. */\n hasTriggerPhrases: boolean\n}\n\nexport interface SkillUsageReport {\n generatedFromTraces: number\n records: SkillUsageRecord[]\n}\n\nexport interface SkillUsageScanConfig {\n /** Dirs holding `*.jsonl` transcripts (Claude `~/.claude/projects`, Codex sessions). */\n transcriptDirs: string[]\n /** Skill roots to scan; each dir directly under `root` with a `SKILL.md` is a skill. */\n skillRoots: { root: string; kind: SkillKind }[]\n /** Roots scanned for `<root>/.evolve/<skill>` artifact dirs. */\n artifactRoots?: string[]\n /** Token-prefixed mappings: skill name → extra artifact subpaths under an artifactRoot\n * (e.g. reflect → `.evolve/reflections`). Catches non-eponymous artifact dirs. */\n artifactAliases?: Record<string, string[]>\n /** Cap files read per transcript dir (bounds a huge corpus); 0 = unbounded. */\n maxTranscriptsPerDir?: number\n}\n\n// ── Deterministic thresholds ─────────────────────────────────────────\n\n/** Anthropic's authoring guidance keeps SKILL.md short; past this with no\n * `references/` split the body burns context budget every session. */\nconst BLOAT_LINE_THRESHOLD = 300\n\nconst TANGLE_PRIVATE_RE =\n /\\b(cli-bridge|tangletools|ops-board|drew-gtr-pro|@tangle-network\\/|~\\/company|tangle\\.tools|gtm-agent)\\b|\\bkimi\\b|\\btcloud\\b/gi\nconst TRIGGER_RE = /triggers?\\s*[:-]/i\n\n// ── Report builder (fs scan — slow, runs once at the registry boundary) ──\n\nfunction listSkillDirs(root: string): { name: string; path: string }[] {\n if (!existsSync(root)) return []\n const out: { name: string; path: string }[] = []\n for (const entry of readdirSync(root, { withFileTypes: true })) {\n if (!entry.isDirectory() && !entry.isSymbolicLink()) continue\n const skillMd = join(root, entry.name, 'SKILL.md')\n if (existsSync(skillMd)) out.push({ name: entry.name, path: skillMd })\n }\n return out\n}\n\nfunction walkJsonl(dir: string, cap: number): string[] {\n if (!existsSync(dir)) return []\n const files: string[] = []\n const stack = [dir]\n while (stack.length) {\n const cur = stack.pop()!\n let entries: Dirent[]\n try {\n entries = readdirSync(cur, { withFileTypes: true })\n } catch {\n continue\n }\n for (const e of entries) {\n const full = join(cur, e.name)\n if (e.isDirectory()) stack.push(full)\n else if (e.name.endsWith('.jsonl')) {\n files.push(full)\n if (cap > 0 && files.length >= cap) return files\n }\n }\n }\n return files\n}\n\nfunction frontmatterDescription(body: string): string {\n const fm = /^---\\n([\\s\\S]*?)\\n---/.exec(body)\n const block = fm?.[1] ?? ''\n const m = /description:\\s*(.+)/i.exec(block)\n return m?.[1] ?? ''\n}\n\nfunction countArtifacts(roots: string[], name: string, aliases: string[]): number {\n let n = 0\n for (const root of roots) {\n const candidates = [join(root, '.evolve', name), ...aliases.map((a) => join(root, a))]\n for (const dir of candidates) {\n if (!existsSync(dir)) continue\n try {\n if (statSync(dir).isDirectory()) n += readdirSync(dir).length\n else n += 1\n } catch {\n /* unreadable — skip */\n }\n }\n }\n return n\n}\n\n/** Scan the corpus + skill roots into a {@link SkillUsageReport}. Deterministic. */\nexport function buildSkillUsageReport(config: SkillUsageScanConfig): SkillUsageReport {\n const skills = config.skillRoots.flatMap(({ root, kind }) =>\n listSkillDirs(root).map((s) => ({ ...s, kind })),\n )\n const names = skills.map((s) => s.name)\n\n // One pass over the corpus accumulating direct + slash counts per skill.\n const direct = new Map<string, number>(names.map((n) => [n, 0]))\n const slash = new Map<string, number>(names.map((n) => [n, 0]))\n const skillRe = /\"skill\"\\s*:\\s*\"([a-z0-9_:-]+)\"/g\n const cmdRe = /<command-name>\\/?([a-z0-9_:-]+)<\\/command-name>/g\n let transcripts = 0\n for (const dir of config.transcriptDirs) {\n for (const file of walkJsonl(dir, config.maxTranscriptsPerDir ?? 0)) {\n transcripts += 1\n let data: string\n try {\n data = readFileSync(file, 'utf8')\n } catch {\n continue\n }\n for (const m of data.matchAll(skillRe)) {\n const g = m[1]\n if (!g) continue\n const n = g.split(':').pop() ?? g\n const prev = direct.get(n)\n if (prev !== undefined) direct.set(n, prev + 1)\n }\n for (const m of data.matchAll(cmdRe)) {\n const g = m[1]\n if (g === undefined) continue\n const prev = slash.get(g)\n if (prev !== undefined) slash.set(g, prev + 1)\n }\n }\n }\n\n // Read each skill body once; compute structure + inbound refs across siblings.\n const bodies = new Map<string, string>()\n for (const s of skills) {\n try {\n bodies.set(s.name, readFileSync(s.path, 'utf8'))\n } catch {\n bodies.set(s.name, '')\n }\n }\n const inbound = new Map<string, number>(names.map((n) => [n, 0]))\n for (const target of names) {\n const ref = new RegExp(`/${target}\\\\b|\\\\[\\\\[${target}\\\\]\\\\]`)\n for (const s of skills) {\n if (s.name === target) continue\n if (ref.test(bodies.get(s.name) ?? '')) inbound.set(target, inbound.get(target)! + 1)\n }\n }\n\n const records: SkillUsageRecord[] = skills.map((s) => {\n const body = bodies.get(s.name) ?? ''\n const dir = s.path.replace(/\\/SKILL\\.md$/, '')\n return {\n name: s.name,\n kind: s.kind,\n path: s.path,\n lines: body ? body.split('\\n').length : 0,\n directInvocations: direct.get(s.name) ?? 0,\n slashInvocations: slash.get(s.name) ?? 0,\n inboundRefs: inbound.get(s.name) ?? 0,\n artifactCount: countArtifacts(\n config.artifactRoots ?? [],\n s.name,\n config.artifactAliases?.[s.name] ?? [],\n ),\n tanglePrivateRefs: (body.match(TANGLE_PRIVATE_RE) ?? []).length,\n hasReferencesDir: existsSync(join(dir, 'references')),\n hasEvalsDir: existsSync(join(dir, 'evals')),\n logsRuns: body.includes('skill-runs.jsonl'),\n hasTriggerPhrases: TRIGGER_RE.test(frontmatterDescription(body) || body.slice(0, 600)),\n }\n })\n return { generatedFromTraces: transcripts, records }\n}\n\n// ── Finding emission (pure — unit-testable, no LLM, no fs) ────────────\n\nconst ANALYST_ID = 'skill-usage'\n\nfunction finding(\n area: string,\n subject: string,\n claim: string,\n severity: AnalystSeverity,\n confidence: number,\n producedAt: string,\n recommended: string,\n evidenceUri: string,\n rationale?: string,\n): AnalystFinding {\n return {\n schema_version: '1.0.0',\n finding_id: computeFindingId({ analyst_id: ANALYST_ID, area, subject, claim }),\n analyst_id: ANALYST_ID,\n produced_at: producedAt,\n severity,\n area,\n claim,\n rationale,\n evidence_refs: [{ kind: 'artifact', uri: evidenceUri }],\n recommended_action: recommended,\n confidence,\n subject,\n }\n}\n\n/** Pure rule pass over a report → findings. Exported for direct/unit use. */\nexport function emitSkillUsageFindings(\n report: SkillUsageReport,\n producedAt: string,\n): AnalystFinding[] {\n const out: AnalystFinding[] = []\n for (const r of report.records) {\n const directTotal = r.directInvocations + r.slashInvocations\n const trueUsage = directTotal + r.inboundRefs + r.artifactCount\n\n // 1. Dead: no usage signal of ANY kind. The only real deprecation candidate.\n if (trueUsage === 0) {\n out.push(\n finding(\n 'skill-usage',\n r.name,\n `Skill '${r.name}' has zero usage across all signals (direct, slash, inbound-refs, artifacts)`,\n 'high',\n 0.6,\n producedAt,\n 'Confirm the skill covers a real recurring job; if not, deprecate. Zero true usage is the only deterministic deprecation candidate.',\n r.path,\n 'No Skill-tool call, no slash invocation, no sibling dispatches to it, and no on-disk artifacts.',\n ),\n )\n } else if (directTotal === 0 && r.inboundRefs + r.artifactCount > 0) {\n // 2. Measurement-invisible: real use via orchestration/artifacts, never invoked directly.\n out.push(\n finding(\n 'skill-usage',\n r.name,\n `Skill '${r.name}' shows 0 direct invocations but is used via orchestration/artifacts (inbound=${r.inboundRefs}, artifacts=${r.artifactCount})`,\n 'info',\n 0.8,\n producedAt,\n 'Do NOT treat as unused — usage is real but logged under parent skills or on disk. Strengthen direct-invocation discovery only if direct use is desired.',\n r.path,\n 'The Skill-tool counter undercounts orchestrated/chained leaf skills.',\n ),\n )\n }\n\n // 3. Discovery gap: low direct use AND weak trigger surface.\n if (directTotal <= 2 && !r.hasTriggerPhrases) {\n out.push(\n finding(\n 'discoverability',\n r.name,\n `Skill '${r.name}' is rarely invoked directly and its description has no explicit trigger phrases`,\n 'medium',\n 0.7,\n producedAt,\n 'Add a `Triggers:` clause with verbatim user phrases to the frontmatter description so the model auto-invokes it.',\n r.path,\n ),\n )\n }\n\n // 4. Public-repo leak.\n if (r.kind === 'public' && r.tanglePrivateRefs > 0) {\n out.push(\n finding(\n 'safety',\n r.name,\n `Public skill '${r.name}' carries ${r.tanglePrivateRefs} Tangle-private reference(s)`,\n 'high',\n 0.75,\n producedAt,\n 'Sanitize incidental internal refs (cli-bridge/kimi/tcloud/~company/private repos) or relocate to a private repo. Verify @tangle-network/* refs are to PUBLISHED packages before treating as a leak.',\n r.path,\n ),\n )\n }\n\n // 5. Bloat / no progressive disclosure.\n if (r.lines > BLOAT_LINE_THRESHOLD && !r.hasReferencesDir) {\n out.push(\n finding(\n 'maintainability',\n r.name,\n `Skill '${r.name}' is ${r.lines} lines with no references/ split (progressive disclosure)`,\n 'medium',\n 0.8,\n producedAt,\n `Split detail into references/ loaded on demand; keep SKILL.md a short overview. ${r.lines} lines load into every session's context budget.`,\n r.path,\n ),\n )\n }\n\n // 6. No evals (Anthropic's \">=3 evals before docs\" rule).\n if (!r.hasEvalsDir) {\n out.push(\n finding(\n 'data-quality',\n r.name,\n `Skill '${r.name}' ships no evals/`,\n 'low',\n 0.6,\n producedAt,\n 'Add evals/evals.json with >=3 scenarios proving the skill beats baseline; gives regression coverage.',\n r.path,\n ),\n )\n }\n\n // 7. No run logging → invisible to /reflect and /governor.\n if (!r.logsRuns) {\n out.push(\n finding(\n 'observability',\n r.name,\n `Skill '${r.name}' never appends to .evolve/skill-runs.jsonl`,\n 'low',\n 0.55,\n producedAt,\n 'Append one run line to .evolve/skill-runs.jsonl on completion, or declare it a non-logging leaf, so the self-improvement loop can see it ran.',\n r.path,\n ),\n )\n }\n }\n return out\n}\n\n// ── The Analyst ──────────────────────────────────────────────────────\n\nclass SkillUsageAnalyst implements ExactCapableAnalyst<SkillUsageReport> {\n readonly id = ANALYST_ID\n readonly description =\n 'Deterministic multi-signal skill-usage analysis: flags dead skills, measurement-invisible (orchestrated) usage, discovery gaps, public-repo leaks, bloat, missing evals, and missing run-logging.'\n readonly inputKind = 'custom' as const\n readonly cost = { kind: 'deterministic' as const, est_usd_per_run: 0 }\n readonly version = '1.0.0'\n readonly executionConfig = {\n kind: 'skill-usage',\n bloat_line_threshold: BLOAT_LINE_THRESHOLD,\n produced_at_source: 'tags.producedAt-or-system-clock',\n } as const\n\n async analyze(input: SkillUsageReport, ctx: AnalystContext): Promise<AnalystFinding[]> {\n const producedAt = ctx.tags?.producedAt ?? new Date().toISOString()\n ctx.log?.(\n `skill-usage: ${input.records.length} skills over ${input.generatedFromTraces} transcripts`,\n )\n return emitSkillUsageFindings(input, producedAt)\n }\n}\n\nexport const SKILL_USAGE_ANALYST = new SkillUsageAnalyst()\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;AA2BA,MAAM,cAAc;AAIpB,SAAS,aAAa,GAAmC;CACvD,QAAQ,GAAR;EACE,KAAK,YACH,OAAO;EACT,KAAK,SACH,OAAO;EACT,KAAK,SACH,OAAO;EACT,KAAK,QACH,OAAO;CACX;AACF;AAaA,SAAgB,kCACd,MACoC;CACpC,MAAM,KAAK,KAAK,MAAM;CACtB,MAAM,OAAO,KAAK,QAAQ;CAC1B,MAAM,sBAAsB,+BAA+B,KAAK,mBAAmB;CACnF,OAAO;EACL;EACA,aACE;EACF,WAAW;EACX,MAAM;GACJ,MAAM;GACN,QAAQ,KAAK,QAAQ,QAAQ,CAAC,KAAK,QAAQ,KAAK,IAAI,KAAA;GACpD,uBAAuB;EACzB;EACA,SAAS,GAAG,+BAA+B,WAAW;EACtD,MAAM,QAAQ,OAAO,KAAK;GACxB,MAAM,aAAa,IAAI,WAAW,IAAI,SAAS;GAC/C,IAAI;GACJ,IAAI;IACF,SAAS,MAAM,wBAAwB,OAAO;KAC5C,GAAG,KAAK;KACR;KACA,QAAQ,IAAI;IACd,CAAC;GACH,UAAU;IACR,MAAM,QAAQ,MAAM,iCAAiC,YAAY;KAC/D,SAAS;KACT,WAAW;IACb,CAAC;IACD,IAAI,CAAC,MAAM,SACT,IAAI,MAAM,wDAAwD;KAChE,eAAe,MAAM;KACrB,YAAY;IACd,CAAC;IAEH,IAAI,cAAc,MAAM,OAAO;GACjC;GACA,IAAI,CAAC,OAAO,WACV,OAAO,CACL,YAAY;IACV,YAAY;IACZ;IACA,OAAO;IACP,WAAW,OAAO;IAClB,UAAU;IACV,YAAY;IACZ,eAAe,CAAC;IAChB,UAAU,EAAE,QAAQ,OAAO,MAAM;GACnC,CAAC,CACH;GAEF,MAAM,MAAwB,CAAC;GAC/B,KAAK,MAAM,KAAK,OAAO,UAAU;IAG/B,IAAI,EAAE,WAAW,EAAE,SAAS,GAAG;IAC/B,IAAI,KACF,YAAY;KACV,YAAY;KACZ;KACA,SAAS,EAAE;KACX,OAAO,EAAE,UACL,YAAY,EAAE,QAAQ,aAAa,EAAE,MAAM,QAC3C,YAAY,EAAE,QAAQ;KAC1B,WAAW,EAAE;KACb,UAAU,aAAa,EAAE,QAAQ;KACjC,YAAY;KACZ,eAAe,CAAC;MAAE,MAAM;MAAY,KAAK;MAAmB,SAAS,EAAE;KAAS,CAAC;KACjF,UAAU;MACR,SAAS,EAAE;MACX,SAAS,EAAE;MACX,UAAU,EAAE;KACd;IACF,CAAC,CACH;GACF;GACA,OAAO;EACT;CACF;AACF;;;;;;;AC3FA,SAAgB,YACd,YACA,YAC0C;CAC1C,MAAM,OAAO,WAAW,WAAW;CACnC,IAAI,SAAS,YAAY,WAAW,SAAS,gBAC3C,OAAO,2BAA2B,YAAY;EAC5C,SAAS,WAAW;EACpB,OAAO,WAAW;EAClB,GAAI,WAAW,YAAY,EAAE,WAAW,WAAW,UAAU,IAAI,CAAC;EAClE,GAAI,WAAW,UAAU,EAAE,SAAS,WAAW,QAAQ,IAAI,CAAC;CAC9D,CAAC;CAEH,IAAI,SAAS,aAAa,WAAW,SAAS,eAC5C,OAAO,4BAA4B,YAAY,WAAW,MAAM;CAElE,IAAI,SAAS,mBAAmB,WAAW,SAAS,eAClD,OAAO,iCAGL,YACA,WAAW,MACb;CAEF,MAAM,IAAI,2BACR,2BAA2B,KAAK,uBAAuB,WAAW,KAAK,2BACtD,WAAW,GAAG,wCACjC;AACF;;;;;;;;;;AC7DA,SAAgB,mBAAmB,YAA4D;CAC7F,KAAK,MAAM,CAAC,MAAM,UAAU;EAC1B,CAAC,MAAM,WAAW,EAAE;EACpB,CAAC,eAAe,WAAW,WAAW;EACtC,CAAC,QAAQ,WAAW,IAAI;EACxB,CAAC,WAAW,WAAW,OAAO;EAC9B,CAAC,gBAAgB,WAAW,YAAY;CAC1C,GACE,IAAI,OAAO,UAAU,YAAY,CAAC,MAAM,KAAK,GAC3C,MAAM,IAAI,UAAU,uBAAuB,KAAK,4BAA4B;CAGhF,OAAO;EACL,GAAG;EACH,QAAQ,WAAW,SAAS,EAAE,GAAG,WAAW,OAAO,IAAI,KAAA;CACzD;AACF;AAsBA,SAAgB,oBACd,SACuE;CACvE,IAAI,CAAC,QAAQ,GAAG,KAAK,GAAG,MAAM,IAAI,UAAU,2CAA2C;CACvF,IAAI,CAAC,QAAQ,YAAY,KAAK,GAC5B,MAAM,IAAI,UAAU,oDAAoD;CAE1E,IAAI,QAAQ,SAAS,KAAA,GACnB,MAAM,IAAI,UAAU,4CAA4C;CAElE,IACE,qBAAqB,YACpB,CAAC,QAAQ,mBACR,OAAO,QAAQ,oBAAoB,YACnC,MAAM,QAAQ,QAAQ,eAAe,IAEvC,MAAM,IAAI,UAAU,wDAAwD;CAE9E,OAAO;EACL,IAAI,QAAQ;EACZ,aAAa,QAAQ;EACrB,SAAS,QAAQ,WAAW;EAC5B,WAAW;EACX,MAAM,QAAQ;EACd,SAAS,QAAQ;EACjB,GAAI,qBAAqB,UAAU,EAAE,iBAAiB,QAAQ,gBAAgB,IAAI,CAAC;CACrF;AACF;;;;;;;;;;;;;;;;;;;;;;;;;ACGA,MAAM,uBAAuB;AAE7B,MAAM,oBACJ;AACF,MAAM,aAAa;AAInB,SAAS,cAAc,MAAgD;CACrE,IAAI,CAAC,WAAW,IAAI,GAAG,OAAO,CAAC;CAC/B,MAAM,MAAwC,CAAC;CAC/C,KAAK,MAAM,SAAS,YAAY,MAAM,EAAE,eAAe,KAAK,CAAC,GAAG;EAC9D,IAAI,CAAC,MAAM,YAAY,KAAK,CAAC,MAAM,eAAe,GAAG;EACrD,MAAM,UAAU,KAAK,MAAM,MAAM,MAAM,UAAU;EACjD,IAAI,WAAW,OAAO,GAAG,IAAI,KAAK;GAAE,MAAM,MAAM;GAAM,MAAM;EAAQ,CAAC;CACvE;CACA,OAAO;AACT;AAEA,SAAS,UAAU,KAAa,KAAuB;CACrD,IAAI,CAAC,WAAW,GAAG,GAAG,OAAO,CAAC;CAC9B,MAAM,QAAkB,CAAC;CACzB,MAAM,QAAQ,CAAC,GAAG;CAClB,OAAO,MAAM,QAAQ;EACnB,MAAM,MAAM,MAAM,IAAI;EACtB,IAAI;EACJ,IAAI;GACF,UAAU,YAAY,KAAK,EAAE,eAAe,KAAK,CAAC;EACpD,QAAQ;GACN;EACF;EACA,KAAK,MAAM,KAAK,SAAS;GACvB,MAAM,OAAO,KAAK,KAAK,EAAE,IAAI;GAC7B,IAAI,EAAE,YAAY,GAAG,MAAM,KAAK,IAAI;QAC/B,IAAI,EAAE,KAAK,SAAS,QAAQ,GAAG;IAClC,MAAM,KAAK,IAAI;IACf,IAAI,MAAM,KAAK,MAAM,UAAU,KAAK,OAAO;GAC7C;EACF;CACF;CACA,OAAO;AACT;AAEA,SAAS,uBAAuB,MAAsB;CAEpD,MAAM,QADK,wBAAwB,KAAK,IACzB,CAAC,GAAG,MAAM;CAEzB,OADU,uBAAuB,KAAK,KAC/B,CAAC,GAAG,MAAM;AACnB;AAEA,SAAS,eAAe,OAAiB,MAAc,SAA2B;CAChF,IAAI,IAAI;CACR,KAAK,MAAM,QAAQ,OAAO;EACxB,MAAM,aAAa,CAAC,KAAK,MAAM,WAAW,IAAI,GAAG,GAAG,QAAQ,KAAK,MAAM,KAAK,MAAM,CAAC,CAAC,CAAC;EACrF,KAAK,MAAM,OAAO,YAAY;GAC5B,IAAI,CAAC,WAAW,GAAG,GAAG;GACtB,IAAI;IACF,IAAI,SAAS,GAAG,CAAC,CAAC,YAAY,GAAG,KAAK,YAAY,GAAG,CAAC,CAAC;SAClD,KAAK;GACZ,QAAQ,CAER;EACF;CACF;CACA,OAAO;AACT;;AAGA,SAAgB,sBAAsB,QAAgD;CACpF,MAAM,SAAS,OAAO,WAAW,SAAS,EAAE,MAAM,WAChD,cAAc,IAAI,CAAC,CAAC,KAAK,OAAO;EAAE,GAAG;EAAG;CAAK,EAAE,CACjD;CACA,MAAM,QAAQ,OAAO,KAAK,MAAM,EAAE,IAAI;CAGtC,MAAM,SAAS,IAAI,IAAoB,MAAM,KAAK,MAAM,CAAC,GAAG,CAAC,CAAC,CAAC;CAC/D,MAAM,QAAQ,IAAI,IAAoB,MAAM,KAAK,MAAM,CAAC,GAAG,CAAC,CAAC,CAAC;CAC9D,MAAM,UAAU;CAChB,MAAM,QAAQ;CACd,IAAI,cAAc;CAClB,KAAK,MAAM,OAAO,OAAO,gBACvB,KAAK,MAAM,QAAQ,UAAU,KAAK,OAAO,wBAAwB,CAAC,GAAG;EACnE,eAAe;EACf,IAAI;EACJ,IAAI;GACF,OAAO,aAAa,MAAM,MAAM;EAClC,QAAQ;GACN;EACF;EACA,KAAK,MAAM,KAAK,KAAK,SAAS,OAAO,GAAG;GACtC,MAAM,IAAI,EAAE;GACZ,IAAI,CAAC,GAAG;GACR,MAAM,IAAI,EAAE,MAAM,GAAG,CAAC,CAAC,IAAI,KAAK;GAChC,MAAM,OAAO,OAAO,IAAI,CAAC;GACzB,IAAI,SAAS,KAAA,GAAW,OAAO,IAAI,GAAG,OAAO,CAAC;EAChD;EACA,KAAK,MAAM,KAAK,KAAK,SAAS,KAAK,GAAG;GACpC,MAAM,IAAI,EAAE;GACZ,IAAI,MAAM,KAAA,GAAW;GACrB,MAAM,OAAO,MAAM,IAAI,CAAC;GACxB,IAAI,SAAS,KAAA,GAAW,MAAM,IAAI,GAAG,OAAO,CAAC;EAC/C;CACF;CAIF,MAAM,yBAAS,IAAI,IAAoB;CACvC,KAAK,MAAM,KAAK,QACd,IAAI;EACF,OAAO,IAAI,EAAE,MAAM,aAAa,EAAE,MAAM,MAAM,CAAC;CACjD,QAAQ;EACN,OAAO,IAAI,EAAE,MAAM,EAAE;CACvB;CAEF,MAAM,UAAU,IAAI,IAAoB,MAAM,KAAK,MAAM,CAAC,GAAG,CAAC,CAAC,CAAC;CAChE,KAAK,MAAM,UAAU,OAAO;EAC1B,MAAM,MAAM,IAAI,OAAO,IAAI,OAAO,YAAY,OAAO,OAAO;EAC5D,KAAK,MAAM,KAAK,QAAQ;GACtB,IAAI,EAAE,SAAS,QAAQ;GACvB,IAAI,IAAI,KAAK,OAAO,IAAI,EAAE,IAAI,KAAK,EAAE,GAAG,QAAQ,IAAI,QAAQ,QAAQ,IAAI,MAAM,IAAK,CAAC;EACtF;CACF;CAEA,MAAM,UAA8B,OAAO,KAAK,MAAM;EACpD,MAAM,OAAO,OAAO,IAAI,EAAE,IAAI,KAAK;EACnC,MAAM,MAAM,EAAE,KAAK,QAAQ,gBAAgB,EAAE;EAC7C,OAAO;GACL,MAAM,EAAE;GACR,MAAM,EAAE;GACR,MAAM,EAAE;GACR,OAAO,OAAO,KAAK,MAAM,IAAI,CAAC,CAAC,SAAS;GACxC,mBAAmB,OAAO,IAAI,EAAE,IAAI,KAAK;GACzC,kBAAkB,MAAM,IAAI,EAAE,IAAI,KAAK;GACvC,aAAa,QAAQ,IAAI,EAAE,IAAI,KAAK;GACpC,eAAe,eACb,OAAO,iBAAiB,CAAC,GACzB,EAAE,MACF,OAAO,kBAAkB,EAAE,SAAS,CAAC,CACvC;GACA,oBAAoB,KAAK,MAAM,iBAAiB,KAAK,CAAC,EAAA,CAAG;GACzD,kBAAkB,WAAW,KAAK,KAAK,YAAY,CAAC;GACpD,aAAa,WAAW,KAAK,KAAK,OAAO,CAAC;GAC1C,UAAU,KAAK,SAAS,kBAAkB;GAC1C,mBAAmB,WAAW,KAAK,uBAAuB,IAAI,KAAK,KAAK,MAAM,GAAG,GAAG,CAAC;EACvF;CACF,CAAC;CACD,OAAO;EAAE,qBAAqB;EAAa;CAAQ;AACrD;AAIA,MAAM,aAAa;AAEnB,SAAS,QACP,MACA,SACA,OACA,UACA,YACA,YACA,aACA,aACA,WACgB;CAChB,OAAO;EACL,gBAAgB;EAChB,YAAY,iBAAiB;GAAE,YAAY;GAAY;GAAM;GAAS;EAAM,CAAC;EAC7E,YAAY;EACZ,aAAa;EACb;EACA;EACA;EACA;EACA,eAAe,CAAC;GAAE,MAAM;GAAY,KAAK;EAAY,CAAC;EACtD,oBAAoB;EACpB;EACA;CACF;AACF;;AAGA,SAAgB,uBACd,QACA,YACkB;CAClB,MAAM,MAAwB,CAAC;CAC/B,KAAK,MAAM,KAAK,OAAO,SAAS;EAC9B,MAAM,cAAc,EAAE,oBAAoB,EAAE;EAI5C,IAHkB,cAAc,EAAE,cAAc,EAAE,kBAGhC,GAChB,IAAI,KACF,QACE,eACA,EAAE,MACF,UAAU,EAAE,KAAK,+EACjB,QACA,IACA,YACA,sIACA,EAAE,MACF,iGACF,CACF;OACK,IAAI,gBAAgB,KAAK,EAAE,cAAc,EAAE,gBAAgB,GAEhE,IAAI,KACF,QACE,eACA,EAAE,MACF,UAAU,EAAE,KAAK,gFAAgF,EAAE,YAAY,cAAc,EAAE,cAAc,IAC7I,QACA,IACA,YACA,2JACA,EAAE,MACF,sEACF,CACF;EAIF,IAAI,eAAe,KAAK,CAAC,EAAE,mBACzB,IAAI,KACF,QACE,mBACA,EAAE,MACF,UAAU,EAAE,KAAK,mFACjB,UACA,IACA,YACA,oHACA,EAAE,IACJ,CACF;EAIF,IAAI,EAAE,SAAS,YAAY,EAAE,oBAAoB,GAC/C,IAAI,KACF,QACE,UACA,EAAE,MACF,iBAAiB,EAAE,KAAK,YAAY,EAAE,kBAAkB,+BACxD,QACA,KACA,YACA,uMACA,EAAE,IACJ,CACF;EAIF,IAAI,EAAE,QAAQ,wBAAwB,CAAC,EAAE,kBACvC,IAAI,KACF,QACE,mBACA,EAAE,MACF,UAAU,EAAE,KAAK,OAAO,EAAE,MAAM,4DAChC,UACA,IACA,YACA,mFAAmF,EAAE,MAAM,mDAC3F,EAAE,IACJ,CACF;EAIF,IAAI,CAAC,EAAE,aACL,IAAI,KACF,QACE,gBACA,EAAE,MACF,UAAU,EAAE,KAAK,oBACjB,OACA,IACA,YACA,wGACA,EAAE,IACJ,CACF;EAIF,IAAI,CAAC,EAAE,UACL,IAAI,KACF,QACE,iBACA,EAAE,MACF,UAAU,EAAE,KAAK,8CACjB,OACA,KACA,YACA,iJACA,EAAE,IACJ,CACF;CAEJ;CACA,OAAO;AACT;AAIA,IAAM,oBAAN,MAAyE;CACvE,KAAc;CACd,cACE;CACF,YAAqB;CACrB,OAAgB;EAAE,MAAM;EAA0B,iBAAiB;CAAE;CACrE,UAAmB;CACnB,kBAA2B;EACzB,MAAM;EACN,sBAAsB;EACtB,oBAAoB;CACtB;CAEA,MAAM,QAAQ,OAAyB,KAAgD;EACrF,MAAM,aAAa,IAAI,MAAM,+BAAc,IAAI,KAAK,EAAA,CAAE,YAAY;EAClE,IAAI,MACF,gBAAgB,MAAM,QAAQ,OAAO,eAAe,MAAM,oBAAoB,aAChF;EACA,OAAO,uBAAuB,OAAO,UAAU;CACjD;AACF;AAEA,MAAa,sBAAsB,IAAI,kBAAkB"}
|
|
1
|
+
{"version":3,"file":"index.js","names":[],"sources":["../../src/analyst/adapters.ts","../../src/analyst/bind.ts","../../src/analyst/define.ts","../../src/analyst/kinds/skill-usage.ts"],"sourcesContent":["/**\n * Adapter factory — lifts the semantic-concept judge into the Analyst\n * contract without re-implementing it.\n *\n * It builds an Analyst with a stable id (caller chooses; a default is given),\n * a version derived from the judge's version plus an adapter revision, and an\n * `analyze()` that calls the judge and lifts its output to `AnalystFinding[]`\n * through `makeFinding()`. The judge's `Severity` ('critical' | 'major' |\n * 'minor' | 'info') projects onto `AnalystSeverity`: 'major' becomes 'high'\n * and 'minor' becomes 'medium'.\n *\n * The adapter owns no state. Calling the factory twice is safe.\n */\n\nimport { CostLedger } from '../cost-ledger'\nimport type { Severity as LayerSeverity } from '../multi-layer-verifier'\nimport {\n runSemanticConceptJudge,\n SEMANTIC_CONCEPT_JUDGE_VERSION,\n type SemanticConceptJudgeInput,\n type SemanticConceptJudgeOptions,\n type SemanticConceptJudgeResult,\n} from '../semantic-concept-judge'\nimport type { Analyst, AnalystFinding, AnalystSeverity } from './types'\nimport { makeFinding } from './types'\nimport { settleUsageReceiptFromCostLedger, validateUsageSettlementTimeout } from './usage-receipt'\n\nconst ADAPTER_REV = '1'\n\n// ── Severity bridges ───────────────────────────────────────────────\n\nfunction liftSeverity(s: LayerSeverity): AnalystSeverity {\n switch (s) {\n case 'critical':\n return 'critical'\n case 'major':\n return 'high'\n case 'minor':\n return 'medium'\n case 'info':\n return 'info'\n }\n}\n\n// ── SemanticConceptJudge → Analyst ─────────────────────────────────\n\nexport interface SemanticConceptJudgeAdapterOpts {\n id?: string\n area?: string\n /** Registry context owns cancellation and the per-analyst cost ledger. */\n options: Omit<SemanticConceptJudgeOptions, 'costLedger' | 'signal'>\n /** Maximum post-cancellation wait for a provider receipt. Default 5 seconds. */\n settlementTimeoutMs?: number\n}\n\nexport function createSemanticConceptJudgeAdapter(\n opts: SemanticConceptJudgeAdapterOpts,\n): Analyst<SemanticConceptJudgeInput> {\n const id = opts.id ?? 'semantic-concept-judge'\n const area = opts.area ?? 'concept-coverage'\n const settlementTimeoutMs = validateUsageSettlementTimeout(opts.settlementTimeoutMs)\n return {\n id,\n description:\n 'Runs the semantic-concept judge and surfaces missing / weak concepts as findings.',\n inputKind: 'custom',\n cost: {\n kind: 'llm',\n models: opts.options.model ? [opts.options.model] : undefined,\n settlement_timeout_ms: settlementTimeoutMs,\n },\n version: `${SEMANTIC_CONCEPT_JUDGE_VERSION}-adapter-${ADAPTER_REV}`,\n async analyze(input, ctx) {\n const costLedger = new CostLedger(ctx.budgetUsd)\n let result: SemanticConceptJudgeResult\n try {\n result = await runSemanticConceptJudge(input, {\n ...opts.options,\n costLedger,\n signal: ctx.signal,\n })\n } finally {\n const usage = await settleUsageReceiptFromCostLedger(costLedger, {\n channel: 'judge',\n timeoutMs: settlementTimeoutMs,\n })\n if (!usage.settled) {\n ctx.log?.('semantic-concept judge provider settlement timed out', {\n pending_calls: usage.pendingCalls,\n timeout_ms: settlementTimeoutMs,\n })\n }\n ctx.recordUsage?.(usage.receipt)\n }\n if (!result.available) {\n return [\n makeFinding({\n analyst_id: id,\n area,\n claim: 'semantic-concept judge unavailable',\n rationale: result.error,\n severity: 'info',\n confidence: 1,\n evidence_refs: [],\n metadata: { reason: result.error },\n }),\n ]\n }\n const out: AnalystFinding[] = []\n for (const f of result.findings) {\n // Only surface gaps: missing concepts or low scores. Concepts at\n // 7+/10 with present=true are not findings — they're successes.\n if (f.present && f.score >= 7) continue\n out.push(\n makeFinding({\n analyst_id: id,\n area,\n subject: f.concept,\n claim: f.present\n ? `concept \"${f.concept}\" is weak (${f.score}/10)`\n : `concept \"${f.concept}\" is missing`,\n rationale: f.evidence,\n severity: liftSeverity(f.severity),\n confidence: 0.85,\n evidence_refs: [{ kind: 'artifact', uri: 'inline:evidence', excerpt: f.evidence }],\n metadata: {\n concept: f.concept,\n present: f.present,\n score_10: f.score,\n },\n }),\n )\n }\n return out\n },\n }\n}\n","/**\n * bindAnalyst — compile an `AnalystDefinition` into a runnable arm.\n *\n * The definition declares protocol (question, task text, reply grammar,\n * evidence projection, budget, repair turns); the transport binding supplies\n * execution (a bridge endpoint, or the caller-owned model path). Each\n * projection mode dispatches to the same strategy its arm's entry point runs,\n * so a compiled definition and the historical creator send byte-identical\n * requests — the definition parity suite asserts this per arm.\n *\n * A projection × transport pair with no strategy fails loud with\n * `AnalystExpressivenessError` naming the pair: an arm shape this layer cannot\n * express must be reported, never approximated.\n */\n\nimport type { CustomTokenPricing } from '../cost-ledger'\nimport type { AnalystBenchmarkRunner } from './benchmark'\nimport { runChunkedAnalystDefinition } from './benchmark-public-model'\nimport { runReplVariableAnalystDefinition } from './benchmark-public-rlm'\nimport type { PublicAnalystBenchmarkModelConfig } from './benchmark-public-types'\nimport { runInlineAnalystDefinition } from './benchmark-runner-prime'\nimport { type AnalystDefinition, AnalystExpressivenessError } from './definition'\nimport type { PrimeBridgeTransport } from './prime-bridge-transport'\nimport type { AnalystRunInputs } from './types'\n\n/** Execution half of a binding: who actually reaches the model. */\nexport type AnalystTransportBinding =\n | {\n /** An OpenAI-compatible cli-bridge endpoint (inline projections). */\n readonly kind: 'prime-bridge'\n readonly baseUrl: string\n readonly model: string\n readonly transport?: PrimeBridgeTransport\n readonly pricing?: CustomTokenPricing\n }\n | {\n /** The caller-owned model path (chunked and repl-variable projections). */\n readonly kind: 'model-owner'\n readonly config: PublicAnalystBenchmarkModelConfig\n }\n\n/**\n * Compile a definition plus a transport binding into a runnable arm. Dispatch\n * is total over the expressible pairs; everything else is a loud refusal.\n */\nexport function bindAnalyst<TRow, TAssignment, TBlock>(\n definition: AnalystDefinition<TRow, TAssignment, TBlock>,\n transports: AnalystTransportBinding,\n): AnalystBenchmarkRunner<AnalystRunInputs> {\n const mode = definition.projection.mode\n if (mode === 'inline' && transports.kind === 'prime-bridge') {\n return runInlineAnalystDefinition(definition, {\n baseUrl: transports.baseUrl,\n model: transports.model,\n ...(transports.transport ? { transport: transports.transport } : {}),\n ...(transports.pricing ? { pricing: transports.pricing } : {}),\n })\n }\n if (mode === 'chunked' && transports.kind === 'model-owner') {\n return runChunkedAnalystDefinition(definition, transports.config)\n }\n if (mode === 'repl-variable' && transports.kind === 'model-owner') {\n return runReplVariableAnalystDefinition(\n // The repl-variable strategy is written against the engine's raw-finding\n // row type; the definition's own row declaration carries it.\n definition as Parameters<typeof runReplVariableAnalystDefinition>[0],\n transports.config,\n )\n }\n throw new AnalystExpressivenessError(\n `no strategy compiles a '${mode}' projection over a '${transports.kind}' transport; ` +\n `definition '${definition.id}' cannot be expressed by this layer yet`,\n )\n}\n","import type { TraceAnalysisStore } from '../trace-analyst/store'\nimport type { ExactCapableAnalyst } from './exact-types'\nimport type { TraceAnalystDefinition } from './kind-factory'\nimport type { Analyst, AnalystCost } from './types'\n\n/**\n * Define a reusable trace-research question.\n *\n * The returned value contains no model, credentials, or execution state. Bind\n * it to any TraceAnalysisEngine with `runTraceAnalyst` or\n * `createTraceAnalyst`.\n */\nexport function defineTraceAnalyst(definition: TraceAnalystDefinition): TraceAnalystDefinition {\n for (const [name, value] of [\n ['id', definition.id],\n ['description', definition.description],\n ['area', definition.area],\n ['version', definition.version],\n ['instructions', definition.instructions],\n ] as const) {\n if (typeof value !== 'string' || !value.trim()) {\n throw new TypeError(`defineTraceAnalyst: ${name} must be a non-empty string`)\n }\n }\n return {\n ...definition,\n limits: definition.limits ? { ...definition.limits } : undefined,\n }\n}\n\nexport interface DefineCustomAnalystOptions {\n id: string\n description: string\n version?: string\n cost: AnalystCost\n analyze: Analyst<TraceAnalysisStore>['analyze']\n}\n\nexport interface DefineExactCustomAnalystOptions extends DefineCustomAnalystOptions {\n /** Canonical JSON for behavior knobs not already bound by `version`. */\n executionConfig: Readonly<Record<string, unknown>>\n}\n\n/** Construct a registrable analyst from a hand-written analyze function. */\nexport function defineCustomAnalyst(\n options: DefineExactCustomAnalystOptions,\n): ExactCapableAnalyst<TraceAnalysisStore>\nexport function defineCustomAnalyst(\n options: DefineCustomAnalystOptions,\n): Analyst<TraceAnalysisStore>\nexport function defineCustomAnalyst(\n options: DefineCustomAnalystOptions | DefineExactCustomAnalystOptions,\n): Analyst<TraceAnalysisStore> | ExactCapableAnalyst<TraceAnalysisStore> {\n if (!options.id.trim()) throw new TypeError('defineCustomAnalyst: id must not be empty')\n if (!options.description.trim()) {\n throw new TypeError('defineCustomAnalyst: description must not be empty')\n }\n if (options.cost === undefined) {\n throw new TypeError('defineCustomAnalyst: cost must be declared')\n }\n if (\n 'executionConfig' in options &&\n (!options.executionConfig ||\n typeof options.executionConfig !== 'object' ||\n Array.isArray(options.executionConfig))\n ) {\n throw new TypeError('defineCustomAnalyst: executionConfig must be an object')\n }\n return {\n id: options.id,\n description: options.description,\n version: options.version ?? '1.0.0',\n inputKind: 'trace-store',\n cost: options.cost,\n analyze: options.analyze,\n ...('executionConfig' in options ? { executionConfig: options.executionConfig } : {}),\n }\n}\n","/**\n * Skill-usage analyst — a DETERMINISTIC `Analyst` over a Claude/Codex skill\n * library + its trace corpus. Unlike the trace-store kinds (failure-mode,\n * improvement, ...) this kind calls no LLM: it mines real usage and skill\n * structure and emits findings by rule.\n *\n * It exists because the naive \"Skill-tool invocation count\" lies low — it\n * misses orchestrated sub-dispatch (a leaf skill run BY /pursue or /governor\n * logs under the parent), slash-command entry, local-script bypass, and\n * on-disk artifacts. The 2026-05-30 skill audit found 39/53 skills at zero\n * direct invocations, yet only one was a genuine cut: the rest were\n * measurement-invisible or discovery-limited. This analyst encodes that\n * lesson as a multi-signal usage model so a cheap repeatable pass can keep\n * the library honest, and so the expensive audit workflow's verdicts can\n * GEPA-distill it toward agreement (see `gold/skill-verdicts.gold.jsonl`).\n *\n * Report-building (`buildSkillUsageReport`, an fs scan) is separated from\n * finding emission (`SkillUsageAnalyst.analyze`, pure) so the slow scan runs\n * once at the registry boundary and the rule logic stays unit-testable.\n */\n\nimport { type Dirent, existsSync, readdirSync, readFileSync, statSync } from 'node:fs'\nimport { join } from 'node:path'\nimport type { ExactCapableAnalyst } from '../exact-types'\nimport type { AnalystContext, AnalystFinding, AnalystSeverity } from '../types'\nimport { computeFindingId } from '../types'\n\n// ── Input model ──────────────────────────────────────────────────────\n\nexport type SkillKind = 'public' | 'private'\n\n/** One skill's multi-signal usage + structure. All counts are deterministic. */\nexport interface SkillUsageRecord {\n name: string\n kind: SkillKind\n /** Absolute path to the skill's SKILL.md. */\n path: string\n lines: number\n /** `\"skill\":\"<name>\"` Skill-tool invocations across the trace corpus. */\n directInvocations: number\n /** `<command-name>/<name>` slash invocations across the trace corpus. */\n slashInvocations: number\n /** Sibling skills whose SKILL.md dispatches to this one (`/<name>`). Proxy\n * for orchestrated sub-dispatch the per-skill counter cannot see. */\n inboundRefs: number\n /** On-disk artifacts attributable to the skill (e.g. `.evolve/<name>/**`). */\n artifactCount: number\n /** Tangle-private reference count in the body (leak signal for public skills). */\n tanglePrivateRefs: number\n hasReferencesDir: boolean\n hasEvalsDir: boolean\n /** Body mentions `skill-runs.jsonl` (visible to /reflect + /governor). */\n logsRuns: boolean\n /** Description carries an explicit `Triggers:` clause / trigger phrases. */\n hasTriggerPhrases: boolean\n}\n\nexport interface SkillUsageReport {\n generatedFromTraces: number\n records: SkillUsageRecord[]\n}\n\nexport interface SkillUsageScanConfig {\n /** Dirs holding `*.jsonl` transcripts (Claude `~/.claude/projects`, Codex sessions). */\n transcriptDirs: string[]\n /** Skill roots to scan; each dir directly under `root` with a `SKILL.md` is a skill. */\n skillRoots: { root: string; kind: SkillKind }[]\n /** Roots scanned for `<root>/.evolve/<skill>` artifact dirs. */\n artifactRoots?: string[]\n /** Token-prefixed mappings: skill name → extra artifact subpaths under an artifactRoot\n * (e.g. reflect → `.evolve/reflections`). Catches non-eponymous artifact dirs. */\n artifactAliases?: Record<string, string[]>\n /** Cap files read per transcript dir (bounds a huge corpus); 0 = unbounded. */\n maxTranscriptsPerDir?: number\n}\n\n// ── Deterministic thresholds ─────────────────────────────────────────\n\n/** Anthropic's authoring guidance keeps SKILL.md short; past this with no\n * `references/` split the body burns context budget every session. */\nconst BLOAT_LINE_THRESHOLD = 300\n\nconst TANGLE_PRIVATE_RE =\n /\\b(cli-bridge|tangletools|ops-board|drew-gtr-pro|@tangle-network\\/|~\\/company|tangle\\.tools|gtm-agent)\\b|\\bkimi\\b|\\btcloud\\b/gi\nconst TRIGGER_RE = /triggers?\\s*[:-]/i\n\n// ── Report builder (fs scan — slow, runs once at the registry boundary) ──\n\nfunction listSkillDirs(root: string): { name: string; path: string }[] {\n if (!existsSync(root)) return []\n const out: { name: string; path: string }[] = []\n for (const entry of readdirSync(root, { withFileTypes: true })) {\n if (!entry.isDirectory() && !entry.isSymbolicLink()) continue\n const skillMd = join(root, entry.name, 'SKILL.md')\n if (existsSync(skillMd)) out.push({ name: entry.name, path: skillMd })\n }\n return out\n}\n\nfunction walkJsonl(dir: string, cap: number): string[] {\n if (!existsSync(dir)) return []\n const files: string[] = []\n const stack = [dir]\n while (stack.length) {\n const cur = stack.pop()!\n let entries: Dirent[]\n try {\n entries = readdirSync(cur, { withFileTypes: true })\n } catch {\n continue\n }\n for (const e of entries) {\n const full = join(cur, e.name)\n if (e.isDirectory()) stack.push(full)\n else if (e.name.endsWith('.jsonl')) {\n files.push(full)\n if (cap > 0 && files.length >= cap) return files\n }\n }\n }\n return files\n}\n\nfunction frontmatterDescription(body: string): string {\n const fm = /^---\\n([\\s\\S]*?)\\n---/.exec(body)\n const block = fm?.[1] ?? ''\n const m = /description:\\s*(.+)/i.exec(block)\n return m?.[1] ?? ''\n}\n\nfunction countArtifacts(roots: string[], name: string, aliases: string[]): number {\n let n = 0\n for (const root of roots) {\n const candidates = [join(root, '.evolve', name), ...aliases.map((a) => join(root, a))]\n for (const dir of candidates) {\n if (!existsSync(dir)) continue\n try {\n if (statSync(dir).isDirectory()) n += readdirSync(dir).length\n else n += 1\n } catch {\n /* unreadable — skip */\n }\n }\n }\n return n\n}\n\n/** Scan the corpus + skill roots into a {@link SkillUsageReport}. Deterministic. */\nexport function buildSkillUsageReport(config: SkillUsageScanConfig): SkillUsageReport {\n const skills = config.skillRoots.flatMap(({ root, kind }) =>\n listSkillDirs(root).map((s) => ({ ...s, kind })),\n )\n const names = skills.map((s) => s.name)\n\n // One pass over the corpus accumulating direct + slash counts per skill.\n const direct = new Map<string, number>(names.map((n) => [n, 0]))\n const slash = new Map<string, number>(names.map((n) => [n, 0]))\n const skillRe = /\"skill\"\\s*:\\s*\"([a-z0-9_:-]+)\"/g\n const cmdRe = /<command-name>\\/?([a-z0-9_:-]+)<\\/command-name>/g\n let transcripts = 0\n for (const dir of config.transcriptDirs) {\n for (const file of walkJsonl(dir, config.maxTranscriptsPerDir ?? 0)) {\n transcripts += 1\n let data: string\n try {\n data = readFileSync(file, 'utf8')\n } catch {\n continue\n }\n for (const m of data.matchAll(skillRe)) {\n const g = m[1]\n if (!g) continue\n const n = g.split(':').pop() ?? g\n const prev = direct.get(n)\n if (prev !== undefined) direct.set(n, prev + 1)\n }\n for (const m of data.matchAll(cmdRe)) {\n const g = m[1]\n if (g === undefined) continue\n const prev = slash.get(g)\n if (prev !== undefined) slash.set(g, prev + 1)\n }\n }\n }\n\n // Read each skill body once; compute structure + inbound refs across siblings.\n const bodies = new Map<string, string>()\n for (const s of skills) {\n try {\n bodies.set(s.name, readFileSync(s.path, 'utf8'))\n } catch {\n bodies.set(s.name, '')\n }\n }\n const inbound = new Map<string, number>(names.map((n) => [n, 0]))\n for (const target of names) {\n const ref = new RegExp(`/${target}\\\\b|\\\\[\\\\[${target}\\\\]\\\\]`)\n for (const s of skills) {\n if (s.name === target) continue\n if (ref.test(bodies.get(s.name) ?? '')) inbound.set(target, inbound.get(target)! + 1)\n }\n }\n\n const records: SkillUsageRecord[] = skills.map((s) => {\n const body = bodies.get(s.name) ?? ''\n const dir = s.path.replace(/\\/SKILL\\.md$/, '')\n return {\n name: s.name,\n kind: s.kind,\n path: s.path,\n lines: body ? body.split('\\n').length : 0,\n directInvocations: direct.get(s.name) ?? 0,\n slashInvocations: slash.get(s.name) ?? 0,\n inboundRefs: inbound.get(s.name) ?? 0,\n artifactCount: countArtifacts(\n config.artifactRoots ?? [],\n s.name,\n config.artifactAliases?.[s.name] ?? [],\n ),\n tanglePrivateRefs: (body.match(TANGLE_PRIVATE_RE) ?? []).length,\n hasReferencesDir: existsSync(join(dir, 'references')),\n hasEvalsDir: existsSync(join(dir, 'evals')),\n logsRuns: body.includes('skill-runs.jsonl'),\n hasTriggerPhrases: TRIGGER_RE.test(frontmatterDescription(body) || body.slice(0, 600)),\n }\n })\n return { generatedFromTraces: transcripts, records }\n}\n\n// ── Finding emission (pure — unit-testable, no LLM, no fs) ────────────\n\nconst ANALYST_ID = 'skill-usage'\n\nfunction finding(\n area: string,\n subject: string,\n claim: string,\n severity: AnalystSeverity,\n confidence: number,\n producedAt: string,\n recommended: string,\n evidenceUri: string,\n rationale?: string,\n): AnalystFinding {\n return {\n schema_version: '1.0.0',\n finding_id: computeFindingId({ analyst_id: ANALYST_ID, area, subject, claim }),\n analyst_id: ANALYST_ID,\n produced_at: producedAt,\n severity,\n area,\n claim,\n rationale,\n evidence_refs: [{ kind: 'artifact', uri: evidenceUri }],\n recommended_action: recommended,\n confidence,\n subject,\n }\n}\n\n/** Pure rule pass over a report → findings. Exported for direct/unit use. */\nexport function emitSkillUsageFindings(\n report: SkillUsageReport,\n producedAt: string,\n): AnalystFinding[] {\n const out: AnalystFinding[] = []\n for (const r of report.records) {\n const directTotal = r.directInvocations + r.slashInvocations\n const trueUsage = directTotal + r.inboundRefs + r.artifactCount\n\n // 1. Dead: no usage signal of ANY kind. The only real deprecation candidate.\n if (trueUsage === 0) {\n out.push(\n finding(\n 'skill-usage',\n r.name,\n `Skill '${r.name}' has zero usage across all signals (direct, slash, inbound-refs, artifacts)`,\n 'high',\n 0.6,\n producedAt,\n 'Confirm the skill covers a real recurring job; if not, deprecate. Zero true usage is the only deterministic deprecation candidate.',\n r.path,\n 'No Skill-tool call, no slash invocation, no sibling dispatches to it, and no on-disk artifacts.',\n ),\n )\n } else if (directTotal === 0 && r.inboundRefs + r.artifactCount > 0) {\n // 2. Measurement-invisible: real use via orchestration/artifacts, never invoked directly.\n out.push(\n finding(\n 'skill-usage',\n r.name,\n `Skill '${r.name}' shows 0 direct invocations but is used via orchestration/artifacts (inbound=${r.inboundRefs}, artifacts=${r.artifactCount})`,\n 'info',\n 0.8,\n producedAt,\n 'Do NOT treat as unused — usage is real but logged under parent skills or on disk. Strengthen direct-invocation discovery only if direct use is desired.',\n r.path,\n 'The Skill-tool counter undercounts orchestrated/chained leaf skills.',\n ),\n )\n }\n\n // 3. Discovery gap: low direct use AND weak trigger surface.\n if (directTotal <= 2 && !r.hasTriggerPhrases) {\n out.push(\n finding(\n 'discoverability',\n r.name,\n `Skill '${r.name}' is rarely invoked directly and its description has no explicit trigger phrases`,\n 'medium',\n 0.7,\n producedAt,\n 'Add a `Triggers:` clause with verbatim user phrases to the frontmatter description so the model auto-invokes it.',\n r.path,\n ),\n )\n }\n\n // 4. Public-repo leak.\n if (r.kind === 'public' && r.tanglePrivateRefs > 0) {\n out.push(\n finding(\n 'safety',\n r.name,\n `Public skill '${r.name}' carries ${r.tanglePrivateRefs} Tangle-private reference(s)`,\n 'high',\n 0.75,\n producedAt,\n 'Sanitize incidental internal refs (cli-bridge/kimi/tcloud/~company/private repos) or relocate to a private repo. Verify @tangle-network/* refs are to PUBLISHED packages before treating as a leak.',\n r.path,\n ),\n )\n }\n\n // 5. Bloat / no progressive disclosure.\n if (r.lines > BLOAT_LINE_THRESHOLD && !r.hasReferencesDir) {\n out.push(\n finding(\n 'maintainability',\n r.name,\n `Skill '${r.name}' is ${r.lines} lines with no references/ split (progressive disclosure)`,\n 'medium',\n 0.8,\n producedAt,\n `Split detail into references/ loaded on demand; keep SKILL.md a short overview. ${r.lines} lines load into every session's context budget.`,\n r.path,\n ),\n )\n }\n\n // 6. No evals (Anthropic's \">=3 evals before docs\" rule).\n if (!r.hasEvalsDir) {\n out.push(\n finding(\n 'data-quality',\n r.name,\n `Skill '${r.name}' ships no evals/`,\n 'low',\n 0.6,\n producedAt,\n 'Add evals/evals.json with >=3 scenarios proving the skill beats baseline; gives regression coverage.',\n r.path,\n ),\n )\n }\n\n // 7. No run logging → invisible to /reflect and /governor.\n if (!r.logsRuns) {\n out.push(\n finding(\n 'observability',\n r.name,\n `Skill '${r.name}' never appends to .evolve/skill-runs.jsonl`,\n 'low',\n 0.55,\n producedAt,\n 'Append one run line to .evolve/skill-runs.jsonl on completion, or declare it a non-logging leaf, so the self-improvement loop can see it ran.',\n r.path,\n ),\n )\n }\n }\n return out\n}\n\n// ── The Analyst ──────────────────────────────────────────────────────\n\nclass SkillUsageAnalyst implements ExactCapableAnalyst<SkillUsageReport> {\n readonly id = ANALYST_ID\n readonly description =\n 'Deterministic multi-signal skill-usage analysis: flags dead skills, measurement-invisible (orchestrated) usage, discovery gaps, public-repo leaks, bloat, missing evals, and missing run-logging.'\n readonly inputKind = 'custom' as const\n readonly cost = { kind: 'deterministic' as const, est_usd_per_run: 0 }\n readonly version = '1.0.0'\n readonly executionConfig = {\n kind: 'skill-usage',\n bloat_line_threshold: BLOAT_LINE_THRESHOLD,\n produced_at_source: 'tags.producedAt-or-system-clock',\n } as const\n\n async analyze(input: SkillUsageReport, ctx: AnalystContext): Promise<AnalystFinding[]> {\n const producedAt = ctx.tags?.producedAt ?? new Date().toISOString()\n ctx.log?.(\n `skill-usage: ${input.records.length} skills over ${input.generatedFromTraces} transcripts`,\n )\n return emitSkillUsageFindings(input, producedAt)\n }\n}\n\nexport const SKILL_USAGE_ANALYST = new SkillUsageAnalyst()\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;AA2BA,MAAM,cAAc;AAIpB,SAAS,aAAa,GAAmC;CACvD,QAAQ,GAAR;EACE,KAAK,YACH,OAAO;EACT,KAAK,SACH,OAAO;EACT,KAAK,SACH,OAAO;EACT,KAAK,QACH,OAAO;CACX;AACF;AAaA,SAAgB,kCACd,MACoC;CACpC,MAAM,KAAK,KAAK,MAAM;CACtB,MAAM,OAAO,KAAK,QAAQ;CAC1B,MAAM,sBAAsB,+BAA+B,KAAK,mBAAmB;CACnF,OAAO;EACL;EACA,aACE;EACF,WAAW;EACX,MAAM;GACJ,MAAM;GACN,QAAQ,KAAK,QAAQ,QAAQ,CAAC,KAAK,QAAQ,KAAK,IAAI,KAAA;GACpD,uBAAuB;EACzB;EACA,SAAS,GAAG,+BAA+B,WAAW;EACtD,MAAM,QAAQ,OAAO,KAAK;GACxB,MAAM,aAAa,IAAI,WAAW,IAAI,SAAS;GAC/C,IAAI;GACJ,IAAI;IACF,SAAS,MAAM,wBAAwB,OAAO;KAC5C,GAAG,KAAK;KACR;KACA,QAAQ,IAAI;IACd,CAAC;GACH,UAAU;IACR,MAAM,QAAQ,MAAM,iCAAiC,YAAY;KAC/D,SAAS;KACT,WAAW;IACb,CAAC;IACD,IAAI,CAAC,MAAM,SACT,IAAI,MAAM,wDAAwD;KAChE,eAAe,MAAM;KACrB,YAAY;IACd,CAAC;IAEH,IAAI,cAAc,MAAM,OAAO;GACjC;GACA,IAAI,CAAC,OAAO,WACV,OAAO,CACL,YAAY;IACV,YAAY;IACZ;IACA,OAAO;IACP,WAAW,OAAO;IAClB,UAAU;IACV,YAAY;IACZ,eAAe,CAAC;IAChB,UAAU,EAAE,QAAQ,OAAO,MAAM;GACnC,CAAC,CACH;GAEF,MAAM,MAAwB,CAAC;GAC/B,KAAK,MAAM,KAAK,OAAO,UAAU;IAG/B,IAAI,EAAE,WAAW,EAAE,SAAS,GAAG;IAC/B,IAAI,KACF,YAAY;KACV,YAAY;KACZ;KACA,SAAS,EAAE;KACX,OAAO,EAAE,UACL,YAAY,EAAE,QAAQ,aAAa,EAAE,MAAM,QAC3C,YAAY,EAAE,QAAQ;KAC1B,WAAW,EAAE;KACb,UAAU,aAAa,EAAE,QAAQ;KACjC,YAAY;KACZ,eAAe,CAAC;MAAE,MAAM;MAAY,KAAK;MAAmB,SAAS,EAAE;KAAS,CAAC;KACjF,UAAU;MACR,SAAS,EAAE;MACX,SAAS,EAAE;MACX,UAAU,EAAE;KACd;IACF,CAAC,CACH;GACF;GACA,OAAO;EACT;CACF;AACF;;;;;;;AC3FA,SAAgB,YACd,YACA,YAC0C;CAC1C,MAAM,OAAO,WAAW,WAAW;CACnC,IAAI,SAAS,YAAY,WAAW,SAAS,gBAC3C,OAAO,2BAA2B,YAAY;EAC5C,SAAS,WAAW;EACpB,OAAO,WAAW;EAClB,GAAI,WAAW,YAAY,EAAE,WAAW,WAAW,UAAU,IAAI,CAAC;EAClE,GAAI,WAAW,UAAU,EAAE,SAAS,WAAW,QAAQ,IAAI,CAAC;CAC9D,CAAC;CAEH,IAAI,SAAS,aAAa,WAAW,SAAS,eAC5C,OAAO,4BAA4B,YAAY,WAAW,MAAM;CAElE,IAAI,SAAS,mBAAmB,WAAW,SAAS,eAClD,OAAO,iCAGL,YACA,WAAW,MACb;CAEF,MAAM,IAAI,2BACR,2BAA2B,KAAK,uBAAuB,WAAW,KAAK,2BACtD,WAAW,GAAG,wCACjC;AACF;;;;;;;;;;AC7DA,SAAgB,mBAAmB,YAA4D;CAC7F,KAAK,MAAM,CAAC,MAAM,UAAU;EAC1B,CAAC,MAAM,WAAW,EAAE;EACpB,CAAC,eAAe,WAAW,WAAW;EACtC,CAAC,QAAQ,WAAW,IAAI;EACxB,CAAC,WAAW,WAAW,OAAO;EAC9B,CAAC,gBAAgB,WAAW,YAAY;CAC1C,GACE,IAAI,OAAO,UAAU,YAAY,CAAC,MAAM,KAAK,GAC3C,MAAM,IAAI,UAAU,uBAAuB,KAAK,4BAA4B;CAGhF,OAAO;EACL,GAAG;EACH,QAAQ,WAAW,SAAS,EAAE,GAAG,WAAW,OAAO,IAAI,KAAA;CACzD;AACF;AAsBA,SAAgB,oBACd,SACuE;CACvE,IAAI,CAAC,QAAQ,GAAG,KAAK,GAAG,MAAM,IAAI,UAAU,2CAA2C;CACvF,IAAI,CAAC,QAAQ,YAAY,KAAK,GAC5B,MAAM,IAAI,UAAU,oDAAoD;CAE1E,IAAI,QAAQ,SAAS,KAAA,GACnB,MAAM,IAAI,UAAU,4CAA4C;CAElE,IACE,qBAAqB,YACpB,CAAC,QAAQ,mBACR,OAAO,QAAQ,oBAAoB,YACnC,MAAM,QAAQ,QAAQ,eAAe,IAEvC,MAAM,IAAI,UAAU,wDAAwD;CAE9E,OAAO;EACL,IAAI,QAAQ;EACZ,aAAa,QAAQ;EACrB,SAAS,QAAQ,WAAW;EAC5B,WAAW;EACX,MAAM,QAAQ;EACd,SAAS,QAAQ;EACjB,GAAI,qBAAqB,UAAU,EAAE,iBAAiB,QAAQ,gBAAgB,IAAI,CAAC;CACrF;AACF;;;;;;;;;;;;;;;;;;;;;;;;;ACGA,MAAM,uBAAuB;AAE7B,MAAM,oBACJ;AACF,MAAM,aAAa;AAInB,SAAS,cAAc,MAAgD;CACrE,IAAI,CAAC,WAAW,IAAI,GAAG,OAAO,CAAC;CAC/B,MAAM,MAAwC,CAAC;CAC/C,KAAK,MAAM,SAAS,YAAY,MAAM,EAAE,eAAe,KAAK,CAAC,GAAG;EAC9D,IAAI,CAAC,MAAM,YAAY,KAAK,CAAC,MAAM,eAAe,GAAG;EACrD,MAAM,UAAU,KAAK,MAAM,MAAM,MAAM,UAAU;EACjD,IAAI,WAAW,OAAO,GAAG,IAAI,KAAK;GAAE,MAAM,MAAM;GAAM,MAAM;EAAQ,CAAC;CACvE;CACA,OAAO;AACT;AAEA,SAAS,UAAU,KAAa,KAAuB;CACrD,IAAI,CAAC,WAAW,GAAG,GAAG,OAAO,CAAC;CAC9B,MAAM,QAAkB,CAAC;CACzB,MAAM,QAAQ,CAAC,GAAG;CAClB,OAAO,MAAM,QAAQ;EACnB,MAAM,MAAM,MAAM,IAAI;EACtB,IAAI;EACJ,IAAI;GACF,UAAU,YAAY,KAAK,EAAE,eAAe,KAAK,CAAC;EACpD,QAAQ;GACN;EACF;EACA,KAAK,MAAM,KAAK,SAAS;GACvB,MAAM,OAAO,KAAK,KAAK,EAAE,IAAI;GAC7B,IAAI,EAAE,YAAY,GAAG,MAAM,KAAK,IAAI;QAC/B,IAAI,EAAE,KAAK,SAAS,QAAQ,GAAG;IAClC,MAAM,KAAK,IAAI;IACf,IAAI,MAAM,KAAK,MAAM,UAAU,KAAK,OAAO;GAC7C;EACF;CACF;CACA,OAAO;AACT;AAEA,SAAS,uBAAuB,MAAsB;CAEpD,MAAM,QADK,wBAAwB,KAAK,IACzB,CAAC,GAAG,MAAM;CAEzB,OADU,uBAAuB,KAAK,KAC/B,CAAC,GAAG,MAAM;AACnB;AAEA,SAAS,eAAe,OAAiB,MAAc,SAA2B;CAChF,IAAI,IAAI;CACR,KAAK,MAAM,QAAQ,OAAO;EACxB,MAAM,aAAa,CAAC,KAAK,MAAM,WAAW,IAAI,GAAG,GAAG,QAAQ,KAAK,MAAM,KAAK,MAAM,CAAC,CAAC,CAAC;EACrF,KAAK,MAAM,OAAO,YAAY;GAC5B,IAAI,CAAC,WAAW,GAAG,GAAG;GACtB,IAAI;IACF,IAAI,SAAS,GAAG,CAAC,CAAC,YAAY,GAAG,KAAK,YAAY,GAAG,CAAC,CAAC;SAClD,KAAK;GACZ,QAAQ,CAER;EACF;CACF;CACA,OAAO;AACT;;AAGA,SAAgB,sBAAsB,QAAgD;CACpF,MAAM,SAAS,OAAO,WAAW,SAAS,EAAE,MAAM,WAChD,cAAc,IAAI,CAAC,CAAC,KAAK,OAAO;EAAE,GAAG;EAAG;CAAK,EAAE,CACjD;CACA,MAAM,QAAQ,OAAO,KAAK,MAAM,EAAE,IAAI;CAGtC,MAAM,SAAS,IAAI,IAAoB,MAAM,KAAK,MAAM,CAAC,GAAG,CAAC,CAAC,CAAC;CAC/D,MAAM,QAAQ,IAAI,IAAoB,MAAM,KAAK,MAAM,CAAC,GAAG,CAAC,CAAC,CAAC;CAC9D,MAAM,UAAU;CAChB,MAAM,QAAQ;CACd,IAAI,cAAc;CAClB,KAAK,MAAM,OAAO,OAAO,gBACvB,KAAK,MAAM,QAAQ,UAAU,KAAK,OAAO,wBAAwB,CAAC,GAAG;EACnE,eAAe;EACf,IAAI;EACJ,IAAI;GACF,OAAO,aAAa,MAAM,MAAM;EAClC,QAAQ;GACN;EACF;EACA,KAAK,MAAM,KAAK,KAAK,SAAS,OAAO,GAAG;GACtC,MAAM,IAAI,EAAE;GACZ,IAAI,CAAC,GAAG;GACR,MAAM,IAAI,EAAE,MAAM,GAAG,CAAC,CAAC,IAAI,KAAK;GAChC,MAAM,OAAO,OAAO,IAAI,CAAC;GACzB,IAAI,SAAS,KAAA,GAAW,OAAO,IAAI,GAAG,OAAO,CAAC;EAChD;EACA,KAAK,MAAM,KAAK,KAAK,SAAS,KAAK,GAAG;GACpC,MAAM,IAAI,EAAE;GACZ,IAAI,MAAM,KAAA,GAAW;GACrB,MAAM,OAAO,MAAM,IAAI,CAAC;GACxB,IAAI,SAAS,KAAA,GAAW,MAAM,IAAI,GAAG,OAAO,CAAC;EAC/C;CACF;CAIF,MAAM,yBAAS,IAAI,IAAoB;CACvC,KAAK,MAAM,KAAK,QACd,IAAI;EACF,OAAO,IAAI,EAAE,MAAM,aAAa,EAAE,MAAM,MAAM,CAAC;CACjD,QAAQ;EACN,OAAO,IAAI,EAAE,MAAM,EAAE;CACvB;CAEF,MAAM,UAAU,IAAI,IAAoB,MAAM,KAAK,MAAM,CAAC,GAAG,CAAC,CAAC,CAAC;CAChE,KAAK,MAAM,UAAU,OAAO;EAC1B,MAAM,MAAM,IAAI,OAAO,IAAI,OAAO,YAAY,OAAO,OAAO;EAC5D,KAAK,MAAM,KAAK,QAAQ;GACtB,IAAI,EAAE,SAAS,QAAQ;GACvB,IAAI,IAAI,KAAK,OAAO,IAAI,EAAE,IAAI,KAAK,EAAE,GAAG,QAAQ,IAAI,QAAQ,QAAQ,IAAI,MAAM,IAAK,CAAC;EACtF;CACF;CAEA,MAAM,UAA8B,OAAO,KAAK,MAAM;EACpD,MAAM,OAAO,OAAO,IAAI,EAAE,IAAI,KAAK;EACnC,MAAM,MAAM,EAAE,KAAK,QAAQ,gBAAgB,EAAE;EAC7C,OAAO;GACL,MAAM,EAAE;GACR,MAAM,EAAE;GACR,MAAM,EAAE;GACR,OAAO,OAAO,KAAK,MAAM,IAAI,CAAC,CAAC,SAAS;GACxC,mBAAmB,OAAO,IAAI,EAAE,IAAI,KAAK;GACzC,kBAAkB,MAAM,IAAI,EAAE,IAAI,KAAK;GACvC,aAAa,QAAQ,IAAI,EAAE,IAAI,KAAK;GACpC,eAAe,eACb,OAAO,iBAAiB,CAAC,GACzB,EAAE,MACF,OAAO,kBAAkB,EAAE,SAAS,CAAC,CACvC;GACA,oBAAoB,KAAK,MAAM,iBAAiB,KAAK,CAAC,EAAA,CAAG;GACzD,kBAAkB,WAAW,KAAK,KAAK,YAAY,CAAC;GACpD,aAAa,WAAW,KAAK,KAAK,OAAO,CAAC;GAC1C,UAAU,KAAK,SAAS,kBAAkB;GAC1C,mBAAmB,WAAW,KAAK,uBAAuB,IAAI,KAAK,KAAK,MAAM,GAAG,GAAG,CAAC;EACvF;CACF,CAAC;CACD,OAAO;EAAE,qBAAqB;EAAa;CAAQ;AACrD;AAIA,MAAM,aAAa;AAEnB,SAAS,QACP,MACA,SACA,OACA,UACA,YACA,YACA,aACA,aACA,WACgB;CAChB,OAAO;EACL,gBAAgB;EAChB,YAAY,iBAAiB;GAAE,YAAY;GAAY;GAAM;GAAS;EAAM,CAAC;EAC7E,YAAY;EACZ,aAAa;EACb;EACA;EACA;EACA;EACA,eAAe,CAAC;GAAE,MAAM;GAAY,KAAK;EAAY,CAAC;EACtD,oBAAoB;EACpB;EACA;CACF;AACF;;AAGA,SAAgB,uBACd,QACA,YACkB;CAClB,MAAM,MAAwB,CAAC;CAC/B,KAAK,MAAM,KAAK,OAAO,SAAS;EAC9B,MAAM,cAAc,EAAE,oBAAoB,EAAE;EAI5C,IAHkB,cAAc,EAAE,cAAc,EAAE,kBAGhC,GAChB,IAAI,KACF,QACE,eACA,EAAE,MACF,UAAU,EAAE,KAAK,+EACjB,QACA,IACA,YACA,sIACA,EAAE,MACF,iGACF,CACF;OACK,IAAI,gBAAgB,KAAK,EAAE,cAAc,EAAE,gBAAgB,GAEhE,IAAI,KACF,QACE,eACA,EAAE,MACF,UAAU,EAAE,KAAK,gFAAgF,EAAE,YAAY,cAAc,EAAE,cAAc,IAC7I,QACA,IACA,YACA,2JACA,EAAE,MACF,sEACF,CACF;EAIF,IAAI,eAAe,KAAK,CAAC,EAAE,mBACzB,IAAI,KACF,QACE,mBACA,EAAE,MACF,UAAU,EAAE,KAAK,mFACjB,UACA,IACA,YACA,oHACA,EAAE,IACJ,CACF;EAIF,IAAI,EAAE,SAAS,YAAY,EAAE,oBAAoB,GAC/C,IAAI,KACF,QACE,UACA,EAAE,MACF,iBAAiB,EAAE,KAAK,YAAY,EAAE,kBAAkB,+BACxD,QACA,KACA,YACA,uMACA,EAAE,IACJ,CACF;EAIF,IAAI,EAAE,QAAQ,wBAAwB,CAAC,EAAE,kBACvC,IAAI,KACF,QACE,mBACA,EAAE,MACF,UAAU,EAAE,KAAK,OAAO,EAAE,MAAM,4DAChC,UACA,IACA,YACA,mFAAmF,EAAE,MAAM,mDAC3F,EAAE,IACJ,CACF;EAIF,IAAI,CAAC,EAAE,aACL,IAAI,KACF,QACE,gBACA,EAAE,MACF,UAAU,EAAE,KAAK,oBACjB,OACA,IACA,YACA,wGACA,EAAE,IACJ,CACF;EAIF,IAAI,CAAC,EAAE,UACL,IAAI,KACF,QACE,iBACA,EAAE,MACF,UAAU,EAAE,KAAK,8CACjB,OACA,KACA,YACA,iJACA,EAAE,IACJ,CACF;CAEJ;CACA,OAAO;AACT;AAIA,IAAM,oBAAN,MAAyE;CACvE,KAAc;CACd,cACE;CACF,YAAqB;CACrB,OAAgB;EAAE,MAAM;EAA0B,iBAAiB;CAAE;CACrE,UAAmB;CACnB,kBAA2B;EACzB,MAAM;EACN,sBAAsB;EACtB,oBAAoB;CACtB;CAEA,MAAM,QAAQ,OAAyB,KAAgD;EACrF,MAAM,aAAa,IAAI,MAAM,+BAAc,IAAI,KAAK,EAAA,CAAE,YAAY;EAClE,IAAI,MACF,gBAAgB,MAAM,QAAQ,OAAO,eAAe,MAAM,oBAAoB,aAChF;EACA,OAAO,uBAAuB,OAAO,UAAU;CACjD;AACF;AAEA,MAAa,sBAAsB,IAAI,kBAAkB"}
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { p as EvidenceRef } from "./types-
|
|
1
|
+
import { p as EvidenceRef } from "./types-DN2WdT5S.js";
|
|
2
2
|
//#region src/statistics/power-and-mde.d.ts
|
|
3
3
|
/**
|
|
4
4
|
* Required N per arm for a two-sample comparison at target effect size,
|
|
@@ -310,4 +310,4 @@ declare function attest(report: unknown, provenance: AttestationProvenance): Att
|
|
|
310
310
|
declare function verifyAttestation(report: unknown, attested: AttestedReport): AttestationVerification;
|
|
311
311
|
//#endregion
|
|
312
312
|
export { requiredSampleSize as C, requiredPairedSampleSize as S, inMemoryExperimentStore as _, attest as a, mcnemarRequiredN as b, ExperimentRep as c, ExperimentVerdict as d, ImprovementThresholds as f, improvementVerdict as g, fileExperimentStore as h, AttestedReport as i, ExperimentStats as l, computeExperimentStats as m, AttestationProvenance as n, verifyAttestation as o, ImprovementVerdictResult as p, AttestationVerification as r, Experiment as s, ATTESTATION_ALGORITHM as t, ExperimentTracker as u, mulberry32 as v, pairedMde as x, mcnemarPower as y };
|
|
313
|
-
//# sourceMappingURL=attestation-
|
|
313
|
+
//# sourceMappingURL=attestation-c1QvaBdX.d.ts.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"attestation-
|
|
1
|
+
{"version":3,"file":"attestation-c1QvaBdX.d.ts","names":[],"sources":["../src/statistics/power-and-mde.ts","../src/statistics/random.ts","../src/experiment-tracker.ts","../src/attestation.ts"],"mappings":";;;;;;;;iBAWgB,mBAAmB;EACjC;EACA;EACA;EACA;;;;;;;;;;;iBAsBc,yBAAyB;EACvC;EACA;EACA;EACA;;;;;;;;iBAkBc,UAAU;EACxB;EACA;EACA;EACA;;;;;;;;;;;;;;;;iBAyBc,iBAAiB;EAC/B;EACA;EACA;EACA;EACA;;;;;;;iBAyBc,aAAa;EAC3B;EACA;EACA;EACA;EACA;;;;;;;;iBCrHc,WAAW;;;;;KCyBf;;UAGK;;EAEf;;EAEA;;EAEA;;;;UAKe;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA,WAAW;;EAEX;;EAEA,UAAU;;UAGK;EACf;EACA;EACA;EACA;;EAEA;;EAEA;;;EAGA;;EAEA;;EAEA;;UAGe;;EAEf;;EAEA;;EAEA,YAAY;;EAEZ;;EAEA;EACA,MAAM;EACN,OAAO;EACP,SAAS;;EAET;;UAGe;;EAEf;;EAEA;;EAEA;;EAEA;;;EAGA;;UAGe;EACf,SAAS;;EAET;;EAEA;;;;;;;iBAkDc,uBACd,MAAM,iBACN,aAAa,wBACZ;;;;;;;iBAgDa,mBACd,WAAW,iBACX,QAAQ,wBACR,aAAa,wBACZ;;;KA6CS,yBAAyB,uBAAuB,QAAQ;;;UAInD;EACf,QAAQ,QAAQ;EAChB,KAAK,aAAa,eAAe;;;;iBAoBnB,wBAAwB,UAAS,eAAoB;;iBAarD,oBAAoB,eAAe;UAyBlC;EACf,QAAQ;EACR,mBAAmB;EACnB,aAAa;;EAEb;;UAGe;EACf;EACA;EACA;EACA;;EAEA,aAAa;;;;;;;;;cAUF;mBACM;mBACA;mBACA;mBACA;EAEjB,YAAY,UAAS;EAOf,OAAO,OAAO,wBAAwB,QAAQ;;;EA6B9C,OACJ,sBACA,KAAK,KAAK;IAAwC;IAAc;MAC/D,QAAQ;EAqCL,IAAI,uBAAuB,QAAQ;EAMnC,QAAQ,QAAQ;;EAKhB,WAAW,uBAAuB,QAAQ;;;;;;;;;;;;;;;;;;;;;;;;cCzarC;UAEI;;EAEf,eAAe;;EAEf;;;EAGA;;EAEA;;EAEA;;;EAGA;;UAGe;;EAEf;EACA,YAAY;EACZ,kBAAkB;;;;;;EAMlB;;UAGe;EACf;;EAEA;;EAEA;;;;;;;;iBAiBc,OAAO,iBAAiB,YAAY,wBAAwB;;;;;;;;;;;iBAqB5D,kBACd,iBACA,UAAU,iBACT"}
|
|
@@ -1,8 +1,7 @@
|
|
|
1
1
|
import { t as AgentEvalError } from "./errors-DEE6u6ot.js";
|
|
2
2
|
import { c as CostLedgerHandle } from "./cost-ledger-DbQdN3nO.js";
|
|
3
|
-
import { a as RunRecord } from "./run-record-
|
|
4
|
-
import {
|
|
5
|
-
import { p as ChatClient } from "./types-Bfk0uxRj.js";
|
|
3
|
+
import { a as RunRecord } from "./run-record-DTv1MdjK.js";
|
|
4
|
+
import { Y as RawProviderSink, p as ChatClient } from "./types-gvRsyJLh.js";
|
|
6
5
|
import { a as CheckerIdentity, n as VerdictCertification, t as DefaultVerdict, x as VerificationStrategySource } from "./verdict-E4eRNf7-.js";
|
|
7
6
|
//#region src/artifact-validator.d.ts
|
|
8
7
|
/**
|
|
@@ -278,4 +277,4 @@ declare function assertRealBackend(records: ReadonlyArray<RunRecord>, opts?: {
|
|
|
278
277
|
}): BackendIntegrityReport;
|
|
279
278
|
//#endregion
|
|
280
279
|
export { SatisfiedBy as _, ArtifactEventLike as a, createLlmCorrectnessChecker as b, ToolCallEventLike as c, CompletionVerdict as d, CorrectnessChecker as f, RequirementCheck as g, ProducedState as h, summarizeBackendIntegrity as i, extractProducedState as l, ProducedProposal as m, BackendIntegrityReport as n, ProposalEventLike as o, LlmCorrectnessCheckerOpts as p, assertRealBackend as r, RuntimeEventLike as s, BackendIntegrityError as t, CompletionRequirement as u, TaskGold as v, verifyCompletion as x, completionVerdict as y };
|
|
281
|
-
//# sourceMappingURL=backend-integrity-
|
|
280
|
+
//# sourceMappingURL=backend-integrity-CeuTgqsd.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"backend-integrity-CeuTgqsd.d.ts","names":[],"sources":["../src/artifact-validator.ts","../src/completion-verifier.ts","../src/produced-state.ts","../src/integrity/backend-integrity.ts"],"mappings":";;;;;;;;;;;;;;;;;;UAaiB;;EAEf;;EAEA;;EAEA;;EAEA,QAAQ;;EAER,WAAW;;;;;KCsBD;UAEK;;EAEf;;EAEA;;EAEA;;EAEA,cAAc;;UAGC;EACf;EACA,cAAc;;UAGC;EACf;EACA;EACA;;EAEA;;;UAIe;;EAEf,WAAW;;EAEX,WAAW;;EAEX;;UAGe;EACf;EACA;;EAEA;;;;;;EAMA;;EAEA;;;;;;;;;EASA;;EAEA;;EAEA;;;;;UAMe,0BAA0B;EACzC;EACA,cAAc;;EAEd;;EAEA;;EAEA;;;;;;;;;iBAUc,kBAAkB;EAChC;EACA,cAAc;;;EAGd,gBAAgB;IACd;;;;;;UAqCa;EACf,UAAU;EACV,SAAS;EACT;;;;;;;;;;;;UAae;GAEb,aAAa,uBACb,kBACC;IAAU;IAAkB;;EAC/B,cAAc;;;;;;;;;iBAsLM,iBACpB,MAAM,UACN,OAAO,eACP,kBAAkB,qBACjB,QAAQ;UAyGM;EACf;;EAEA,aAAa;EACb;EACA,WAAW;EACX,SAAS;;EAET;;;;;;EAMA;;;;;;EAMA,UAAU;;;;;;;;iBA2CI,4BACd,MAAM,YACN,OAAM,4BACL;;;;UCnhBc;EACf;EACA;;;;;;;UAQe;EACf;EACA;EACA;EACA;EACA;EACA;;;UAIe;EACf;EACA;EACA;EACA;EAIA;;;;;;;;KASU,mBACR,oBACA,oBACA;EACE;;;;;;;;;;;;iBAmBU,qBAAqB,iBAAiB,qBAAqB;;;UCpD1D;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;;;;;;cAQW,8BAA8B;WAGvB,QAAQ;EAF1B,YACE,iBACgB,QAAQ;;;;;;;iBAWZ,0BACd,SAAS,cAAc,aACtB;;;;;;;;iBAuHa,kBACd,SAAS,cAAc,YACvB;EAAQ;IACP"}
|
|
@@ -1,5 +1,5 @@
|
|
|
1
|
-
import { c as AnalystRunInputs, f as AnalystUsageReceipt, i as AnalystFinding, p as EvidenceRef, w as TraceAnalysisStore } from "./types-
|
|
2
|
-
import { c as RegistryRunOpts, n as AnalystRegistry } from "./registry-
|
|
1
|
+
import { c as AnalystRunInputs, f as AnalystUsageReceipt, i as AnalystFinding, p as EvidenceRef, w as TraceAnalysisStore } from "./types-DN2WdT5S.js";
|
|
2
|
+
import { c as RegistryRunOpts, n as AnalystRegistry } from "./registry-7pOUBrtX.js";
|
|
3
3
|
//#region src/analyst/benchmark-scoring.d.ts
|
|
4
4
|
declare function scoreAnalystFindings(testCase: Pick<AnalystBenchmarkCase, 'id' | 'expectedIssues' | 'labeledEvidence'>, findings: readonly AnalystFinding[]): AnalystFindingScore;
|
|
5
5
|
//#endregion
|
|
@@ -233,4 +233,4 @@ declare function registryBenchmarkRunner(options: {
|
|
|
233
233
|
}): AnalystBenchmarkRunner<AnalystRunInputs>;
|
|
234
234
|
//#endregion
|
|
235
235
|
export { scoreAnalystFindings as C, traceStoreEvidenceResolver as S, AnalystIssueExpectation as _, AnalystBenchmarkLabelState as a, registryBenchmarkRunner as b, AnalystBenchmarkProvenance as c, AnalystBenchmarkSummary as d, AnalystEvidenceExpectation as f, AnalystFindingScore as g, AnalystEvidenceResolver as h, AnalystBenchmarkError as i, AnalystBenchmarkResult as l, AnalystEvidenceResolutionError as m, AnalystBenchmarkDatasetRef as n, AnalystBenchmarkObservation as o, AnalystEvidenceResolution as p, AnalystBenchmarkDescriptor as r, AnalystBenchmarkOutput as s, AnalystBenchmarkCase as t, AnalystBenchmarkRunner as u, AnalystLatencyDistribution as v, runAnalystBenchmark as x, RunAnalystBenchmarkOptions as y };
|
|
236
|
-
//# sourceMappingURL=benchmark-
|
|
236
|
+
//# sourceMappingURL=benchmark-BjLGkfnN.d.ts.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"benchmark-
|
|
1
|
+
{"version":3,"file":"benchmark-BjLGkfnN.d.ts","names":[],"sources":["../src/analyst/benchmark-scoring.ts","../src/analyst/benchmark.ts"],"mappings":";;;iBASgB,qBACd,UAAU,KAAK,oEACf,mBAAmB,mBAClB;;;UCIc;EACf;EACA,OAAO;;UAGQ;EACf;EACA;EACA;EACA;EACA,oBAAoB;EACpB;;EAEA,4BAA4B;;KAGlB;UAEK,qBAAqB;EACpC;;EAEA;;EAEA,YAAY;EACZ,OAAO;EACP,yBAAyB;;EAEzB,2BAA2B;EAC3B;EACA,WAAW;;UAGI;EACf;EACA;EACA;EACA;EACA;EACA,mBAAmB;EACnB;EACA;EACA;EACA;;EAEA;;EAEA;;EAEA;EACA;;UAGe;EACf,UAAU;EACV;EACA;;UAGe;EACf;EACA;EACA,oBAAoB;EACpB,QAAQ;;EAER;;KAGU,wBAAwB,qBAAqB;EACvD;EACA,WAAW;EACX,UAAU;EACV,SAAS;gBACK;;;;;iBAMA,2BAA2B,QACzC,WAAW,OAAO,WAAW,qBAC5B,wBAAwB;UAoBV;EACf,mBAAmB;EACnB,QAAQ;EACR,WAAW;;;;;EAKX;;EAEA,QAAQ;;UAGO;EACf;EACA;EACA;EACA;;UAGe,uBAAuB;EACtC;EACA,QACE,OAAO,QACP;IAAW;IAAgB;IAAoB,SAAS;MACvD,yBAAyB,QAAQ;;UAGrB;EACf;EACA;EACA;EACA,YAAY;EACZ;EACA;EACA;EACA;EACA,mBAAmB;EACnB,OAAO;EACP,qBAAqB;EACrB;EACA,eAAe;EACf,QAAQ;EACR,iBAAiB;EACjB,QAAQ;;UAGO;EACf;EACA;EACA;EACA;EACA;;UAGe;EACf;EACA;EACA;EACA;EACA;EACA;EACA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;;EAEA;;EAEA;;EAEA;;EAEA;EACA,WAAW;EACX;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;;UAGe;EACf;EACA;EACA;;UAGe;EACf;EACA,UAAU;EACV;EACA,cAAc;EACd,WAAW;;UAGI,mCAAmC;EAClD;EACA;EACA;EACA;EACA;EACA;EACA;;UAGe;EACf,YAAY;EACZ,cAAc;EACd,WAAW;;UAGI,2BAA2B;EAC1C,gBAAgB,qBAAqB;EACrC,kBAAkB,uBAAuB;EACzC;EACA;EACA;EACA,kBAAkB,wBAAwB;EAC1C,YAAY;;EAEZ,+BAA+B;EAC/B,iBAAiB,aAAa,uCAAuC;EACrE,SAAS;;iBAGW,oBAAoB,QACxC,SAAS,2BAA2B,UACnC,QAAQ;iBAyDK,wBAAwB;EACtC;EACA,UAAU;EACV,aAAa,KAAK;;EAElB;IACE,uBAAuB"}
|
|
@@ -4,14 +4,14 @@ import { r as pairedBootstrap } from "./paired-tests-C8iCsioC.js";
|
|
|
4
4
|
import { a as resolveModelPricing } from "./metrics-Qv-cpptD.js";
|
|
5
5
|
import { a as CostLedgerPersistenceError, i as CostLedger, n as CostCallConflictError, o as CostReceiptCaptureError, r as CostCeilingReachedError, s as CostReservationExceededError, t as CostAccountingIncompleteError } from "./cost-ledger-B1qx30B4.js";
|
|
6
6
|
import { c as writeLedgerFileAtomically, s as withLedgerFileLock } from "./ledger-core-PIfjCbKn.js";
|
|
7
|
-
import { S as fsCampaignStorage, i as startExternalOptimizerModelProxy, r as runWithCleanup, x as createRunCostLedger, y as resolveExternalOptimizerProcessLimits } from "./external-optimizer-subprocess-
|
|
7
|
+
import { S as fsCampaignStorage, i as startExternalOptimizerModelProxy, r as runWithCleanup, x as createRunCostLedger, y as resolveExternalOptimizerProcessLimits } from "./external-optimizer-subprocess-DgNebftP.js";
|
|
8
8
|
import { r as makeFinding, s as usageReceiptFromCostLedger } from "./types-CiWITkGo.js";
|
|
9
|
-
import {
|
|
10
|
-
import { a as callLlmJson, r as LlmResponseError, t as LlmCallError } from "./llm-client-
|
|
11
|
-
import { t as createDspyRlmTraceEngine } from "./dspy-rlm-engine-
|
|
12
|
-
import { i as otlpTextToTraceAnalysisStore, r as createOtlpBufferTraceStore, t as DEFAULT_MAX_TRACE_FILE_BYTES } from "./store-otlp-
|
|
9
|
+
import { A as RAW_FINDING_SCHEMA_PROMPT, M as evidenceRefsFromRawFinding, j as RawAnalystFindingSchema, m as TRACE_ANALYSIS_LIMITS, r as runTraceAnalyst } from "./kind-factory-gP6lDySe.js";
|
|
10
|
+
import { a as callLlmJson, r as LlmResponseError, t as LlmCallError } from "./llm-client-CxQtdtd6.js";
|
|
11
|
+
import { t as createDspyRlmTraceEngine } from "./dspy-rlm-engine-D5byiHn9.js";
|
|
12
|
+
import { i as otlpTextToTraceAnalysisStore, r as createOtlpBufferTraceStore, t as DEFAULT_MAX_TRACE_FILE_BYTES } from "./store-otlp-Dow0pk_5.js";
|
|
13
13
|
import { i as summarizeAnalystBenchmarkRunner, n as runAnalystBenchmark, r as traceStoreEvidenceResolver } from "./benchmark-C4wk_Sjr.js";
|
|
14
|
-
import { n as acquireSingleRunLock } from "./external-optimizer-process-
|
|
14
|
+
import { n as acquireSingleRunLock } from "./external-optimizer-process-Cq_Pg15r.js";
|
|
15
15
|
import { c as primeProtocolSha256, d as runPrimeExchange, f as decodeReplyRows, n as buildPrimePrompt, p as assertEqualDeclarativeTerms, t as analystUsageReceiptFromPrimeUsage, u as projectPrimeTrajectory } from "./prime-protocol-6tZTVsWm.js";
|
|
16
16
|
import { createHash, randomUUID } from "node:crypto";
|
|
17
17
|
import { constants, existsSync, lstatSync, readFileSync } from "node:fs";
|
|
@@ -1412,7 +1412,7 @@ const ANALYST_BENCHMARK_DEPENDENCY_LOCK_FILES = Object.freeze([
|
|
|
1412
1412
|
"package.json",
|
|
1413
1413
|
"pnpm-lock.yaml"
|
|
1414
1414
|
]);
|
|
1415
|
-
const ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256 = "
|
|
1415
|
+
const ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256 = "8e4b431bbba1d1d485847fd21acea07ce78bde91f43b9f64750046d70b7373b1";
|
|
1416
1416
|
const ANALYST_BENCHMARK_EVIDENCE_DEPENDENCY_LOCK_SHA256 = "1e03f2daed356d60316aabefb407ec1e437ac94d408d61eea4ae096e9c6fbb5b";
|
|
1417
1417
|
const ANALYST_BENCHMARK_EVIDENCE_IMPLEMENTATION_SHA256 = "4dba263b6256a30d56c7fdb2d992d3a953c0035d731f359b704db806f68f75ac";
|
|
1418
1418
|
const ANALYST_BENCHMARK_IMPLEMENTATION_FILES = Object.freeze([
|
|
@@ -1525,7 +1525,7 @@ const ANALYST_BENCHMARK_IMPLEMENTATION_FILES = Object.freeze([
|
|
|
1525
1525
|
"src/trace/raw-provider-sink.ts",
|
|
1526
1526
|
"src/verdict-cache.ts"
|
|
1527
1527
|
]);
|
|
1528
|
-
const ANALYST_BENCHMARK_IMPLEMENTATION_SHA256 = "
|
|
1528
|
+
const ANALYST_BENCHMARK_IMPLEMENTATION_SHA256 = "d099d13e08710074eeced469aea8220ef50acf41738372a5c17cb26155628cf7";
|
|
1529
1529
|
function analystBenchmarkImplementationDigest() {
|
|
1530
1530
|
return ANALYST_BENCHMARK_IMPLEMENTATION_SHA256;
|
|
1531
1531
|
}
|
|
@@ -6351,4 +6351,4 @@ function shellQuote(value) {
|
|
|
6351
6351
|
//#endregion
|
|
6352
6352
|
export { ANALYST_BENCHMARK_LOCAL_RECEIPT_FILE as $, effectiveAnalystProtocolSha256 as A, loadCodeTraceVerificationArtifacts as B, adaptPublicBenchmarkFindings as C, renderCodeTraceCalibrationMarkdown as D, readAnalystBenchmarkArtifact as E, publicBenchmarkProtocolSha256 as F, ANALYST_BENCHMARK_EVIDENCE_DEPENDENCY_LOCK_SHA256 as G, ANALYST_BENCHMARK_DEPENDENCY_LOCK_DIGEST_ALGORITHM as H, publicBenchmarkRlmInstructions as I, ANALYST_BENCHMARK_IMPLEMENTATION_FILES as J, ANALYST_BENCHMARK_EVIDENCE_IMPLEMENTATION_SHA256 as K, publicBenchmarkSystemPrompt as L, CODE_TRACE_BENCH_ANALYST_PROMPT as M, MAX_INCORRECT_BLOCKS as N, summarizeCodeTraceCalibration as O, MAX_INCORRECT_BLOCK_STEPS as P, ANALYST_BENCHMARK_COST_LEDGER_FILE as Q, DEFAULT_MAX_VERIFICATION_ARTIFACT_BYTES as R, analystDefinitionProtocolSha256 as S, expandCodeTraceFailureBlocks as T, ANALYST_BENCHMARK_DEPENDENCY_LOCK_FILES as U, parseVerificationOutcome as V, ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256 as W, analystBenchmarkDependencyLockDigest as X, ANALYST_BENCHMARK_IMPLEMENTATION_SHA256 as Y, analystBenchmarkImplementationDigest as Z, createPublicBenchmarkDirectRunner as _, primeCodeTraceAnalystDefinition as a, summarizeAgentRxCalibration as at, AnalystExpressivenessError as b, loadPublicBenchmarkRows as c, agentRxBenchmarkCase as ct, publicBenchmarkSelectionReport as d, roundAgentRxStep as dt, ANALYST_BENCHMARK_MANIFEST_FILE as et, selectPublicBenchmarkRows as f, normalizeBenchmarkLabel as ft, runReplVariableAnalystDefinition as g, rlmEngineLimits as h, primeAnalystProtocolSha256 as i, renderAgentRxCalibrationMarkdown as it, readAnalystInstructionsOverride as j, analystInstructionsOverrideFromText as k, preparePublicAnalystBenchmark as l, agentRxPredictionsToFindings as lt, publicRlmAnalystDefinition as m, renderAnalystBenchmarkMarkdown as n, compareAnalystRunners as nt, runInlineAnalystDefinition as o, codeTraceBenchCase as ot, createPublicBenchmarkRlmRunner as p, ANALYST_BENCHMARK_IMPLEMENTATION_DIGEST_ALGORITHM as q, createPrimeBenchmarkRunner as r, AGENT_RX_UPSTREAM_REVISION as rt, nodeHttpPrimeBridgeTransport as s, codeTracerPredictionsToFindings as st, runAnalystBenchmarkCommand as t, ANALYST_BENCHMARK_OBSERVATIONS_FILE as tt, publicBenchmarkDistributions as u, normalizeAgentRxCategory as ut, publicDirectAnalystDefinition as v, emptyPublicBenchmarkRunner as w, analystDefinitionAsymmetries as x, runChunkedAnalystDefinition as y, appendVerificationArtifactsToOtlp as z };
|
|
6353
6353
|
|
|
6354
|
-
//# sourceMappingURL=benchmark-command-
|
|
6354
|
+
//# sourceMappingURL=benchmark-command-9S20PRel.js.map
|