@tangle-network/agent-eval 0.136.0 → 0.137.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +38 -1
- package/README.md +4 -2
- package/dist/{agent-profile-cell-OhuTee9n.js → agent-profile-cell-CbfBm2g6.js} +2 -2
- package/dist/{agent-profile-cell-OhuTee9n.js.map → agent-profile-cell-CbfBm2g6.js.map} +1 -1
- package/dist/{agent-profile-cell-CCm3l2v2.d.ts → agent-profile-cell-Cw0PVwDr.d.ts} +2 -2
- package/dist/{agent-profile-cell-CCm3l2v2.d.ts.map → agent-profile-cell-Cw0PVwDr.d.ts.map} +1 -1
- package/dist/analyst/index.d.ts +139 -17
- package/dist/analyst/index.d.ts.map +1 -1
- package/dist/analyst/index.js +606 -4
- package/dist/analyst/index.js.map +1 -1
- package/dist/{analyze-runs-jjCmF8pU.js → analyze-runs-BScZvqMV.js} +6 -6
- package/dist/{analyze-runs-jjCmF8pU.js.map → analyze-runs-BScZvqMV.js.map} +1 -1
- package/dist/{analyze-runs-Cda5Xkj1.d.ts → analyze-runs-PVtnfjvA.d.ts} +6 -6
- package/dist/{analyze-runs-Cda5Xkj1.d.ts.map → analyze-runs-PVtnfjvA.d.ts.map} +1 -1
- package/dist/{baseline-BUeFcgrn.js → baseline-C-GocmIW.js} +2 -2
- package/dist/{baseline-BUeFcgrn.js.map → baseline-C-GocmIW.js.map} +1 -1
- package/dist/benchmark-CHX4orG7.d.ts +184 -0
- package/dist/benchmark-CHX4orG7.d.ts.map +1 -0
- package/dist/benchmark-YDrpumqB.js +414 -0
- package/dist/benchmark-YDrpumqB.js.map +1 -0
- package/dist/benchmarks/index.d.ts +1 -1
- package/dist/benchmarks/index.js +1 -1
- package/dist/{benchmarks-Dfgm9ts5.js → benchmarks-DCLkQOmc.js} +3 -3
- package/dist/{benchmarks-Dfgm9ts5.js.map → benchmarks-DCLkQOmc.js.map} +1 -1
- package/dist/builder-eval/index.js +2 -2
- package/dist/campaign/index.d.ts +6 -6
- package/dist/campaign/index.js +3 -3
- package/dist/{campaign-Dz8uQnhC.js → campaign-lgObcHFC.js} +212 -78
- package/dist/campaign-lgObcHFC.js.map +1 -0
- package/dist/cli.js +1 -1
- package/dist/client-C8L6h6Wf.d.ts +202 -0
- package/dist/client-C8L6h6Wf.d.ts.map +1 -0
- package/dist/completion-verifier-DSyRNVzU.d.ts +240 -0
- package/dist/completion-verifier-DSyRNVzU.d.ts.map +1 -0
- package/dist/contract/index.d.ts +10 -9
- package/dist/contract/index.d.ts.map +1 -1
- package/dist/contract/index.js +10 -10
- package/dist/control.d.ts +2 -2
- package/dist/control.js +1 -1
- package/dist/{cost-ledger-DHAjwNj7.js → cost-ledger-D-5_-dhi.js} +2 -2
- package/dist/{cost-ledger-DHAjwNj7.js.map → cost-ledger-D-5_-dhi.js.map} +1 -1
- package/dist/{cost-ledger-fGS_u_O1.d.ts → cost-ledger-D2o6JOrL.d.ts} +2 -2
- package/dist/{cost-ledger-fGS_u_O1.d.ts.map → cost-ledger-D2o6JOrL.d.ts.map} +1 -1
- package/dist/{dataset-BvtnC8Dc.d.ts → dataset-v_Y5902-.d.ts} +2 -2
- package/dist/{dataset-BvtnC8Dc.d.ts.map → dataset-v_Y5902-.d.ts.map} +1 -1
- package/dist/{default-registry-CHmdy2An.js → default-registry-CLXbRt0f.js} +119 -38
- package/dist/default-registry-CLXbRt0f.js.map +1 -0
- package/dist/{default-registry-Brxr728w.d.ts → default-registry-Dc5D_Loc.d.ts} +63 -137
- package/dist/default-registry-Dc5D_Loc.d.ts.map +1 -0
- package/dist/{errors-8YnH8WlF.js → errors-D-LKuDhb.js} +8 -2
- package/dist/errors-D-LKuDhb.js.map +1 -0
- package/dist/{errors-CEk209JS.d.ts → errors-DkfjIDvD.d.ts} +9 -3
- package/dist/errors-DkfjIDvD.d.ts.map +1 -0
- package/dist/{eval-campaign-Cc8WZJ6b.js → eval-campaign-CHqfLnff.js} +6 -6
- package/dist/{eval-campaign-Cc8WZJ6b.js.map → eval-campaign-CHqfLnff.js.map} +1 -1
- package/dist/{extract-usage-DIQpN-ww.js → extract-usage-p-56bh8q.js} +3 -3
- package/dist/{extract-usage-DIQpN-ww.js.map → extract-usage-p-56bh8q.js.map} +1 -1
- package/dist/{feedback-trajectory-CVaeREXV.d.ts → feedback-trajectory-N_F0PwHz.d.ts} +90 -3
- package/dist/feedback-trajectory-N_F0PwHz.d.ts.map +1 -0
- package/dist/fuzz.d.ts +1 -1
- package/dist/fuzz.js +2 -2
- package/dist/hosted/index.d.ts +3 -2
- package/dist/hosted/index.d.ts.map +1 -1
- package/dist/index-BipJlj-C.d.ts +316 -0
- package/dist/index-BipJlj-C.d.ts.map +1 -0
- package/dist/{index-CQsJcqch.d.ts → index-BnP1QJUv.d.ts} +5 -5
- package/dist/{index-CQsJcqch.d.ts.map → index-BnP1QJUv.d.ts.map} +1 -1
- package/dist/{index-B4Fjfo5U.d.ts → index-C-Pr4OWg.d.ts} +88 -317
- package/dist/index-C-Pr4OWg.d.ts.map +1 -0
- package/dist/{index-DuhJaaiH.d.ts → index-DEb46kc6.d.ts} +2 -2
- package/dist/{index-DuhJaaiH.d.ts.map → index-DEb46kc6.d.ts.map} +1 -1
- package/dist/{index-C2fkZhv_.d.ts → index-DRNl6g_N.d.ts} +3 -3
- package/dist/{index-C2fkZhv_.d.ts.map → index-DRNl6g_N.d.ts.map} +1 -1
- package/dist/{index-AbhwHp0V.d.ts → index-U3RHOShi.d.ts} +2 -2
- package/dist/{index-AbhwHp0V.d.ts.map → index-U3RHOShi.d.ts.map} +1 -1
- package/dist/index.d.ts +29 -70
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +507 -33
- package/dist/index.js.map +1 -1
- package/dist/{client-DcvgkaZi.d.ts → insight-report-B9ooYH_g.d.ts} +5 -203
- package/dist/insight-report-B9ooYH_g.d.ts.map +1 -0
- package/dist/integrity-CCXTftiL.js +1360 -0
- package/dist/integrity-CCXTftiL.js.map +1 -0
- package/dist/{integrity-rmVhXWA7.d.ts → integrity-CKxosZ5Z.d.ts} +3 -3
- package/dist/{integrity-rmVhXWA7.d.ts.map → integrity-CKxosZ5Z.d.ts.map} +1 -1
- package/dist/{integrity-BzRbCHzi.js → integrity-fdt8XPAv.js} +2 -2
- package/dist/{integrity-BzRbCHzi.js.map → integrity-fdt8XPAv.js.map} +1 -1
- package/dist/ledger-core/index.d.ts +1 -1
- package/dist/ledger-core/index.js +1 -1
- package/dist/{ledger-core-DAKFKRzi.js → ledger-core-t6sItivm.js} +85 -85
- package/dist/{ledger-core-DAKFKRzi.js.map → ledger-core-t6sItivm.js.map} +1 -1
- package/dist/{llm-client-DHx8pzyJ.js → llm-client-DKB25jV8.js} +3 -3
- package/dist/{llm-client-DHx8pzyJ.js.map → llm-client-DKB25jV8.js.map} +1 -1
- package/dist/meta-eval/index.d.ts +2 -2
- package/dist/meta-eval/index.js +3 -3
- package/dist/{mint-DyRUc9k6.js → mint-Ctwk079K.js} +4 -4
- package/dist/{mint-DyRUc9k6.js.map → mint-Ctwk079K.js.map} +1 -1
- package/dist/multishot/index.d.ts +2 -2
- package/dist/openapi.json +1 -1
- package/dist/{paired-arms-BbFKrAU-.js → paired-arms-iZ08VFMN.js} +3 -3
- package/dist/{paired-arms-BbFKrAU-.js.map → paired-arms-iZ08VFMN.js.map} +1 -1
- package/dist/pipelines/index.js +2 -2
- package/dist/profile-cell.d.ts +1 -1
- package/dist/profile-cell.js +1 -1
- package/dist/{propose-review-control-SQ-n9-We.js → propose-review-control-DLXz4FCX.js} +2 -2
- package/dist/{propose-review-control-SQ-n9-We.js.map → propose-review-control-DLXz4FCX.js.map} +1 -1
- package/dist/registry-BdM7SuTr.d.ts +124 -0
- package/dist/registry-BdM7SuTr.d.ts.map +1 -0
- package/dist/{release-report-DooPguBc.js → release-report-B5XPBvAU.js} +4 -4
- package/dist/{release-report-DooPguBc.js.map → release-report-B5XPBvAU.js.map} +1 -1
- package/dist/{release-report-DpBxGGI1.d.ts → release-report-CofgVNZt.d.ts} +4 -4
- package/dist/{release-report-DpBxGGI1.d.ts.map → release-report-CofgVNZt.d.ts.map} +1 -1
- package/dist/{replay-C6wRg47C.js → replay-Bju0T8Ls.js} +248 -8
- package/dist/replay-Bju0T8Ls.js.map +1 -0
- package/dist/{replay-BRfMIs81.d.ts → replay-K8FaC0CB.d.ts} +227 -52
- package/dist/replay-K8FaC0CB.d.ts.map +1 -0
- package/dist/reporting.d.ts +4 -4
- package/dist/reporting.js +4 -4
- package/dist/{researcher-Doo95b50.d.ts → researcher-Da0Wj-bt.d.ts} +6 -7
- package/dist/researcher-Da0Wj-bt.d.ts.map +1 -0
- package/dist/{reward-hacking-D-QqXvg-.d.ts → reward-hacking-CQ3hTCO3.d.ts} +2 -2
- package/dist/{reward-hacking-D-QqXvg-.d.ts.map → reward-hacking-CQ3hTCO3.d.ts.map} +1 -1
- package/dist/{reward-hacking-a-kYs0-i.js → reward-hacking-GyN0kMd8.js} +3 -3
- package/dist/{reward-hacking-a-kYs0-i.js.map → reward-hacking-GyN0kMd8.js.map} +1 -1
- package/dist/rl.d.ts +6 -6
- package/dist/rl.js +9 -9
- package/dist/rollout/index.d.ts +1 -1
- package/dist/rollout/index.js +3 -3
- package/dist/{rollout-DLSUIWLu.js → rollout-DQFl0UXA.js} +2 -2
- package/dist/{rollout-DLSUIWLu.js.map → rollout-DQFl0UXA.js.map} +1 -1
- package/dist/{rubric-predictive-validity-BJf-8ejY.js → rubric-predictive-validity-BRR632r1.js} +2 -2
- package/dist/{rubric-predictive-validity-BJf-8ejY.js.map → rubric-predictive-validity-BRR632r1.js.map} +1 -1
- package/dist/{rubric-predictive-validity-C1dCLcvb.d.ts → rubric-predictive-validity-C4sztLR3.d.ts} +2 -2
- package/dist/{rubric-predictive-validity-C1dCLcvb.d.ts.map → rubric-predictive-validity-C4sztLR3.d.ts.map} +1 -1
- package/dist/{run-evidence-DokQtX0-.d.ts → run-evidence-BDIircdA.d.ts} +3 -3
- package/dist/{run-evidence-DokQtX0-.d.ts.map → run-evidence-BDIircdA.d.ts.map} +1 -1
- package/dist/{run-record-DcObtIGh.d.ts → run-record-BPCa2rQ8.d.ts} +4 -4
- package/dist/{run-record-DcObtIGh.d.ts.map → run-record-BPCa2rQ8.d.ts.map} +1 -1
- package/dist/{run-record-BIwU2wdV.js → run-record-vRgqWmJw.js} +3 -3
- package/dist/{run-record-BIwU2wdV.js.map → run-record-vRgqWmJw.js.map} +1 -1
- package/dist/{semantic-concept-judge-Btozx3Vc.js → semantic-concept-judge-Bz64IckK.js} +4 -4
- package/dist/{semantic-concept-judge-Btozx3Vc.js.map → semantic-concept-judge-Bz64IckK.js.map} +1 -1
- package/dist/{server-Bz3WQJs6.js → server-KjXZZUDX.js} +3 -3
- package/dist/{server-Bz3WQJs6.js.map → server-KjXZZUDX.js.map} +1 -1
- package/dist/{skill-usage-BDQVPIG1.d.ts → skill-usage-CFDLLlhF.d.ts} +25 -47
- package/dist/skill-usage-CFDLLlhF.d.ts.map +1 -0
- package/dist/{skillopt-optimization-method-CwSYkv35.d.ts → skillopt-optimization-method-BpbnlvAZ.d.ts} +11 -12
- package/dist/skillopt-optimization-method-BpbnlvAZ.d.ts.map +1 -0
- package/dist/{skillopt-optimization-method-0UmPD6aP.js → skillopt-optimization-method-f4o9sUT4.js} +7 -7
- package/dist/{skillopt-optimization-method-0UmPD6aP.js.map → skillopt-optimization-method-f4o9sUT4.js.map} +1 -1
- package/dist/{statistics-CnGCLLqc.js → statistics-ByxzSiOM.js} +2 -2
- package/dist/{statistics-CnGCLLqc.js.map → statistics-ByxzSiOM.js.map} +1 -1
- package/dist/{statistics-CKOqre5S.d.ts → statistics-_7P642CN.d.ts} +2 -2
- package/dist/{statistics-CKOqre5S.d.ts.map → statistics-_7P642CN.d.ts.map} +1 -1
- package/dist/{summary-report-BEk8OFLs.js → summary-report-9A5y7EsK.js} +4 -4
- package/dist/{summary-report-BEk8OFLs.js.map → summary-report-9A5y7EsK.js.map} +1 -1
- package/dist/{summary-report-CPMINBqs.d.ts → summary-report-DHipz9Kx.d.ts} +3 -3
- package/dist/{summary-report-CPMINBqs.d.ts.map → summary-report-DHipz9Kx.d.ts.map} +1 -1
- package/dist/supervisor-run/index.d.ts +3 -2
- package/dist/supervisor-run/index.js +3 -2
- package/dist/{supervisor-run-Dr5HnTup.js → supervisor-run-B2EWUmQY.js} +28 -454
- package/dist/supervisor-run-B2EWUmQY.js.map +1 -0
- package/dist/{test-graded-scenario-BsqWLmPt.js → test-graded-scenario-JHcKQNpq.js} +2 -2
- package/dist/{test-graded-scenario-BsqWLmPt.js.map → test-graded-scenario-JHcKQNpq.js.map} +1 -1
- package/dist/tools-DZk2Jn64.js +1876 -0
- package/dist/tools-DZk2Jn64.js.map +1 -0
- package/dist/traces.d.ts +6 -7
- package/dist/traces.js +5 -6
- package/dist/{types-DVjczBM9.d.ts → types-CKswbJGO.d.ts} +260 -6
- package/dist/types-CKswbJGO.d.ts.map +1 -0
- package/dist/{types-DiWLru6Z.d.ts → types-CTGbIm57.d.ts} +5 -5
- package/dist/{types-DiWLru6Z.d.ts.map → types-CTGbIm57.d.ts.map} +1 -1
- package/dist/types-CTvKfr5F.d.ts +804 -0
- package/dist/types-CTvKfr5F.d.ts.map +1 -0
- package/dist/{index-CyC1BTmn.d.ts → types-Dea6tiVI.d.ts} +16 -238
- package/dist/types-Dea6tiVI.d.ts.map +1 -0
- package/dist/wire/index.d.ts +3 -3
- package/dist/wire/index.js +1 -1
- package/docs/feedback-trajectories.md +100 -1
- package/docs/trace-analysis.md +374 -58
- package/package.json +5 -1
- package/dist/analyst-BkTS3C58.d.ts +0 -89
- package/dist/analyst-BkTS3C58.d.ts.map +0 -1
- package/dist/analyst-j5je5J7c.js +0 -152
- package/dist/analyst-j5je5J7c.js.map +0 -1
- package/dist/campaign-Dz8uQnhC.js.map +0 -1
- package/dist/client-DcvgkaZi.d.ts.map +0 -1
- package/dist/default-registry-Brxr728w.d.ts.map +0 -1
- package/dist/default-registry-CHmdy2An.js.map +0 -1
- package/dist/errors-8YnH8WlF.js.map +0 -1
- package/dist/errors-CEk209JS.d.ts.map +0 -1
- package/dist/feedback-trajectory-CVaeREXV.d.ts.map +0 -1
- package/dist/index-B4Fjfo5U.d.ts.map +0 -1
- package/dist/index-CyC1BTmn.d.ts.map +0 -1
- package/dist/llm-client-BiK4HW0u.d.ts +0 -290
- package/dist/llm-client-BiK4HW0u.d.ts.map +0 -1
- package/dist/raw-provider-sink-BU29Sh8h.d.ts +0 -134
- package/dist/raw-provider-sink-BU29Sh8h.d.ts.map +0 -1
- package/dist/replay-BRfMIs81.d.ts.map +0 -1
- package/dist/replay-C6wRg47C.js.map +0 -1
- package/dist/researcher-Doo95b50.d.ts.map +0 -1
- package/dist/skill-usage-BDQVPIG1.d.ts.map +0 -1
- package/dist/skillopt-optimization-method-CwSYkv35.d.ts.map +0 -1
- package/dist/store-CxJry_cs.d.ts +0 -229
- package/dist/store-CxJry_cs.d.ts.map +0 -1
- package/dist/supervisor-run-Dr5HnTup.js.map +0 -1
- package/dist/tools-D8yTtNSN.js +0 -1190
- package/dist/tools-D8yTtNSN.js.map +0 -1
- package/dist/types-Cc3qbqzj.d.ts +0 -387
- package/dist/types-Cc3qbqzj.d.ts.map +0 -1
- package/dist/types-DVjczBM9.d.ts.map +0 -1
package/dist/index.js
CHANGED
|
@@ -1,46 +1,47 @@
|
|
|
1
1
|
import { t as __exportAll } from "./rolldown-runtime-8H4AJuhK.js";
|
|
2
|
-
import { _ as observeAll, a as jsonlReviewStore, c as scoreFromEvals, d as runAgentControlLoop, f as stopOnNoProgress, g as noProgressDetector, h as errorStreakDetector, i as inMemoryReviewStore, l as allCriticalPassed, m as subjectiveEval, n as runProposeReviewAsControlLoop, o as runProposeReview, p as stopOnRepeatedAction, r as createLlmReviewer, s as controlRunToRunRecord, t as controlFailureClassFromVerification, u as objectiveEval, v as repeatedActionDetector, y as evaluateActionPolicy } from "./propose-review-control-
|
|
3
|
-
import { a as
|
|
4
|
-
import { _ as verifyManifest, a as assertRunAgentProfileCell, c as groupRunsByAgentProfileCell, d as validateAgentProfileCell, f as verifyAgentProfileCell, g as signManifest, h as hashJson, i as agentProfileCellKey, l as requireAgentProfileCell, m as evaluateHypothesis, n as AgentProfileCellValidationError, o as buildAgentInterfaceProfileCell, p as canonicalize, r as agentProfileCellHashMaterial, s as buildAgentProfileCell, t as AGENT_PROFILE_KINDS, u as toAgentProfileJson } from "./agent-profile-cell-
|
|
5
|
-
import {
|
|
2
|
+
import { _ as observeAll, a as jsonlReviewStore, c as scoreFromEvals, d as runAgentControlLoop, f as stopOnNoProgress, g as noProgressDetector, h as errorStreakDetector, i as inMemoryReviewStore, l as allCriticalPassed, m as subjectiveEval, n as runProposeReviewAsControlLoop, o as runProposeReview, p as stopOnRepeatedAction, r as createLlmReviewer, s as controlRunToRunRecord, t as controlFailureClassFromVerification, u as objectiveEval, v as repeatedActionDetector, y as evaluateActionPolicy } from "./propose-review-control-DLXz4FCX.js";
|
|
3
|
+
import { a as LimitExceededError, c as ValidationError, i as JudgeError, l as VerificationError, n as CaptureIntegrityError, o as NotFoundError, r as ConfigError, s as ReplayError, t as AgentEvalError } from "./errors-D-LKuDhb.js";
|
|
4
|
+
import { _ as verifyManifest, a as assertRunAgentProfileCell, c as groupRunsByAgentProfileCell, d as validateAgentProfileCell, f as verifyAgentProfileCell, g as signManifest, h as hashJson, i as agentProfileCellKey, l as requireAgentProfileCell, m as evaluateHypothesis, n as AgentProfileCellValidationError, o as buildAgentInterfaceProfileCell, p as canonicalize, r as agentProfileCellHashMaterial, s as buildAgentProfileCell, t as AGENT_PROFILE_KINDS, u as toAgentProfileJson } from "./agent-profile-cell-CbfBm2g6.js";
|
|
5
|
+
import { L as computeTraceMetrics, R as createChatClient, a as KNOWLEDGE_GAP_KIND_SPEC, d as emitControlIntegrityFindings, f as createTraceAnalystKind, i as KNOWLEDGE_POISONING_KIND_SPEC, l as CONTROL_INTEGRITY_ANALYST, m as renderUpstreamFindings, n as AnalystRegistry, o as IMPROVEMENT_KIND_SPEC, p as renderPriorFindings, r as DEFAULT_TRACE_ANALYST_KINDS, s as FAILURE_MODE_KIND_SPEC, t as buildDefaultAnalystRegistry, u as ControlIntegrityAnalyst, z as createAnalystAi } from "./default-registry-CLXbRt0f.js";
|
|
6
6
|
import { a as estimateTokens, i as estimateCost, n as MetricsCollector, o as isModelPriced, r as TokenCounter, s as resolveModelPricing, t as MODEL_PRICING } from "./metrics-C9YY1OcL.js";
|
|
7
|
-
import { a as CostLedgerPersistenceError, c as costForTokenPricing, i as CostLedger, l as costForUsage, n as CostCallConflictError, o as CostReceiptCaptureError, r as CostCeilingReachedError, s as CostReservationExceededError, t as CostAccountingIncompleteError, u as modelPriceKey } from "./cost-ledger-
|
|
7
|
+
import { a as CostLedgerPersistenceError, c as costForTokenPricing, i as CostLedger, l as costForUsage, n as CostCallConflictError, o as CostReceiptCaptureError, r as CostCeilingReachedError, s as CostReservationExceededError, t as CostAccountingIncompleteError, u as modelPriceKey } from "./cost-ledger-D-5_-dhi.js";
|
|
8
8
|
import { a as providerFromBaseUrl, i as defaultProviderRedactor, n as InMemoryRawProviderSink, r as NoopRawProviderSink, t as FileSystemRawProviderSink } from "./raw-provider-sink-BQd7mzyT.js";
|
|
9
|
-
import { a as assertLlmRoute, c as callLlmJson, d as isTransientLlmError, f as maximumChargeForLlmRequest, i as LlmRouteAssertionError, l as costReceiptFromLlm, m as stripFencedJson, n as LlmClient, o as backoffMs, p as probeLlm, r as LlmResponseError, s as callLlm, t as LlmCallError, u as costReceiptFromLlmError } from "./llm-client-
|
|
9
|
+
import { a as assertLlmRoute, c as callLlmJson, d as isTransientLlmError, f as maximumChargeForLlmRequest, i as LlmRouteAssertionError, l as costReceiptFromLlm, m as stripFencedJson, n as LlmClient, o as backoffMs, p as probeLlm, r as LlmResponseError, s as callLlm, t as LlmCallError, u as costReceiptFromLlmError } from "./llm-client-DKB25jV8.js";
|
|
10
10
|
import { INPUT_VALUE, LLM_CACHED_TOKENS, LLM_CACHED_TOKEN_ATTR_KEYS, LLM_CACHE_WRITE_TOKENS, LLM_CACHE_WRITE_TOKEN_ATTR_KEYS, LLM_CONTEXT_TOKENS, LLM_COST_ATTR_KEYS, LLM_COST_USD, LLM_INPUT_TOKENS, LLM_INPUT_TOKEN_ATTR_KEYS, LLM_MODEL_ATTR_KEYS, LLM_MODEL_NAME, LLM_OUTPUT_TOKENS, LLM_OUTPUT_TOKEN_ATTR_KEYS, LLM_REASONING_TOKENS, LLM_REASONING_TOKEN_ATTR_KEYS, OPENINFERENCE_SPAN_KIND, OUTPUT_VALUE, RUN_COST_ATTR_KEYS, SPAN_KIND_ATTR_KEYS, TOOL_ARGS_CAPTURED, TOOL_LATENCY_MS, TOOL_NAME, TOOL_NAME_ATTR_KEYS, applyLlmSpanOtlpAttributes, asNumber, contextInputTokens, firstNumberAttr } from "./trace-attributes.js";
|
|
11
|
-
import { C as
|
|
11
|
+
import { A as applyToolSpanOtlpAttributes, C as extractOtlpAttributes, D as readOtlpStatus, E as projectOtlpFlatLine, M as isOtlpModelCall, N as traceSpanKindToOpenInferenceKind, T as inferOtlpKind, _ as TraceFileMalformedError, b as TraceNotFoundError, c as otlpTextToTraceAnalysisStore, d as TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, f as TRACE_ANALYSIS_LIMITS, g as TraceAnalysisValidationError, h as TraceAnalysisStoreContractError, i as traceAnalystFunctionGroup, j as classifyOtlpSpanRole, k as stringField, l as createBoundedTraceAnalysisStore, m as TraceAnalysisLimitError, n as buildTraceAnalysisToolDescriptors, o as OtlpFileTraceStore, p as SpanNotFoundError, r as buildTraceAnalystTools, t as TRACE_ANALYST_TOOL_NAMESPACE, u as DEFAULT_TRACE_ANALYST_BUDGETS, v as TraceFileMissingError, w as firstStringAttr, x as asString, y as TraceFileTooLargeError } from "./tools-DZk2Jn64.js";
|
|
12
12
|
import { a as clamp01, c as makeFinding, i as aggregateRunScore, l as makeProposalFinding, r as DEFAULT_RUN_SCORE_WEIGHTS, s as computeFindingId } from "./proposal-findings-DCawte-y.js";
|
|
13
|
+
import { c as SUPERVISOR_RUN_INTEGRITY_SCHEMA, n as supervisorRunRolloutLines, t as analyzeSupervisorRunIntegrity } from "./integrity-CCXTftiL.js";
|
|
14
|
+
import { c as assertMintedLines, f as isRolloutLine, i as ROLLOUT_SCHEMA, m as validateRolloutLine, p as isTrainableSplit, s as assertMinted, u as assertRolloutLine } from "./schema-C6DW4ZHR.js";
|
|
15
|
+
import { a as scoreOrigin, n as observedScore, o as trainingReward, r as observedSplitScore, s as trainingScore, t as isRealnessGated } from "./reward-nw2xZGZG.js";
|
|
13
16
|
import { n as mapConcurrent, t as Mutex } from "./concurrency-MUjT7VjM.js";
|
|
14
|
-
import { a as RunCritic, d as defaultIsMaterial, f as diffFindings, i as runSemanticConceptJudge, n as SEMANTIC_CONCEPT_JUDGE_VERSION, o as SKILL_USAGE_ANALYST, p as LockedJsonlAppender, r as createSemanticConceptJudge, s as SkillUsageAnalyst, t as DEFAULT_COMPLEXITY_WEIGHTS, u as FindingsStore } from "./semantic-concept-judge-
|
|
15
|
-
import { At as dominates, B as surfaceContentHash, Bt as assertRealAgentReceipts, Ct as llmJudge, D as runCanaries, Dt as fileVerdictCache, Et as contentHash, Ft as decidePairedPromotion, G as DEFAULT_MUTATION_PRIMITIVES, Ht as summarizeAgentReceiptIntegrity, It as pairedDecisionShape, K as buildReflectionPrompt, Lt as minimumPairsForPairedDeltaTest, Mt as paretoFrontierWithCrowding, Nt as scalarScore, Ot as inMemoryVerdictCache, Rt as pairedDeltaTest, St as hashScenarios, Tt as canonicalJson, Ut as summarizeBackendIntegrity, Vt as assertRealBackend, Wt as JudgeParseError, _t as redTeamReport, bt as Dataset, dt as REFERENCE_EQUIVALENCE_INPUT_LIMITS, ft as REFERENCE_EQUIVALENCE_JUDGE_VERSION, gt as redTeamDataset, ht as DEFAULT_RED_TEAM_CORPUS, jt as paretoFrontier, kt as crowdingDistance, mt as runReferenceEquivalenceJudge, pt as createReferenceEquivalenceJudge, q as parseReflectionResponse, vt as scoreRedTeamOutput, wt as cachedJudge, xt as HoldoutLockedError, yt as toolNamesForRun, zt as BackendIntegrityError } from "./skillopt-optimization-method-
|
|
16
|
-
import { $ as selfPreference, A as pairedRiskDifference, B as requiredSampleSize, C as mulberry32, D as pairedCohensDz, E as pairedBootstrap, F as partialCredit, G as wilson, H as weightedComposite, I as passAtK, J as normalCdf, K as studentTCdf, L as pearsonR, M as pairedRiskDifferenceScore, N as pairedSignTest, O as pairedDeltaTieFraction, P as pairedTTest, Q as positionalBias, R as ranks, S as mcnemarRequiredN, T as pairedBinaryScale, U as weightedMean, V as spearmanR, W as wilcoxonSignedRank, X as calibrateJudgeContinuous, Y as calibrateJudge, Z as continuousAgreement, _ as interpretCliffs, a as MANN_WHITNEY_EXACT_MAX_WORK, b as mcnemar, c as bonferroni, d as confidenceInterval, et as verbosityBias, f as corpusInterRaterAgreement, g as interRaterReliability, h as holm, i as MANN_WHITNEY_EXACT_MAX_STATES, j as pairedRiskDifferenceExact, k as pairedMde, l as cliffsDelta, m as eProcess, n as DECISION_PAIRED_DELTA_STATISTIC, o as WILCOXON_EXACT_MAX_N, p as corpusInterRaterAgreementFromJudgeScores, r as DEFAULT_PERMUTATIONS, s as benjaminiHochberg, t as BOOTSTRAP_GATE_MIN_N, u as cohensD, v as isBinaryOutcomeVector, w as normalizeScores, x as mcnemarPower, y as mannWhitneyU, z as requiredPairedSampleSize } from "./statistics-
|
|
17
|
-
import { n as pairArms, r as pairRunRecords, t as comparePairedArms } from "./paired-arms-
|
|
17
|
+
import { a as RunCritic, d as defaultIsMaterial, f as diffFindings, i as runSemanticConceptJudge, n as SEMANTIC_CONCEPT_JUDGE_VERSION, o as SKILL_USAGE_ANALYST, p as LockedJsonlAppender, r as createSemanticConceptJudge, s as SkillUsageAnalyst, t as DEFAULT_COMPLEXITY_WEIGHTS, u as FindingsStore } from "./semantic-concept-judge-Bz64IckK.js";
|
|
18
|
+
import { At as dominates, B as surfaceContentHash, Bt as assertRealAgentReceipts, Ct as llmJudge, D as runCanaries, Dt as fileVerdictCache, Et as contentHash, Ft as decidePairedPromotion, G as DEFAULT_MUTATION_PRIMITIVES, Ht as summarizeAgentReceiptIntegrity, It as pairedDecisionShape, K as buildReflectionPrompt, Lt as minimumPairsForPairedDeltaTest, Mt as paretoFrontierWithCrowding, Nt as scalarScore, Ot as inMemoryVerdictCache, Rt as pairedDeltaTest, St as hashScenarios, Tt as canonicalJson, Ut as summarizeBackendIntegrity, Vt as assertRealBackend, Wt as JudgeParseError, _t as redTeamReport, bt as Dataset, dt as REFERENCE_EQUIVALENCE_INPUT_LIMITS, ft as REFERENCE_EQUIVALENCE_JUDGE_VERSION, gt as redTeamDataset, ht as DEFAULT_RED_TEAM_CORPUS, jt as paretoFrontier, kt as crowdingDistance, mt as runReferenceEquivalenceJudge, pt as createReferenceEquivalenceJudge, q as parseReflectionResponse, vt as scoreRedTeamOutput, wt as cachedJudge, xt as HoldoutLockedError, yt as toolNamesForRun, zt as BackendIntegrityError } from "./skillopt-optimization-method-f4o9sUT4.js";
|
|
19
|
+
import { $ as selfPreference, A as pairedRiskDifference, B as requiredSampleSize, C as mulberry32, D as pairedCohensDz, E as pairedBootstrap, F as partialCredit, G as wilson, H as weightedComposite, I as passAtK, J as normalCdf, K as studentTCdf, L as pearsonR, M as pairedRiskDifferenceScore, N as pairedSignTest, O as pairedDeltaTieFraction, P as pairedTTest, Q as positionalBias, R as ranks, S as mcnemarRequiredN, T as pairedBinaryScale, U as weightedMean, V as spearmanR, W as wilcoxonSignedRank, X as calibrateJudgeContinuous, Y as calibrateJudge, Z as continuousAgreement, _ as interpretCliffs, a as MANN_WHITNEY_EXACT_MAX_WORK, b as mcnemar, c as bonferroni, d as confidenceInterval, et as verbosityBias, f as corpusInterRaterAgreement, g as interRaterReliability, h as holm, i as MANN_WHITNEY_EXACT_MAX_STATES, j as pairedRiskDifferenceExact, k as pairedMde, l as cliffsDelta, m as eProcess, n as DECISION_PAIRED_DELTA_STATISTIC, o as WILCOXON_EXACT_MAX_N, p as corpusInterRaterAgreementFromJudgeScores, r as DEFAULT_PERMUTATIONS, s as benjaminiHochberg, t as BOOTSTRAP_GATE_MIN_N, u as cohensD, v as isBinaryOutcomeVector, w as normalizeScores, x as mcnemarPower, y as mannWhitneyU, z as requiredPairedSampleSize } from "./statistics-ByxzSiOM.js";
|
|
20
|
+
import { n as pairArms, r as pairRunRecords, t as comparePairedArms } from "./paired-arms-iZ08VFMN.js";
|
|
18
21
|
import { n as llmSpanFromProvider, t as TraceEmitter } from "./emitter-CPBAhxum.js";
|
|
19
|
-
import {
|
|
22
|
+
import { h as hashCanonical, m as canonicalString } from "./ledger-core-t6sItivm.js";
|
|
20
23
|
import { a as isRetrievalSpan, i as isLlmSpan, n as TRACE_SCHEMA_VERSION, o as isSandboxSpan, r as isJudgeSpan, s as isToolSpan, t as FAILURE_CLASSES } from "./schema-CRhEY1SO.js";
|
|
21
|
-
import { a as roundTripRunRecord, i as parseRunRecordSafe, n as isRunRecord, o as runTaskScore, r as modelHasSnapshot, s as validateRunRecord, t as RunRecordValidationError } from "./run-record-
|
|
22
|
-
import { a as evaluateReleaseConfidence, i as assertReleaseConfidence, n as bootstrapCi, r as judgeReplayGate, t as renderReleaseReport } from "./release-report-
|
|
23
|
-
import { c as assertMintedLines, f as isRolloutLine, i as ROLLOUT_SCHEMA, m as validateRolloutLine, p as isTrainableSplit, s as assertMinted, u as assertRolloutLine } from "./schema-C6DW4ZHR.js";
|
|
24
|
+
import { a as roundTripRunRecord, i as parseRunRecordSafe, n as isRunRecord, o as runTaskScore, r as modelHasSnapshot, s as validateRunRecord, t as RunRecordValidationError } from "./run-record-vRgqWmJw.js";
|
|
25
|
+
import { a as evaluateReleaseConfidence, i as assertReleaseConfidence, n as bootstrapCi, r as judgeReplayGate, t as renderReleaseReport } from "./release-report-B5XPBvAU.js";
|
|
24
26
|
import { i as toRewardRows, r as toJsonl, s as toSftRows } from "./exporters-q9iL-2Jf.js";
|
|
25
|
-
import { a as toHarborTrajectories, i as relabelImportedSplit, n as HARBOR_IMPORT_GAP, o as toHarborTrajectory, r as fromHarborTrajectory, t as ATIF_SCHEMA_VERSION } from "./rollout-
|
|
27
|
+
import { a as toHarborTrajectories, i as relabelImportedSplit, n as HARBOR_IMPORT_GAP, o as toHarborTrajectory, r as fromHarborTrajectory, t as ATIF_SCHEMA_VERSION } from "./rollout-DQFl0UXA.js";
|
|
26
28
|
import { t as buildTrajectory } from "./trajectory-D_7rLrvE.js";
|
|
27
|
-
import { n as unmintableReasons, t as mintRolloutRows } from "./mint-
|
|
28
|
-
import {
|
|
29
|
-
import {
|
|
30
|
-
import {
|
|
31
|
-
import {
|
|
32
|
-
import {
|
|
33
|
-
import { n as iqr, r as welchsTTest, t as compareToBaseline } from "./baseline-BUeFcgrn.js";
|
|
29
|
+
import { n as unmintableReasons, t as mintRolloutRows } from "./mint-Ctwk079K.js";
|
|
30
|
+
import { C as SUPERVISOR_RUN_SCHEMA, T as showMeasured, _ as readClaudeCodeSupervisorRun, b as rollupSupervisorRuns, c as writeSupervisorRunReport, d as renderSupervisorRunHeadline, f as renderSupervisorRunMarkdown, g as claudeCodeSupervisorRunReader, t as analyzeSupervisorRun, v as analyzeSupervisorRunSources, w as isUnavailable } from "./supervisor-run-B2EWUmQY.js";
|
|
31
|
+
import { A as TRACE_ANALYST_ACTOR_DESCRIPTION, C as domainEvidencePattern, D as tokenizeDomainWords, E as scoreTraceInsightReadiness, O as traceAnalystOnRunComplete, S as describeTraceInsightScope, T as planTraceInsightQuestions, _ as otlpToTraceRunRecords, a as convertTraceStoresToOtlp, b as buildTraceInsightPrompt, c as otelRunCompleteHook, d as captureFetchToRawSink, f as ToolTraceMissingError, g as otlpToRunRecords, h as otlpRowsToTraceRunRecords, i as iterateRawCalls, j as TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION, k as analyzeTraces, l as OTEL_AGENT_EVAL_SCOPE, m as otlpRowsToRunRecords, n as ReplayCacheMissError, o as createOtelExporter, p as toolSpansToTraceAnalysisStore, r as createReplayFetch, s as createOtelTracingStore, t as ReplayCache, u as exportRunAsOtlp, v as flattenOtlpExportToNdjson, w as inferDomainKeywords, x as defaultTraceInsightPanel, y as buildTraceInsightContext } from "./replay-Bju0T8Ls.js";
|
|
32
|
+
import { n as extractUsageFromResponse, r as extractUsageFromSse, t as extractUsage } from "./extract-usage-p-56bh8q.js";
|
|
33
|
+
import { B as agentProfileModelId, G as createLlmCorrectnessChecker, H as harnessAxisOf, I as CODING_HARNESSES, J as verifyCompletion, K as createTokenRecallChecker, L as HARNESS_NATIVE_MODEL, R as agentProfileHash, U as extractProducedState, V as expandProfileAxes, W as completionVerdict, q as parseCorrectnessResponse, z as agentProfileId } from "./campaign-lgObcHFC.js";
|
|
34
|
+
import { n as iqr, r as welchsTTest, t as compareToBaseline } from "./baseline-C-GocmIW.js";
|
|
34
35
|
import { a as judgeSpans, c as runsForScenario, i as hasCapturedToolArgs, l as toolSpans, n as argHash, o as llmSpans, r as groupBy, s as runFailureClass, t as aggregateLlm } from "./query-Di7eEQ79.js";
|
|
35
|
-
import { a as checkBehavioralCanary, i as canaryLeakView, o as checkCanaries, r as HoldoutAuditor, s as runBehavioralCanaries, t as analyzeRuns } from "./analyze-runs-
|
|
36
|
-
import { a as summaryTable, i as researchReport, n as gainHistogram, r as paretoChart, t as RESEARCH_REPORT_HARD_PAIR_FLOOR } from "./summary-report-
|
|
37
|
-
import { a as composeParsers, c as vitestTestParser, i as SubprocessSandboxDriver, n as DockerSandboxDriver, o as jestTestParser, r as SandboxHarness, s as pytestTestParser, t as runTestGradedScenario } from "./test-graded-scenario-
|
|
38
|
-
import { a as InMemoryTraceStore, i as FileSystemTraceStore, n as assertRunCaptured, r as throwIfRunIncomplete, t as RunIntegrityError } from "./integrity-
|
|
36
|
+
import { a as checkBehavioralCanary, i as canaryLeakView, o as checkCanaries, r as HoldoutAuditor, s as runBehavioralCanaries, t as analyzeRuns } from "./analyze-runs-BScZvqMV.js";
|
|
37
|
+
import { a as summaryTable, i as researchReport, n as gainHistogram, r as paretoChart, t as RESEARCH_REPORT_HARD_PAIR_FLOOR } from "./summary-report-9A5y7EsK.js";
|
|
38
|
+
import { a as composeParsers, c as vitestTestParser, i as SubprocessSandboxDriver, n as DockerSandboxDriver, o as jestTestParser, r as SandboxHarness, s as pytestTestParser, t as runTestGradedScenario } from "./test-graded-scenario-JHcKQNpq.js";
|
|
39
|
+
import { a as InMemoryTraceStore, i as FileSystemTraceStore, n as assertRunCaptured, r as throwIfRunIncomplete, t as RunIntegrityError } from "./integrity-fdt8XPAv.js";
|
|
39
40
|
import { i as redactValue, n as REDACTION_VERSION, r as redactString, t as DEFAULT_REDACTION_RULES } from "./redact-7Aq1ukl-.js";
|
|
40
41
|
import { n as DEFAULT_RULES, r as classifyFailure, t as computeToolUseMetrics } from "./tool-use-metrics-DEGMKycK.js";
|
|
41
42
|
import { t as analyzeSeries } from "./series-convergence-CjO2QdRW.js";
|
|
42
|
-
import { _ as deterministicSplit, g as BENCHMARK_SPLIT_SEED, t as benchmarks_exports } from "./benchmarks-
|
|
43
|
-
import { t as runEvalCampaign } from "./eval-campaign-
|
|
43
|
+
import { _ as deterministicSplit, g as BENCHMARK_SPLIT_SEED, t as benchmarks_exports } from "./benchmarks-DCLkQOmc.js";
|
|
44
|
+
import { t as runEvalCampaign } from "./eval-campaign-CHqfLnff.js";
|
|
44
45
|
import { n as pairedEvalueSequence, t as evaluateInterimReleaseConfidence } from "./sequential-Br0mAPHA.js";
|
|
45
46
|
import { AxGEPA, AxSignature, ax } from "@ax-llm/ax";
|
|
46
47
|
import { accessSync, appendFileSync, constants, cpSync, existsSync, mkdirSync, promises, readFileSync, readdirSync, statSync, writeFileSync } from "node:fs";
|
|
@@ -1602,6 +1603,343 @@ async function decideNextUserTurn(chat, opts) {
|
|
|
1602
1603
|
return paid.value.content.trim();
|
|
1603
1604
|
}
|
|
1604
1605
|
//#endregion
|
|
1606
|
+
//#region src/feedback-trajectory-review.ts
|
|
1607
|
+
/** Bind an analyst finding's complete canonical JSON content to a stable digest. */
|
|
1608
|
+
function analystFindingDigest(finding) {
|
|
1609
|
+
return hashCanonical(snapshotAnalystFinding(finding, "analyst finding"));
|
|
1610
|
+
}
|
|
1611
|
+
/** Bind the complete analyst result to one immutable review target. */
|
|
1612
|
+
function analystRunDigest(run) {
|
|
1613
|
+
return hashCanonical(snapshotAnalystRun(run, "analyst run"));
|
|
1614
|
+
}
|
|
1615
|
+
function snapshotAnalystRun(value, context = "analyst run") {
|
|
1616
|
+
let snapshot;
|
|
1617
|
+
try {
|
|
1618
|
+
snapshot = JSON.parse(canonicalString(value));
|
|
1619
|
+
} catch (cause) {
|
|
1620
|
+
throw new TypeError(`${context} must have a canonical JSON representation`, { cause });
|
|
1621
|
+
}
|
|
1622
|
+
if (!isRecord$1(snapshot)) throw new TypeError(`${context} must be an object`);
|
|
1623
|
+
assertOnlyKeys(snapshot, [
|
|
1624
|
+
"run_id",
|
|
1625
|
+
"correlation_id",
|
|
1626
|
+
"started_at",
|
|
1627
|
+
"ended_at",
|
|
1628
|
+
"findings",
|
|
1629
|
+
"per_analyst",
|
|
1630
|
+
"total_cost_usd",
|
|
1631
|
+
"total_cost_provenance"
|
|
1632
|
+
], context);
|
|
1633
|
+
requiredString(snapshot.run_id, `${context} run_id`);
|
|
1634
|
+
requiredString(snapshot.correlation_id, `${context} correlation_id`);
|
|
1635
|
+
canonicalTimestamp(snapshot.started_at, `${context} started_at`);
|
|
1636
|
+
canonicalTimestamp(snapshot.ended_at, `${context} ended_at`);
|
|
1637
|
+
snapshot.findings = snapshotAnalystFindings(snapshot.findings, `${context} findings`);
|
|
1638
|
+
if (!Array.isArray(snapshot.per_analyst)) throw new TypeError(`${context} per_analyst must be an array`);
|
|
1639
|
+
for (const [index, summary] of snapshot.per_analyst.entries()) {
|
|
1640
|
+
if (!isRecord$1(summary)) throw new TypeError(`${context} per_analyst ${index} must be an object`);
|
|
1641
|
+
requiredString(summary.analyst_id, `${context} per_analyst ${index} analyst_id`);
|
|
1642
|
+
}
|
|
1643
|
+
if (typeof snapshot.total_cost_usd !== "number" || !Number.isFinite(snapshot.total_cost_usd) || snapshot.total_cost_usd < 0) throw new TypeError(`${context} total_cost_usd must be a finite non-negative number`);
|
|
1644
|
+
if (snapshot.total_cost_provenance !== void 0 && !isRecord$1(snapshot.total_cost_provenance)) throw new TypeError(`${context} total_cost_provenance must be an object`);
|
|
1645
|
+
return snapshot;
|
|
1646
|
+
}
|
|
1647
|
+
function snapshotAnalystFindings(value, context = "analyst run findings") {
|
|
1648
|
+
if (!Array.isArray(value)) throw new TypeError(`${context} must be an array`);
|
|
1649
|
+
const findings = value.map((finding, index) => snapshotAnalystFinding(finding, `${context} finding ${index}`));
|
|
1650
|
+
assertUniqueFindingIds(findings.map((finding) => finding.finding_id));
|
|
1651
|
+
return findings;
|
|
1652
|
+
}
|
|
1653
|
+
function readAnalystReview(trajectory) {
|
|
1654
|
+
const analystAttempts = trajectory.attempts.filter((attempt) => isRecord$1(attempt.artifact) && attempt.artifact.type === "analyst-run");
|
|
1655
|
+
const analysis = isRecord$1(trajectory.metadata?.analysis) ? trajectory.metadata.analysis : void 0;
|
|
1656
|
+
if (analystAttempts.length === 0) {
|
|
1657
|
+
if (analysis?.kind === "analyst-run") throw new TypeError(`feedbackTrajectoryToOptimizerRow: analyst trajectory "${trajectory.id}" is missing its archived run`);
|
|
1658
|
+
return;
|
|
1659
|
+
}
|
|
1660
|
+
if (analystAttempts.length !== 1) throw new TypeError(`feedbackTrajectoryToOptimizerRow: analyst trajectory "${trajectory.id}" must contain exactly one archived run`);
|
|
1661
|
+
if (analysis?.kind !== "analyst-run") throw new TypeError(`feedbackTrajectoryToOptimizerRow: analyst trajectory "${trajectory.id}" is missing review state`);
|
|
1662
|
+
const artifact = analystAttempts[0].artifact;
|
|
1663
|
+
if (!isRecord$1(artifact)) throw new TypeError(`feedbackTrajectoryToOptimizerRow: analyst trajectory "${trajectory.id}" has an invalid archived run`);
|
|
1664
|
+
const runId = requiredString(artifact.analystRunId, `analyst trajectory "${trajectory.id}" run id`);
|
|
1665
|
+
if (analysis.runId !== runId) throw new TypeError(`feedbackTrajectoryToOptimizerRow: analyst trajectory "${trajectory.id}" run identity does not match its review state`);
|
|
1666
|
+
const artifactRunDigest = requiredDigest(artifact.runDigest, `analyst trajectory "${trajectory.id}" archived run digest`);
|
|
1667
|
+
const storedRunDigest = requiredDigest(analysis.runDigest, `analyst trajectory "${trajectory.id}" review run digest`);
|
|
1668
|
+
if (artifactRunDigest !== storedRunDigest) throw new TypeError(`feedbackTrajectoryToOptimizerRow: analyst trajectory "${trajectory.id}" run digest does not match its review state`);
|
|
1669
|
+
const findings = snapshotAnalystFindings(artifact.findings, `analyst trajectory "${trajectory.id}"`);
|
|
1670
|
+
const findingIds = findings.map((finding) => finding.finding_id);
|
|
1671
|
+
const analystIds = stringArray(artifact.analystIds, `analyst trajectory "${trajectory.id}" analyst ids`);
|
|
1672
|
+
const attemptMetadata = analystAttempts[0].metadata;
|
|
1673
|
+
if (!isRecord$1(attemptMetadata)) throw new TypeError(`feedbackTrajectoryToOptimizerRow: analyst trajectory "${trajectory.id}" is missing archived run metadata`);
|
|
1674
|
+
const archivedRun = snapshotAnalystRun({
|
|
1675
|
+
run_id: runId,
|
|
1676
|
+
correlation_id: artifact.correlationId,
|
|
1677
|
+
started_at: analysis.startedAt,
|
|
1678
|
+
ended_at: analysis.endedAt,
|
|
1679
|
+
findings,
|
|
1680
|
+
per_analyst: attemptMetadata.perAnalyst,
|
|
1681
|
+
total_cost_usd: analysis.knownCostUsd,
|
|
1682
|
+
...analysis.costProvenance === void 0 ? {} : { total_cost_provenance: analysis.costProvenance }
|
|
1683
|
+
}, `analyst trajectory "${trajectory.id}" archived run`);
|
|
1684
|
+
const knownAnalystIds = new Set(analystIds);
|
|
1685
|
+
for (const [index, finding] of findings.entries()) if (!knownAnalystIds.has(finding.analyst_id)) throw new TypeError(`feedbackTrajectoryToOptimizerRow: analyst trajectory "${trajectory.id}" omits generating analyst "${finding.analyst_id}" at finding ${index}`);
|
|
1686
|
+
const reviewDecisions = validateAnalystReviewDecisions({
|
|
1687
|
+
runId,
|
|
1688
|
+
runDigest: storedRunDigest,
|
|
1689
|
+
findings,
|
|
1690
|
+
analystIds,
|
|
1691
|
+
decisions: analysis.reviewDecisions,
|
|
1692
|
+
requireComplete: true
|
|
1693
|
+
});
|
|
1694
|
+
const expectedRunDigest = analystRunDigest(archivedRun);
|
|
1695
|
+
if (storedRunDigest !== expectedRunDigest) throw new TypeError(`feedbackTrajectoryToOptimizerRow: analyst trajectory "${trajectory.id}" archived run digest mismatch`);
|
|
1696
|
+
return {
|
|
1697
|
+
runId,
|
|
1698
|
+
runDigest: expectedRunDigest,
|
|
1699
|
+
findings,
|
|
1700
|
+
findingIds,
|
|
1701
|
+
analystIds,
|
|
1702
|
+
reviewDecisions
|
|
1703
|
+
};
|
|
1704
|
+
}
|
|
1705
|
+
function completedAnalystReviewQuality(review) {
|
|
1706
|
+
const findingDecisions = review.reviewDecisions.filter((decision) => decision.verdict !== "completeness_assessed");
|
|
1707
|
+
const completeness = review.reviewDecisions.filter((decision) => decision.verdict === "completeness_assessed");
|
|
1708
|
+
if (completeness.length !== 1) throw new TypeError("feedbackTrajectoryToOptimizerRow: analyst run requires exactly one independent completeness_assessed decision");
|
|
1709
|
+
const confirmed = findingDecisions.filter((decision) => decision.verdict === "confirmed").length;
|
|
1710
|
+
const rejected = findingDecisions.length - confirmed;
|
|
1711
|
+
const emitted = review.findingIds.length;
|
|
1712
|
+
const missed = completeness[0].missedIssues.length;
|
|
1713
|
+
const precision = emitted === 0 ? 1 : confirmed / emitted;
|
|
1714
|
+
const recallDenominator = confirmed + missed;
|
|
1715
|
+
const recall = recallDenominator === 0 ? 1 : confirmed / recallDenominator;
|
|
1716
|
+
return {
|
|
1717
|
+
precision,
|
|
1718
|
+
recall,
|
|
1719
|
+
f1: precision + recall === 0 ? 0 : 2 * precision * recall / (precision + recall),
|
|
1720
|
+
counts: {
|
|
1721
|
+
emitted,
|
|
1722
|
+
confirmed,
|
|
1723
|
+
rejected,
|
|
1724
|
+
missed
|
|
1725
|
+
}
|
|
1726
|
+
};
|
|
1727
|
+
}
|
|
1728
|
+
function validateAnalystReviewDecisions(input) {
|
|
1729
|
+
if (!Array.isArray(input.decisions)) throw new TypeError("analyst review decisions must be an array");
|
|
1730
|
+
const findings = snapshotAnalystFindings(input.findings);
|
|
1731
|
+
const expectedRunDigest = requiredDigest(input.runDigest, "analyst review run digest");
|
|
1732
|
+
const findingsById = new Map(findings.map((finding) => [finding.finding_id, finding]));
|
|
1733
|
+
const generatingAnalystIds = new Set(input.analystIds);
|
|
1734
|
+
const seenFindingIds = /* @__PURE__ */ new Set();
|
|
1735
|
+
let completenessCount = 0;
|
|
1736
|
+
const decisions = input.decisions.map((value, index) => {
|
|
1737
|
+
if (!isRecord$1(value)) throw new TypeError(`analyst review decision ${index} must be an object`);
|
|
1738
|
+
const source = requiredString(value.source, `analyst review decision ${index} source`);
|
|
1739
|
+
if (!isAnalystReviewSource(source)) throw new TypeError(`analyst review decision ${index} source must be user, judge, environment, metric, or policy`);
|
|
1740
|
+
const reviewerId = requiredString(value.reviewerId, `analyst review decision ${index} reviewerId`);
|
|
1741
|
+
if (generatingAnalystIds.has(reviewerId)) throw new TypeError(`analyst review decision ${index} reviewerId must differ from the generating analyst`);
|
|
1742
|
+
const reviewId = requiredString(value.reviewId, `analyst review decision ${index} reviewId`);
|
|
1743
|
+
if (reviewId === input.runId) throw new TypeError(`analyst review decision ${index} reviewId must identify an independent review`);
|
|
1744
|
+
const reason = requiredString(value.reason, `analyst review decision ${index} reason`);
|
|
1745
|
+
const decidedAt = canonicalTimestamp(value.decidedAt, `analyst review decision ${index} decidedAt`);
|
|
1746
|
+
const runDigest = requiredDigest(value.runDigest, `analyst review decision ${index} runDigest`);
|
|
1747
|
+
if (runDigest !== expectedRunDigest) throw new TypeError(`analyst review decision ${index} run digest mismatch`);
|
|
1748
|
+
if (value.verdict === "completeness_assessed") {
|
|
1749
|
+
assertOnlyKeys(value, [
|
|
1750
|
+
"runDigest",
|
|
1751
|
+
"verdict",
|
|
1752
|
+
"missedIssues",
|
|
1753
|
+
"source",
|
|
1754
|
+
"reviewerId",
|
|
1755
|
+
"reviewId",
|
|
1756
|
+
"reason",
|
|
1757
|
+
"decidedAt"
|
|
1758
|
+
], `analyst review decision ${index}`);
|
|
1759
|
+
completenessCount += 1;
|
|
1760
|
+
if (completenessCount > 1) throw new TypeError("duplicate completeness_assessed analyst review decision");
|
|
1761
|
+
return {
|
|
1762
|
+
runDigest,
|
|
1763
|
+
verdict: "completeness_assessed",
|
|
1764
|
+
missedIssues: validateMissedIssues(value.missedIssues, findingsById, `analyst review decision ${index}`),
|
|
1765
|
+
source,
|
|
1766
|
+
reviewerId,
|
|
1767
|
+
reviewId,
|
|
1768
|
+
reason,
|
|
1769
|
+
decidedAt
|
|
1770
|
+
};
|
|
1771
|
+
}
|
|
1772
|
+
if (value.verdict !== "confirmed" && value.verdict !== "rejected") throw new TypeError(`analyst review decision ${index} verdict must be confirmed, rejected, or completeness_assessed`);
|
|
1773
|
+
assertOnlyKeys(value, [
|
|
1774
|
+
"runDigest",
|
|
1775
|
+
"findingId",
|
|
1776
|
+
"findingDigest",
|
|
1777
|
+
"verdict",
|
|
1778
|
+
"source",
|
|
1779
|
+
"reviewerId",
|
|
1780
|
+
"reviewId",
|
|
1781
|
+
"reason",
|
|
1782
|
+
"decidedAt"
|
|
1783
|
+
], `analyst review decision ${index}`);
|
|
1784
|
+
const findingId = requiredString(value.findingId, `analyst review decision ${index} findingId`);
|
|
1785
|
+
const finding = findingsById.get(findingId);
|
|
1786
|
+
if (!finding) throw new TypeError(`analyst review decision references unknown finding id "${findingId}"`);
|
|
1787
|
+
if (seenFindingIds.has(findingId)) throw new TypeError(`duplicate analyst review decision for finding id "${findingId}"`);
|
|
1788
|
+
seenFindingIds.add(findingId);
|
|
1789
|
+
const findingDigest = requiredString(value.findingDigest, `analyst review decision ${index} findingDigest`);
|
|
1790
|
+
const expectedDigest = analystFindingDigest(finding);
|
|
1791
|
+
if (findingDigest !== expectedDigest) throw new TypeError(`analyst review decision ${index} digest mismatch for finding id "${findingId}"`);
|
|
1792
|
+
return {
|
|
1793
|
+
runDigest,
|
|
1794
|
+
findingId,
|
|
1795
|
+
findingDigest: expectedDigest,
|
|
1796
|
+
verdict: value.verdict,
|
|
1797
|
+
source,
|
|
1798
|
+
reviewerId,
|
|
1799
|
+
reviewId,
|
|
1800
|
+
reason,
|
|
1801
|
+
decidedAt
|
|
1802
|
+
};
|
|
1803
|
+
});
|
|
1804
|
+
if (input.requireComplete) {
|
|
1805
|
+
const missing = findings.map((finding) => finding.finding_id).filter((findingId) => !seenFindingIds.has(findingId));
|
|
1806
|
+
if (missing.length > 0) throw new TypeError(`feedbackTrajectoryToOptimizerRow: missing independent decisions for finding ids: ${missing.join(", ")}`);
|
|
1807
|
+
if (completenessCount !== 1) throw new TypeError("feedbackTrajectoryToOptimizerRow: analyst run requires exactly one independent completeness_assessed decision");
|
|
1808
|
+
}
|
|
1809
|
+
return decisions;
|
|
1810
|
+
}
|
|
1811
|
+
function assertUniqueFindingIds(findingIds) {
|
|
1812
|
+
const seen = /* @__PURE__ */ new Set();
|
|
1813
|
+
for (const findingId of findingIds) {
|
|
1814
|
+
if (findingId.trim().length === 0) throw new TypeError("analyst finding id must not be empty");
|
|
1815
|
+
if (seen.has(findingId)) throw new TypeError(`analyst run contains duplicate finding id "${findingId}"`);
|
|
1816
|
+
seen.add(findingId);
|
|
1817
|
+
}
|
|
1818
|
+
}
|
|
1819
|
+
function snapshotAnalystFinding(value, context) {
|
|
1820
|
+
let snapshot;
|
|
1821
|
+
try {
|
|
1822
|
+
snapshot = JSON.parse(canonicalString(value));
|
|
1823
|
+
} catch (cause) {
|
|
1824
|
+
throw new TypeError(`${context} must have a canonical JSON representation`, { cause });
|
|
1825
|
+
}
|
|
1826
|
+
assertAnalystFinding(snapshot, context);
|
|
1827
|
+
return snapshot;
|
|
1828
|
+
}
|
|
1829
|
+
function assertAnalystFinding(value, context) {
|
|
1830
|
+
if (!isRecord$1(value)) throw new TypeError(`${context} must be an object`);
|
|
1831
|
+
assertOnlyKeys(value, [
|
|
1832
|
+
"schema_version",
|
|
1833
|
+
"finding_id",
|
|
1834
|
+
"analyst_id",
|
|
1835
|
+
"produced_at",
|
|
1836
|
+
"severity",
|
|
1837
|
+
"area",
|
|
1838
|
+
"claim",
|
|
1839
|
+
"rationale",
|
|
1840
|
+
"evidence_refs",
|
|
1841
|
+
"recommended_action",
|
|
1842
|
+
"validation_plan",
|
|
1843
|
+
"confidence",
|
|
1844
|
+
"subject",
|
|
1845
|
+
"derived_from_judge",
|
|
1846
|
+
"metadata"
|
|
1847
|
+
], context);
|
|
1848
|
+
if (value.schema_version !== "1.0.0") throw new TypeError(`${context} schema_version must be "1.0.0"`);
|
|
1849
|
+
requiredString(value.finding_id, `${context} finding_id`);
|
|
1850
|
+
requiredString(value.analyst_id, `${context} analyst_id`);
|
|
1851
|
+
canonicalTimestamp(value.produced_at, `${context} produced_at`);
|
|
1852
|
+
if (value.severity !== "critical" && value.severity !== "high" && value.severity !== "medium" && value.severity !== "low" && value.severity !== "info") throw new TypeError(`${context} severity is invalid`);
|
|
1853
|
+
requiredString(value.area, `${context} area`);
|
|
1854
|
+
requiredString(value.claim, `${context} claim`);
|
|
1855
|
+
optionalString$1(value.rationale, `${context} rationale`);
|
|
1856
|
+
value.evidence_refs = validateEvidenceRefs(value.evidence_refs, `${context} evidence_refs`);
|
|
1857
|
+
optionalString$1(value.recommended_action, `${context} recommended_action`);
|
|
1858
|
+
optionalString$1(value.validation_plan, `${context} validation_plan`);
|
|
1859
|
+
if (typeof value.confidence !== "number" || !Number.isFinite(value.confidence) || value.confidence < 0 || value.confidence > 1) throw new TypeError(`${context} confidence must be a finite number from 0 through 1`);
|
|
1860
|
+
optionalString$1(value.subject, `${context} subject`);
|
|
1861
|
+
if (value.derived_from_judge !== void 0 && typeof value.derived_from_judge !== "boolean") throw new TypeError(`${context} derived_from_judge must be a boolean`);
|
|
1862
|
+
if (value.metadata !== void 0 && !isRecord$1(value.metadata)) throw new TypeError(`${context} metadata must be an object`);
|
|
1863
|
+
}
|
|
1864
|
+
function validateMissedIssues(value, findingsById, context) {
|
|
1865
|
+
if (!Array.isArray(value)) throw new TypeError(`${context} missedIssues must be an array`);
|
|
1866
|
+
const seen = /* @__PURE__ */ new Set();
|
|
1867
|
+
return value.map((issue, index) => {
|
|
1868
|
+
const issueContext = `${context} missedIssues ${index}`;
|
|
1869
|
+
if (!isRecord$1(issue)) throw new TypeError(`${issueContext} must be an object`);
|
|
1870
|
+
assertOnlyKeys(issue, [
|
|
1871
|
+
"id",
|
|
1872
|
+
"reason",
|
|
1873
|
+
"evidence"
|
|
1874
|
+
], issueContext);
|
|
1875
|
+
const id = requiredString(issue.id, `${issueContext} id`);
|
|
1876
|
+
if (findingsById.has(id)) throw new TypeError(`${issueContext} id "${id}" is already an emitted finding id`);
|
|
1877
|
+
if (seen.has(id)) throw new TypeError(`duplicate missed issue id "${id}"`);
|
|
1878
|
+
seen.add(id);
|
|
1879
|
+
return {
|
|
1880
|
+
id,
|
|
1881
|
+
reason: requiredString(issue.reason, `${issueContext} reason`),
|
|
1882
|
+
...issue.evidence === void 0 ? {} : { evidence: validateEvidenceRefs(issue.evidence, `${issueContext} evidence`) }
|
|
1883
|
+
};
|
|
1884
|
+
});
|
|
1885
|
+
}
|
|
1886
|
+
function validateEvidenceRefs(value, context) {
|
|
1887
|
+
if (!Array.isArray(value)) throw new TypeError(`${context} must be an array`);
|
|
1888
|
+
return value.map((evidence, index) => {
|
|
1889
|
+
const evidenceContext = `${context} ${index}`;
|
|
1890
|
+
if (!isRecord$1(evidence)) throw new TypeError(`${evidenceContext} must be an object`);
|
|
1891
|
+
assertOnlyKeys(evidence, [
|
|
1892
|
+
"kind",
|
|
1893
|
+
"uri",
|
|
1894
|
+
"excerpt"
|
|
1895
|
+
], evidenceContext);
|
|
1896
|
+
if (evidence.kind !== "span" && evidence.kind !== "event" && evidence.kind !== "artifact" && evidence.kind !== "finding" && evidence.kind !== "metric") throw new TypeError(`${evidenceContext} kind is invalid`);
|
|
1897
|
+
const uri = requiredString(evidence.uri, `${evidenceContext} uri`);
|
|
1898
|
+
const excerpt = evidence.excerpt;
|
|
1899
|
+
optionalString$1(excerpt, `${evidenceContext} excerpt`);
|
|
1900
|
+
return {
|
|
1901
|
+
kind: evidence.kind,
|
|
1902
|
+
uri,
|
|
1903
|
+
...excerpt === void 0 ? {} : { excerpt }
|
|
1904
|
+
};
|
|
1905
|
+
});
|
|
1906
|
+
}
|
|
1907
|
+
function assertOnlyKeys(value, allowed, name) {
|
|
1908
|
+
const allowedKeys = new Set(allowed);
|
|
1909
|
+
const unexpected = Object.keys(value).filter((key) => !allowedKeys.has(key));
|
|
1910
|
+
if (unexpected.length > 0) throw new TypeError(`${name} contains unknown fields: ${unexpected.sort().join(", ")}`);
|
|
1911
|
+
}
|
|
1912
|
+
function stringArray(value, name) {
|
|
1913
|
+
if (!Array.isArray(value) || value.some((item) => typeof item !== "string")) throw new TypeError(`${name} must be an array of strings`);
|
|
1914
|
+
const strings = value.map((item) => requiredString(item, name));
|
|
1915
|
+
if (new Set(strings).size !== strings.length) throw new TypeError(`${name} must contain unique values`);
|
|
1916
|
+
return strings;
|
|
1917
|
+
}
|
|
1918
|
+
function requiredString(value, name) {
|
|
1919
|
+
if (typeof value !== "string" || value.trim().length === 0) throw new TypeError(`${name} must be a non-empty string`);
|
|
1920
|
+
return value;
|
|
1921
|
+
}
|
|
1922
|
+
function requiredDigest(value, name) {
|
|
1923
|
+
const digest = requiredString(value, name);
|
|
1924
|
+
if (!/^sha256:[a-f0-9]{64}$/.test(digest)) throw new TypeError(`${name} must be a sha256 digest`);
|
|
1925
|
+
return digest;
|
|
1926
|
+
}
|
|
1927
|
+
function optionalString$1(value, name) {
|
|
1928
|
+
if (value !== void 0 && typeof value !== "string") throw new TypeError(`${name} must be a string`);
|
|
1929
|
+
}
|
|
1930
|
+
function canonicalTimestamp(value, name) {
|
|
1931
|
+
const timestamp = requiredString(value, name);
|
|
1932
|
+
const parsed = new Date(timestamp);
|
|
1933
|
+
if (Number.isNaN(parsed.valueOf()) || parsed.toISOString() !== timestamp) throw new TypeError(`${name} must be a canonical ISO 8601 UTC timestamp`);
|
|
1934
|
+
return timestamp;
|
|
1935
|
+
}
|
|
1936
|
+
function isAnalystReviewSource(value) {
|
|
1937
|
+
return value === "user" || value === "judge" || value === "environment" || value === "metric" || value === "policy";
|
|
1938
|
+
}
|
|
1939
|
+
function isRecord$1(value) {
|
|
1940
|
+
return typeof value === "object" && value !== null && !Array.isArray(value);
|
|
1941
|
+
}
|
|
1942
|
+
//#endregion
|
|
1605
1943
|
//#region src/feedback-trajectory.ts
|
|
1606
1944
|
const DEFAULT_SPLIT_POLICY = {
|
|
1607
1945
|
trainPct: 70,
|
|
@@ -1735,6 +2073,115 @@ function createFeedbackTrajectory(input) {
|
|
|
1735
2073
|
metadata: input.metadata
|
|
1736
2074
|
};
|
|
1737
2075
|
}
|
|
2076
|
+
/** Preserve a trace-analysis run for review, whether or not review is complete. */
|
|
2077
|
+
function analystRunToFeedbackTrajectory(run, options) {
|
|
2078
|
+
const archivedRun = snapshotAnalystRun(run, `analyst run "${run.run_id}"`);
|
|
2079
|
+
const runDigest = analystRunDigest(archivedRun);
|
|
2080
|
+
const findings = archivedRun.findings;
|
|
2081
|
+
assertUniqueFindingIds(findings.map((finding) => finding.finding_id));
|
|
2082
|
+
const analystIds = [.../* @__PURE__ */ new Set([...findings.map((finding) => finding.analyst_id), ...archivedRun.per_analyst.map((summary) => summary.analyst_id)])];
|
|
2083
|
+
const reviewDecisions = validateAnalystReviewDecisions({
|
|
2084
|
+
runId: archivedRun.run_id,
|
|
2085
|
+
runDigest,
|
|
2086
|
+
findings,
|
|
2087
|
+
analystIds,
|
|
2088
|
+
decisions: options.reviewDecisions ?? []
|
|
2089
|
+
});
|
|
2090
|
+
const reviewRequests = options.reviewRequests === void 0 ? void 0 : validateAnalystReviewRequests(archivedRun, options.reviewRequests);
|
|
2091
|
+
const createdAt = options.createdAt ?? archivedRun.started_at;
|
|
2092
|
+
const knownCostUsd = archivedRun.total_cost_usd;
|
|
2093
|
+
const capturedCost = archivedRun.total_cost_provenance?.kind === "uncaptured" ? void 0 : archivedRun.total_cost_usd;
|
|
2094
|
+
return createFeedbackTrajectory({
|
|
2095
|
+
id: options.id,
|
|
2096
|
+
projectId: options.projectId,
|
|
2097
|
+
scenarioId: options.scenarioId,
|
|
2098
|
+
task: options.task,
|
|
2099
|
+
attempts: [{
|
|
2100
|
+
id: `analysis:${archivedRun.run_id}`,
|
|
2101
|
+
stepIndex: 0,
|
|
2102
|
+
artifactType: "research",
|
|
2103
|
+
artifact: {
|
|
2104
|
+
type: "analyst-run",
|
|
2105
|
+
analystRunId: archivedRun.run_id,
|
|
2106
|
+
runDigest,
|
|
2107
|
+
correlationId: archivedRun.correlation_id,
|
|
2108
|
+
analystIds,
|
|
2109
|
+
findings
|
|
2110
|
+
},
|
|
2111
|
+
createdAt,
|
|
2112
|
+
metadata: {
|
|
2113
|
+
perAnalyst: archivedRun.per_analyst,
|
|
2114
|
+
trace: options.trace,
|
|
2115
|
+
reviewRequests
|
|
2116
|
+
}
|
|
2117
|
+
}],
|
|
2118
|
+
labels: [...options.labels ?? []],
|
|
2119
|
+
outcome: options.outcome ? {
|
|
2120
|
+
...options.outcome,
|
|
2121
|
+
costUsd: options.outcome.costUsd ?? capturedCost
|
|
2122
|
+
} : capturedCost === void 0 ? void 0 : {
|
|
2123
|
+
costUsd: capturedCost,
|
|
2124
|
+
observedAt: archivedRun.ended_at
|
|
2125
|
+
},
|
|
2126
|
+
split: options.split,
|
|
2127
|
+
tags: options.tags,
|
|
2128
|
+
createdAt,
|
|
2129
|
+
metadata: {
|
|
2130
|
+
...options.metadata,
|
|
2131
|
+
analysis: {
|
|
2132
|
+
kind: "analyst-run",
|
|
2133
|
+
runId: archivedRun.run_id,
|
|
2134
|
+
runDigest,
|
|
2135
|
+
startedAt: archivedRun.started_at,
|
|
2136
|
+
endedAt: archivedRun.ended_at,
|
|
2137
|
+
reviewDecisions,
|
|
2138
|
+
costProvenance: archivedRun.total_cost_provenance,
|
|
2139
|
+
knownCostUsd
|
|
2140
|
+
}
|
|
2141
|
+
}
|
|
2142
|
+
});
|
|
2143
|
+
}
|
|
2144
|
+
/** Convert one immutable analyst run into revision requests for an independent reviewer. */
|
|
2145
|
+
function analystRunToReviewRequests(run, options = {}) {
|
|
2146
|
+
const archivedRun = snapshotAnalystRun(run, `analyst run "${run.run_id}"`);
|
|
2147
|
+
const runDigest = analystRunDigest(archivedRun);
|
|
2148
|
+
const createdAt = options.createdAt ?? (/* @__PURE__ */ new Date()).toISOString();
|
|
2149
|
+
return archivedRun.findings.map((finding) => ({
|
|
2150
|
+
id: analystReviewRequestId(runDigest, finding.finding_id),
|
|
2151
|
+
runId: archivedRun.run_id,
|
|
2152
|
+
runDigest,
|
|
2153
|
+
findingId: finding.finding_id,
|
|
2154
|
+
findingDigest: analystFindingDigest(finding),
|
|
2155
|
+
analystId: finding.analyst_id,
|
|
2156
|
+
area: finding.area,
|
|
2157
|
+
claim: finding.claim,
|
|
2158
|
+
evidence: finding.evidence_refs,
|
|
2159
|
+
recommendedAction: finding.recommended_action,
|
|
2160
|
+
validationPlan: finding.validation_plan,
|
|
2161
|
+
severity: feedbackSeverityFromFinding(finding),
|
|
2162
|
+
confidence: finding.confidence,
|
|
2163
|
+
createdAt
|
|
2164
|
+
}));
|
|
2165
|
+
}
|
|
2166
|
+
function validateAnalystReviewRequests(run, requests) {
|
|
2167
|
+
const runDigest = analystRunDigest(run);
|
|
2168
|
+
const findingsById = new Map(run.findings.map((finding) => [finding.finding_id, finding]));
|
|
2169
|
+
const seenFindingIds = /* @__PURE__ */ new Set();
|
|
2170
|
+
return requests.map((request, index) => {
|
|
2171
|
+
if (request.runId !== run.run_id) throw new TypeError(`analyst review request ${index} run id mismatch`);
|
|
2172
|
+
if (request.runDigest !== runDigest) throw new TypeError(`analyst review request ${index} run digest mismatch`);
|
|
2173
|
+
const finding = findingsById.get(request.findingId);
|
|
2174
|
+
if (!finding) throw new TypeError(`analyst review request ${index} references unknown finding id "${request.findingId}"`);
|
|
2175
|
+
if (seenFindingIds.has(request.findingId)) throw new TypeError(`duplicate analyst review request for finding id "${request.findingId}"`);
|
|
2176
|
+
seenFindingIds.add(request.findingId);
|
|
2177
|
+
if (request.findingDigest !== analystFindingDigest(finding)) throw new TypeError(`analyst review request ${index} finding digest mismatch`);
|
|
2178
|
+
if (request.id !== analystReviewRequestId(runDigest, request.findingId)) throw new TypeError(`analyst review request ${index} id mismatch`);
|
|
2179
|
+
return request;
|
|
2180
|
+
});
|
|
2181
|
+
}
|
|
2182
|
+
function analystReviewRequestId(runDigest, findingId) {
|
|
2183
|
+
return `analyst:${runDigest}:${findingId}`;
|
|
2184
|
+
}
|
|
1738
2185
|
function assignFeedbackSplit(trajectory, policy = {}) {
|
|
1739
2186
|
const split = {
|
|
1740
2187
|
...DEFAULT_SPLIT_POLICY,
|
|
@@ -1772,18 +2219,37 @@ function feedbackTrajectoriesToDatasetScenarios(trajectories) {
|
|
|
1772
2219
|
}
|
|
1773
2220
|
function feedbackTrajectoryToOptimizerRow(trajectory) {
|
|
1774
2221
|
const labels = allLabels(trajectory);
|
|
2222
|
+
const analystReview = readAnalystReview(trajectory);
|
|
2223
|
+
const analystQuality = analystReview ? completedAnalystReviewQuality(analystReview) : void 0;
|
|
2224
|
+
const reviewLabelKinds = analystReview ? analystReview.reviewDecisions.map((decision) => {
|
|
2225
|
+
if (decision.verdict === "rejected") return "reject";
|
|
2226
|
+
if (decision.verdict === "completeness_assessed" && decision.missedIssues.length > 0) return "revision_request";
|
|
2227
|
+
return "approve";
|
|
2228
|
+
}) : [];
|
|
1775
2229
|
return {
|
|
1776
2230
|
scenarioId: trajectory.scenarioId ?? trajectory.id,
|
|
1777
2231
|
trajectoryId: trajectory.id,
|
|
1778
|
-
labelKinds: [
|
|
1779
|
-
score: trajectory.outcome?.score ?? scoreFromLabels(labels),
|
|
2232
|
+
labelKinds: [.../* @__PURE__ */ new Set([...labels.map((label) => label.kind), ...reviewLabelKinds])],
|
|
2233
|
+
score: analystQuality?.f1 ?? trajectory.outcome?.score ?? scoreFromLabels(labels),
|
|
1780
2234
|
metadata: {
|
|
1781
2235
|
projectId: trajectory.projectId,
|
|
1782
2236
|
split: trajectory.split,
|
|
1783
2237
|
intent: trajectory.task.intent,
|
|
1784
2238
|
attempts: trajectory.attempts.length,
|
|
1785
2239
|
outcome: trajectory.outcome,
|
|
1786
|
-
labels
|
|
2240
|
+
labels,
|
|
2241
|
+
...analystReview ? { analystReview: {
|
|
2242
|
+
findingIds: analystReview.findingIds,
|
|
2243
|
+
findingDigests: analystReview.findings.map((finding) => ({
|
|
2244
|
+
findingId: finding.finding_id,
|
|
2245
|
+
findingDigest: analystFindingDigest(finding)
|
|
2246
|
+
})),
|
|
2247
|
+
decisions: analystReview.reviewDecisions,
|
|
2248
|
+
precision: analystQuality?.precision,
|
|
2249
|
+
recall: analystQuality?.recall,
|
|
2250
|
+
f1: analystQuality?.f1,
|
|
2251
|
+
counts: analystQuality?.counts
|
|
2252
|
+
} } : {}
|
|
1787
2253
|
}
|
|
1788
2254
|
};
|
|
1789
2255
|
}
|
|
@@ -1923,6 +2389,14 @@ function allLabels(trajectory) {
|
|
|
1923
2389
|
return true;
|
|
1924
2390
|
});
|
|
1925
2391
|
}
|
|
2392
|
+
function feedbackSeverityFromFinding(finding) {
|
|
2393
|
+
switch (finding.severity) {
|
|
2394
|
+
case "critical": return "critical";
|
|
2395
|
+
case "high": return "error";
|
|
2396
|
+
case "medium": return "warning";
|
|
2397
|
+
default: return "info";
|
|
2398
|
+
}
|
|
2399
|
+
}
|
|
1926
2400
|
function scoreFromLabels(labels) {
|
|
1927
2401
|
if (!labels.length) return void 0;
|
|
1928
2402
|
const scored = labels.map((label) => {
|
|
@@ -12372,6 +12846,6 @@ function assertProductBenchmarkRun(runDir) {
|
|
|
12372
12846
|
return report;
|
|
12373
12847
|
}
|
|
12374
12848
|
//#endregion
|
|
12375
|
-
export { AGENT_PROFILE_KINDS, ATIF_SCHEMA_VERSION, ATTESTATION_ALGORITHM, AgentDriver, AgentEvalError, AgentProfileCellValidationError, AnalystRegistry, AxGepaSteeringOptimizer, BENCHMARK_SPLIT_SEED, BOOTSTRAP_GATE_MIN_N, BackendIntegrityError, BenchmarkRunner, BudgetBreachError, BudgetGuard, CODING_HARNESSES, CallbackResearcher, CaptureIntegrityError, ConfigError, ConvergenceTracker, CostAccountingIncompleteError, CostCallConflictError, CostCeilingReachedError, CostLedger, CostLedgerPersistenceError, CostReceiptCaptureError, CostReservationExceededError, CostTracker, CrossFamilyError, DECISION_PAIRED_DELTA_STATISTIC, DEFAULT_AGENT_SLOS, DEFAULT_COMPLEXITY_WEIGHTS, DEFAULT_RULES as DEFAULT_FAILURE_RULES, DEFAULT_FINDERS, DEFAULT_HARNESS_OBJECTIVES, DEFAULT_MUTATION_PRIMITIVES, DEFAULT_MUTATORS, DEFAULT_PERMUTATIONS, DEFAULT_PR_REVIEW_SCORE_WEIGHTS, DEFAULT_REDACTION_RULES, DEFAULT_RED_TEAM_CORPUS, DEFAULT_RUN_SCORE_WEIGHTS, DEFAULT_SEVERITY_WEIGHTS, DEFAULT_TRACE_ANALYST_BUDGETS, DEFAULT_TRACE_ANALYST_KINDS, Dataset, DescriptionLengthGate, DockerSandboxDriver, DualAgentBench, ERROR_COUNT_PATTERNS, EvalTraceStore, ExperimentTracker, FAILURE_CLASSES, FAILURE_MODE_KIND_SPEC, FileSystemFeedbackTrajectoryStore, FileSystemRawProviderSink, FileSystemTraceStore, FindingsStore, HARBOR_IMPORT_GAP, HARNESS_NATIVE_MODEL, HeldOutGate, HoldoutAuditor, HoldoutLockedError, IMPROVEMENT_KIND_SPEC, INPUT_VALUE, INTENT_MATCH_JUDGE_VERSION, InMemoryFeedbackTrajectoryStore, InMemoryRawProviderSink, InMemoryTraceStore, InMemoryWorkspaceInspector, JudgeError, JudgeParseError, JudgeRunner, KNOWLEDGE_GAP_KIND_SPEC, KNOWLEDGE_POISONING_KIND_SPEC, LLM_CACHED_TOKENS, LLM_CACHED_TOKEN_ATTR_KEYS, LLM_CACHE_WRITE_TOKENS, LLM_CACHE_WRITE_TOKEN_ATTR_KEYS, LLM_CONTEXT_TOKENS, LLM_COST_ATTR_KEYS, LLM_COST_USD, LLM_INPUT_TOKENS, LLM_INPUT_TOKEN_ATTR_KEYS, LLM_MODEL_ATTR_KEYS, LLM_MODEL_NAME, LLM_OUTPUT_TOKENS, LLM_OUTPUT_TOKEN_ATTR_KEYS, LLM_REASONING_TOKENS, LLM_REASONING_TOKEN_ATTR_KEYS, LlmCallError, LlmClient, LlmResponseError, LlmRouteAssertionError, LockedJsonlAppender, MANN_WHITNEY_EXACT_MAX_STATES, MANN_WHITNEY_EXACT_MAX_WORK, MODEL_PRICING, MetricsCollector, ModelsUnreachableError, MultiLayerVerifier, Mutex, NoopRawProviderSink, NoopResearcher, NotFoundError, OPENINFERENCE_SPAN_KIND, OTEL_AGENT_EVAL_SCOPE, OUTPUT_VALUE, OtlpFileTraceStore, PairwiseSteeringOptimizer, ProductClient, PromptRegistry, REDACTION_VERSION, REFERENCE_EQUIVALENCE_INPUT_LIMITS, REFERENCE_EQUIVALENCE_JUDGE_VERSION, RESEARCH_REPORT_HARD_PAIR_FLOOR, ROLLOUT_SCHEMA, RUN_COST_ATTR_KEYS, ReplayCache, ReplayCacheMissError, ReplayError, RunCritic, RunIntegrityError, RunRecordValidationError, SEMANTIC_CONCEPT_JUDGE_VERSION, SKILL_USAGE_ANALYST, SPAN_KIND_ATTR_KEYS, SUPERVISOR_RUN_SCHEMA, SandboxHarness, ScenarioRegistry, SeatUnsetError, SingleBackendError, SkillUsageAnalyst, SpanNotFoundError, SubprocessSandboxDriver, TOOL_ARGS_CAPTURED, TOOL_LATENCY_MS, TOOL_NAME, TOOL_NAME_ATTR_KEYS, TRACE_ANALYST_ACTOR_DESCRIPTION, TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION, TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, TRACE_SCHEMA_VERSION, TokenCounter, ToolTraceMissingError, TraceContractBuilder, TraceEmitter, TraceFileMissingError, TraceNotFoundError, UI_FINDING_SEVERITIES, UI_LENSES, UNIVERSAL_FINDERS, ValidationError, VerificationError, WILCOXON_EXACT_MAX_N, acquisitionPlansForKnowledgeGaps, agentProfileCellHashMaterial, agentProfileCellKey, agentProfileHash, agentProfileId, agentProfileModelId, agentVisibleFields, aggregateJudgeVerdicts, aggregateLlm, aggregatePrReviewScore, aggregateRunScore, allCriticalPassed, analyzeAntiSlop, analyzeSeries, analyzeSupervisorRun, analyzeSupervisorRunSources, analyzeTraces, appendScorecard, applyLlmSpanOtlpAttributes, applyToolSpanOtlpAttributes, argHash, asNumber, asString, assertCapabilityHeadroom, assertCrossFamily, assertLlmRoute, assertMinted, assertMintedLines, assertModelsServed, assertNoHiddenLeak, assertProductBenchmarkRun, assertRealAgentReceipts, assertRealBackend, assertReleaseConfidence, assertRolloutLine, assertRunAgentProfileCell, assertRunCaptured, assertSingleBackend, assignFeedbackSplit, assignHeldOutTag, attachCostToReport, attest, attributeCounterfactuals, backoffMs, deterministicSplit as benchmarkDeterministicSplit, benchmarks_exports as benchmarks, benjaminiHochberg, bisect, blendHeldout, blockingKnowledgeEval, bonferroni, bootstrapCi, buildAgentInterfaceProfileCell, buildAgentProfileCell, buildDefaultAnalystRegistry, buildDriverSystemPrompt, buildProductBenchmarkManifest, buildReflectionPrompt, buildReviewerPrompt, buildTraceAnalystTools, buildTraceInsightContext, buildTraceInsightPrompt, buildTrajectory, buildWorkerDriverSystemPrompt, byteLengthRange, cachedJudge, calibrateJudge, calibrateJudgeContinuous, callLlm, callLlmJson, canaryLeakView, canonicalJson, canonicalize, capabilityHeadroom, captureFetchToRawSink, causalAttribution, checkBehavioralCanary, checkCanaries, checkSlos, checkTraceContracts, clamp01, classifyFailure, classifyOtlpSpanRole, classifyTreatment, claudeCodeSupervisorRunReader, cliffsDelta, clusteredPairedBinary, cohensD, collectionPreserved, commentsForSource, commitBisect, comparePairedArms, compareReferenceReplay, compareToBaseline, compilerJudge, completionVerdict, composeParsers, composeValidators, computeExperimentStats, computeFindingId, computeToolUseMetrics, computeTraceMetrics, confidenceInterval, containsAll, contentHash, contextInputTokens, continuousAgreement, contractJudge, controlFailureClassFromVerification, controlRunToFeedbackTrajectory, controlRunToRunRecord, convertTraceStoresToOtlp, corpusInterRaterAgreement, corpusInterRaterAgreementFromJudgeScores, costForTokenPricing, costForUsage, costReceiptFromLlm, costReceiptFromLlmError, costReport, createAnalystAi, createAntiSlopJudge, createChatClient, createDefaultReviewer, createFeedbackTrajectory, createIntentMatchJudge, createLlmCorrectnessChecker, createLlmReviewer, createOtelExporter, createOtelTracingStore, createReferenceEquivalenceJudge, createReplayFetch, createSandboxPool, createSemanticConceptJudge, createTokenRecallChecker, createTraceAnalystKind, crossTraceDiff, crowdingDistance, dataDescriptionBits, decideNextUserTurn, decidePairedPromotion, decideReferenceReplayPromotion, decideReferenceReplayRunPromotion, defaultBlendWeights, defaultIsMaterial, defaultProviderRedactor, defaultReferenceReplayMatcher, defaultTraceInsightPanel, deployGateLayer, describeTraceInsightScope, diffFindings, diffScorecard, discoverPersonas, distillPlaybook, domainEvidencePattern, dominates, eProcess, ensembleJudge, errorStreakDetector, estimateCost, estimateTokens, evaluateActionPolicy, evaluateContract, evaluateHypothesis, evaluateInterimReleaseConfidence, evaluateOracles, evaluateReleaseConfidence, evaluateTraceContract, executeScenario, expandProfileAxes, expectAgent, exportProductBenchmark, exportProductBenchmarkRuns, exportRewardModel, exportRunAsOtlp, extractAssetUrls, extractErrorCount, extractOtlpAttributes, extractProducedState, extractUsage, extractUsageFromResponse, extractUsageFromSse, feedbackTrajectoriesToDatasetScenarios, feedbackTrajectoriesToOptimizerRows, feedbackTrajectoryToDatasetScenario, feedbackTrajectoryToOptimizerRow, fileContains, fileExists, fileExperimentStore, fileVerdictCache, findAutoMatchNoExpectation, findConstructorCwdDropped, findFallbackToPass, findLiteralTruePass, findProductBenchmarkArtifacts, findSkipCountsAsPass, firstNumberAttr, firstStringAttr, flattenOtlpExportToNdjson, flowLayer, fnv1a32, formatBenchmarkReport, formatDriverReport, formatFindings, formatScorecardDiff, fromHarborTrajectory, gainHistogram, gateTreatmentApplied, gateTreatmentFromMetrics, gateTreatmentFromSpans, gateTreatmentFromToolSpans, ghCliClient, gitProvenanceReader, precision as goldenPrecision, gradeOnHidden, gradeSemanticStatus, groupBy, groupRunsByAgentProfileCell, harnessAxisOf, hasCapturedToolArgs, hashContent, hashJson, hashScenarios, hashToUnit, hiddenGrade, holm, htmlContainsElement, httpGithubClient, improvementVerdict, inMemoryExperimentStore, inMemoryReferenceReplayStore, inMemoryReviewStore, inMemoryRunRecordBackend, inMemoryVerdictCache, inferDomainKeywords, inferOtlpKind, interRaterReliability, interpretCliffs, iqr, isBinaryOutcomeVector, isHiddenDestination, isJudgeSpan, isLlmSpan, isModelPriced, isOtelConfigured, isOtlpModelCall, isRealnessGated, isRetrievalSpan, isRolloutLine, isRunRecord, isSandboxSpan, isToolSpan, isTrainableSplit, isTransientLlmError, isUnavailable, iterateRawCalls, jestTestParser, jsonHasKeys, jsonShape, jsonlReferenceReplayStore, jsonlReviewStore, jsonlRunRecordBackend, judgeFamily, judgeReplayGate, judgeSpans, keyPreserved, knowledgeReadinessTracePayload, leaderboard, linterJudge, llmJudge, llmSpanFromProvider, llmSpans, loadScorecard, loadScorerFromGrader, localCommandRunner, lowercaseMutator, makeEvalTools, makeFinding, makeProposalFinding, mannWhitneyU, mapConcurrent, matchGoldens, matchSpan, maximumChargeForLlmRequest, mcnemar, mcnemarPower, mcnemarRequiredN, mergeLayerResults, mergeSteeringBundle, minimumPairsForPairedDeltaTest, mintRolloutRows, modelDescriptionBits, modelHasSnapshot, modelPriceKey, mulberry32, multiToolchainLayer, noProgressDetector, normalCdf, normalizeScores, notBlocked, objectiveEval, observeAll, observedScore, observedSplitScore, otelRunCompleteHook, otlpRowsToRunRecords, otlpRowsToTraceRunRecords, otlpToRunRecords, otlpToTraceRunRecords, pairArms, pairRunRecords, pairedBinaryScale, pairedBootstrap, pairedCohensDz, pairedDecisionShape, pairedDeltaTest, pairedDeltaTieFraction, pairedEvalueSequence, pairedMde, pairedRiskDifference, pairedRiskDifferenceExact, pairedRiskDifferenceScore, pairedSignTest, pairedTTest, paraphraseRobustness, paraphraseRobustnessScenarios, paretoChart, paretoFrontier, paretoFrontierWithCrowding, parseCorrectnessResponse, parseFeedbackTrajectoriesJsonl, parseReflectionResponse, parseRunRecordSafe, parseRuntimeTrajectoryHookEvent, partialCredit, partitionHeldOut, passAtK, passOrthogonality, pearsonR, pixelDeltaRatio, planTraceInsightQuestions, politenessPrefixMutator, positionalBias, preflightModels, printDriverSummary, probeLlm, productBenchmarkIntegrityFailures, productBenchmarkMutableSurfaces, productBenchmarkRepoIdentity, productBenchmarkSplits, profile_exports as profile, projectOtlpFlatLine, projectRuntimeTrajectoryEvidence, promptBisect, proposeSynthesisTargets, providerFromBaseUrl, pytestTestParser, ranks, readClaudeCodeSupervisorRun, readOtlpStatus, readProductBenchmarkManifest, readProductBenchmarkRecords, recordRuns, recordRunsToScorecard, redTeamDataset, redTeamReport, redactString, redactValue, referenceReplayRunsToSteeringRows, referenceReplayScenarioToRunScore, regexMatch, regexMatches, relabelImportedSplit, renderMarkdownReport, renderPlaybookMarkdown, renderPreferenceMemoryMarkdown, renderPriorFindings, renderReleaseReport, renderSteeringText, renderSupervisorRunHeadline, renderSupervisorRunMarkdown, renderUpstreamFindings, repeatedActionDetector, replayFeedbackTrajectories, replayFeedbackTrajectory, replayScorerOverCorpus, replayTraceThroughJudge, requireAgentProfileCell, requiredPairedSampleSize, requiredSampleSize, researchReport, resolveModelPricing, resolveSeat, rollupSupervisorRuns, roundTripRunRecord, routeFields, rowCount, rowWhere, runAgentControlLoop, runAssertions, runBehavioralCanaries, runCanaries, runCounterfactual, runE2EWorkflow, runEvalCampaign, runExpectations, runFailureClass, runHarnessExperiment, runIntentMatchJudge, runJudgeFleet, runKeywordCoverageJudge, runKeywordCoverageJudgeUrl, runLiveProof, runProposeReview, runProposeReviewAsControlLoop, runRecordToProductBenchmarkRecord, runReferenceEquivalenceJudge, runReferenceReplay, runScore, runSelfPlay, runSemanticConceptJudge, runTaskScore, runTestGradedScenario, runsForScenario, scalarScore, scanForMuffledGates, scoreContinuity, scoreFromEvals, scoreKnowledgeReadiness, scoreOrigin, scorePrReviewComments, scorePrReviewSource, scoreRedTeamOutput, scoreReferenceReplay, scoreTraceInsightReadiness, seatPresets, securityJudge, selectHarnessVariant, selfPreference, sentenceReorderMutator, serializeFeedbackTrajectoriesJsonl, showMeasured, signManifest, spearmanR, statusAdvanced, stopOnNoProgress, stopOnRepeatedAction, stringField, stripFencedJson, studentTCdf, subjectiveEval, summarizeAgentReceiptIntegrity, summarizeBackendIntegrity, summarizeHarnessResults, summarizePrReviewBenchmark, summarizePreferenceMemory, summaryTable, supervisorRunRolloutLines, testJudge, textInSnapshot, throwIfRunIncomplete, toAgentProfileJson, toHarborTrajectories, toHarborTrajectory, toJsonl, toLangfuseEnvelope, toOpenAiTool, toPrometheusText, toRewardRows, toSftRows, tokenizeDomainWords, toolNamesForRun, toolSpans, toolSpansToTraceAnalysisStore, traceAnalystFunctionGroup, traceAnalystOnRunComplete, traceContract, traceJudge, traceJudgeEnsemble, traceSpanKindToOpenInferenceKind, tracedAnalyzeTraces, trainingReward, trainingScore, typoMutator, unmintableReasons, urlContains, userQuestionsForKnowledgeGaps, validateAgentProfileCell, validateProductBenchmarkManifest, validateProductBenchmarkRecord, validateProductBenchmarkRun, validateRolloutLine, validateRunRecord, verbosityBias, verifyAgentProfileCell, verifyAttestation, verifyCompletion, verifyManifest, visualDiff, viteDeployRunner, vitestTestParser, weightedComposite, weightedMean, weightedRecall, welchsTTest, whitespaceCollapseMutator, wilcoxonSignedRank, wilson, withAssignedFeedbackSplit, withHeldoutBlend, withJudgeRetry, withOtelPipeline, wranglerDeployRunner, writeSupervisorRunReport };
|
|
12849
|
+
export { AGENT_PROFILE_KINDS, ATIF_SCHEMA_VERSION, ATTESTATION_ALGORITHM, AgentDriver, AgentEvalError, AgentProfileCellValidationError, AnalystRegistry, AxGepaSteeringOptimizer, BENCHMARK_SPLIT_SEED, BOOTSTRAP_GATE_MIN_N, BackendIntegrityError, BenchmarkRunner, BudgetBreachError, BudgetGuard, CODING_HARNESSES, CONTROL_INTEGRITY_ANALYST, CallbackResearcher, CaptureIntegrityError, ConfigError, ControlIntegrityAnalyst, ConvergenceTracker, CostAccountingIncompleteError, CostCallConflictError, CostCeilingReachedError, CostLedger, CostLedgerPersistenceError, CostReceiptCaptureError, CostReservationExceededError, CostTracker, CrossFamilyError, DECISION_PAIRED_DELTA_STATISTIC, DEFAULT_AGENT_SLOS, DEFAULT_COMPLEXITY_WEIGHTS, DEFAULT_RULES as DEFAULT_FAILURE_RULES, DEFAULT_FINDERS, DEFAULT_HARNESS_OBJECTIVES, DEFAULT_MUTATION_PRIMITIVES, DEFAULT_MUTATORS, DEFAULT_PERMUTATIONS, DEFAULT_PR_REVIEW_SCORE_WEIGHTS, DEFAULT_REDACTION_RULES, DEFAULT_RED_TEAM_CORPUS, DEFAULT_RUN_SCORE_WEIGHTS, DEFAULT_SEVERITY_WEIGHTS, DEFAULT_TRACE_ANALYST_BUDGETS, DEFAULT_TRACE_ANALYST_KINDS, Dataset, DescriptionLengthGate, DockerSandboxDriver, DualAgentBench, ERROR_COUNT_PATTERNS, EvalTraceStore, ExperimentTracker, FAILURE_CLASSES, FAILURE_MODE_KIND_SPEC, FileSystemFeedbackTrajectoryStore, FileSystemRawProviderSink, FileSystemTraceStore, FindingsStore, HARBOR_IMPORT_GAP, HARNESS_NATIVE_MODEL, HeldOutGate, HoldoutAuditor, HoldoutLockedError, IMPROVEMENT_KIND_SPEC, INPUT_VALUE, INTENT_MATCH_JUDGE_VERSION, InMemoryFeedbackTrajectoryStore, InMemoryRawProviderSink, InMemoryTraceStore, InMemoryWorkspaceInspector, JudgeError, JudgeParseError, JudgeRunner, KNOWLEDGE_GAP_KIND_SPEC, KNOWLEDGE_POISONING_KIND_SPEC, LLM_CACHED_TOKENS, LLM_CACHED_TOKEN_ATTR_KEYS, LLM_CACHE_WRITE_TOKENS, LLM_CACHE_WRITE_TOKEN_ATTR_KEYS, LLM_CONTEXT_TOKENS, LLM_COST_ATTR_KEYS, LLM_COST_USD, LLM_INPUT_TOKENS, LLM_INPUT_TOKEN_ATTR_KEYS, LLM_MODEL_ATTR_KEYS, LLM_MODEL_NAME, LLM_OUTPUT_TOKENS, LLM_OUTPUT_TOKEN_ATTR_KEYS, LLM_REASONING_TOKENS, LLM_REASONING_TOKEN_ATTR_KEYS, LimitExceededError, LlmCallError, LlmClient, LlmResponseError, LlmRouteAssertionError, LockedJsonlAppender, MANN_WHITNEY_EXACT_MAX_STATES, MANN_WHITNEY_EXACT_MAX_WORK, MODEL_PRICING, MetricsCollector, ModelsUnreachableError, MultiLayerVerifier, Mutex, NoopRawProviderSink, NoopResearcher, NotFoundError, OPENINFERENCE_SPAN_KIND, OTEL_AGENT_EVAL_SCOPE, OUTPUT_VALUE, OtlpFileTraceStore, PairwiseSteeringOptimizer, ProductClient, PromptRegistry, REDACTION_VERSION, REFERENCE_EQUIVALENCE_INPUT_LIMITS, REFERENCE_EQUIVALENCE_JUDGE_VERSION, RESEARCH_REPORT_HARD_PAIR_FLOOR, ROLLOUT_SCHEMA, RUN_COST_ATTR_KEYS, ReplayCache, ReplayCacheMissError, ReplayError, RunCritic, RunIntegrityError, RunRecordValidationError, SEMANTIC_CONCEPT_JUDGE_VERSION, SKILL_USAGE_ANALYST, SPAN_KIND_ATTR_KEYS, SUPERVISOR_RUN_INTEGRITY_SCHEMA, SUPERVISOR_RUN_SCHEMA, SandboxHarness, ScenarioRegistry, SeatUnsetError, SingleBackendError, SkillUsageAnalyst, SpanNotFoundError, SubprocessSandboxDriver, TOOL_ARGS_CAPTURED, TOOL_LATENCY_MS, TOOL_NAME, TOOL_NAME_ATTR_KEYS, TRACE_ANALYSIS_LIMITS, TRACE_ANALYST_ACTOR_DESCRIPTION, TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION, TRACE_ANALYST_TOOL_NAMESPACE, TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, TRACE_SCHEMA_VERSION, TokenCounter, ToolTraceMissingError, TraceAnalysisLimitError, TraceAnalysisStoreContractError, TraceAnalysisValidationError, TraceContractBuilder, TraceEmitter, TraceFileMalformedError, TraceFileMissingError, TraceFileTooLargeError, TraceNotFoundError, UI_FINDING_SEVERITIES, UI_LENSES, UNIVERSAL_FINDERS, ValidationError, VerificationError, WILCOXON_EXACT_MAX_N, acquisitionPlansForKnowledgeGaps, agentProfileCellHashMaterial, agentProfileCellKey, agentProfileHash, agentProfileId, agentProfileModelId, agentVisibleFields, aggregateJudgeVerdicts, aggregateLlm, aggregatePrReviewScore, aggregateRunScore, allCriticalPassed, analystFindingDigest, analystRunDigest, analystRunToFeedbackTrajectory, analystRunToReviewRequests, analyzeAntiSlop, analyzeSeries, analyzeSupervisorRun, analyzeSupervisorRunIntegrity, analyzeSupervisorRunSources, analyzeTraces, appendScorecard, applyLlmSpanOtlpAttributes, applyToolSpanOtlpAttributes, argHash, asNumber, asString, assertCapabilityHeadroom, assertCrossFamily, assertLlmRoute, assertMinted, assertMintedLines, assertModelsServed, assertNoHiddenLeak, assertProductBenchmarkRun, assertRealAgentReceipts, assertRealBackend, assertReleaseConfidence, assertRolloutLine, assertRunAgentProfileCell, assertRunCaptured, assertSingleBackend, assignFeedbackSplit, assignHeldOutTag, attachCostToReport, attest, attributeCounterfactuals, backoffMs, deterministicSplit as benchmarkDeterministicSplit, benchmarks_exports as benchmarks, benjaminiHochberg, bisect, blendHeldout, blockingKnowledgeEval, bonferroni, bootstrapCi, buildAgentInterfaceProfileCell, buildAgentProfileCell, buildDefaultAnalystRegistry, buildDriverSystemPrompt, buildProductBenchmarkManifest, buildReflectionPrompt, buildReviewerPrompt, buildTraceAnalysisToolDescriptors, buildTraceAnalystTools, buildTraceInsightContext, buildTraceInsightPrompt, buildTrajectory, buildWorkerDriverSystemPrompt, byteLengthRange, cachedJudge, calibrateJudge, calibrateJudgeContinuous, callLlm, callLlmJson, canaryLeakView, canonicalJson, canonicalize, capabilityHeadroom, captureFetchToRawSink, causalAttribution, checkBehavioralCanary, checkCanaries, checkSlos, checkTraceContracts, clamp01, classifyFailure, classifyOtlpSpanRole, classifyTreatment, claudeCodeSupervisorRunReader, cliffsDelta, clusteredPairedBinary, cohensD, collectionPreserved, commentsForSource, commitBisect, comparePairedArms, compareReferenceReplay, compareToBaseline, compilerJudge, completionVerdict, composeParsers, composeValidators, computeExperimentStats, computeFindingId, computeToolUseMetrics, computeTraceMetrics, confidenceInterval, containsAll, contentHash, contextInputTokens, continuousAgreement, contractJudge, controlFailureClassFromVerification, controlRunToFeedbackTrajectory, controlRunToRunRecord, convertTraceStoresToOtlp, corpusInterRaterAgreement, corpusInterRaterAgreementFromJudgeScores, costForTokenPricing, costForUsage, costReceiptFromLlm, costReceiptFromLlmError, costReport, createAnalystAi, createAntiSlopJudge, createBoundedTraceAnalysisStore, createChatClient, createDefaultReviewer, createFeedbackTrajectory, createIntentMatchJudge, createLlmCorrectnessChecker, createLlmReviewer, createOtelExporter, createOtelTracingStore, createReferenceEquivalenceJudge, createReplayFetch, createSandboxPool, createSemanticConceptJudge, createTokenRecallChecker, createTraceAnalystKind, crossTraceDiff, crowdingDistance, dataDescriptionBits, decideNextUserTurn, decidePairedPromotion, decideReferenceReplayPromotion, decideReferenceReplayRunPromotion, defaultBlendWeights, defaultIsMaterial, defaultProviderRedactor, defaultReferenceReplayMatcher, defaultTraceInsightPanel, deployGateLayer, describeTraceInsightScope, diffFindings, diffScorecard, discoverPersonas, distillPlaybook, domainEvidencePattern, dominates, eProcess, emitControlIntegrityFindings, ensembleJudge, errorStreakDetector, estimateCost, estimateTokens, evaluateActionPolicy, evaluateContract, evaluateHypothesis, evaluateInterimReleaseConfidence, evaluateOracles, evaluateReleaseConfidence, evaluateTraceContract, executeScenario, expandProfileAxes, expectAgent, exportProductBenchmark, exportProductBenchmarkRuns, exportRewardModel, exportRunAsOtlp, extractAssetUrls, extractErrorCount, extractOtlpAttributes, extractProducedState, extractUsage, extractUsageFromResponse, extractUsageFromSse, feedbackTrajectoriesToDatasetScenarios, feedbackTrajectoriesToOptimizerRows, feedbackTrajectoryToDatasetScenario, feedbackTrajectoryToOptimizerRow, fileContains, fileExists, fileExperimentStore, fileVerdictCache, findAutoMatchNoExpectation, findConstructorCwdDropped, findFallbackToPass, findLiteralTruePass, findProductBenchmarkArtifacts, findSkipCountsAsPass, firstNumberAttr, firstStringAttr, flattenOtlpExportToNdjson, flowLayer, fnv1a32, formatBenchmarkReport, formatDriverReport, formatFindings, formatScorecardDiff, fromHarborTrajectory, gainHistogram, gateTreatmentApplied, gateTreatmentFromMetrics, gateTreatmentFromSpans, gateTreatmentFromToolSpans, ghCliClient, gitProvenanceReader, precision as goldenPrecision, gradeOnHidden, gradeSemanticStatus, groupBy, groupRunsByAgentProfileCell, harnessAxisOf, hasCapturedToolArgs, hashContent, hashJson, hashScenarios, hashToUnit, hiddenGrade, holm, htmlContainsElement, httpGithubClient, improvementVerdict, inMemoryExperimentStore, inMemoryReferenceReplayStore, inMemoryReviewStore, inMemoryRunRecordBackend, inMemoryVerdictCache, inferDomainKeywords, inferOtlpKind, interRaterReliability, interpretCliffs, iqr, isBinaryOutcomeVector, isHiddenDestination, isJudgeSpan, isLlmSpan, isModelPriced, isOtelConfigured, isOtlpModelCall, isRealnessGated, isRetrievalSpan, isRolloutLine, isRunRecord, isSandboxSpan, isToolSpan, isTrainableSplit, isTransientLlmError, isUnavailable, iterateRawCalls, jestTestParser, jsonHasKeys, jsonShape, jsonlReferenceReplayStore, jsonlReviewStore, jsonlRunRecordBackend, judgeFamily, judgeReplayGate, judgeSpans, keyPreserved, knowledgeReadinessTracePayload, leaderboard, linterJudge, llmJudge, llmSpanFromProvider, llmSpans, loadScorecard, loadScorerFromGrader, localCommandRunner, lowercaseMutator, makeEvalTools, makeFinding, makeProposalFinding, mannWhitneyU, mapConcurrent, matchGoldens, matchSpan, maximumChargeForLlmRequest, mcnemar, mcnemarPower, mcnemarRequiredN, mergeLayerResults, mergeSteeringBundle, minimumPairsForPairedDeltaTest, mintRolloutRows, modelDescriptionBits, modelHasSnapshot, modelPriceKey, mulberry32, multiToolchainLayer, noProgressDetector, normalCdf, normalizeScores, notBlocked, objectiveEval, observeAll, observedScore, observedSplitScore, otelRunCompleteHook, otlpRowsToRunRecords, otlpRowsToTraceRunRecords, otlpTextToTraceAnalysisStore, otlpToRunRecords, otlpToTraceRunRecords, pairArms, pairRunRecords, pairedBinaryScale, pairedBootstrap, pairedCohensDz, pairedDecisionShape, pairedDeltaTest, pairedDeltaTieFraction, pairedEvalueSequence, pairedMde, pairedRiskDifference, pairedRiskDifferenceExact, pairedRiskDifferenceScore, pairedSignTest, pairedTTest, paraphraseRobustness, paraphraseRobustnessScenarios, paretoChart, paretoFrontier, paretoFrontierWithCrowding, parseCorrectnessResponse, parseFeedbackTrajectoriesJsonl, parseReflectionResponse, parseRunRecordSafe, parseRuntimeTrajectoryHookEvent, partialCredit, partitionHeldOut, passAtK, passOrthogonality, pearsonR, pixelDeltaRatio, planTraceInsightQuestions, politenessPrefixMutator, positionalBias, preflightModels, printDriverSummary, probeLlm, productBenchmarkIntegrityFailures, productBenchmarkMutableSurfaces, productBenchmarkRepoIdentity, productBenchmarkSplits, profile_exports as profile, projectOtlpFlatLine, projectRuntimeTrajectoryEvidence, promptBisect, proposeSynthesisTargets, providerFromBaseUrl, pytestTestParser, ranks, readClaudeCodeSupervisorRun, readOtlpStatus, readProductBenchmarkManifest, readProductBenchmarkRecords, recordRuns, recordRunsToScorecard, redTeamDataset, redTeamReport, redactString, redactValue, referenceReplayRunsToSteeringRows, referenceReplayScenarioToRunScore, regexMatch, regexMatches, relabelImportedSplit, renderMarkdownReport, renderPlaybookMarkdown, renderPreferenceMemoryMarkdown, renderPriorFindings, renderReleaseReport, renderSteeringText, renderSupervisorRunHeadline, renderSupervisorRunMarkdown, renderUpstreamFindings, repeatedActionDetector, replayFeedbackTrajectories, replayFeedbackTrajectory, replayScorerOverCorpus, replayTraceThroughJudge, requireAgentProfileCell, requiredPairedSampleSize, requiredSampleSize, researchReport, resolveModelPricing, resolveSeat, rollupSupervisorRuns, roundTripRunRecord, routeFields, rowCount, rowWhere, runAgentControlLoop, runAssertions, runBehavioralCanaries, runCanaries, runCounterfactual, runE2EWorkflow, runEvalCampaign, runExpectations, runFailureClass, runHarnessExperiment, runIntentMatchJudge, runJudgeFleet, runKeywordCoverageJudge, runKeywordCoverageJudgeUrl, runLiveProof, runProposeReview, runProposeReviewAsControlLoop, runRecordToProductBenchmarkRecord, runReferenceEquivalenceJudge, runReferenceReplay, runScore, runSelfPlay, runSemanticConceptJudge, runTaskScore, runTestGradedScenario, runsForScenario, scalarScore, scanForMuffledGates, scoreContinuity, scoreFromEvals, scoreKnowledgeReadiness, scoreOrigin, scorePrReviewComments, scorePrReviewSource, scoreRedTeamOutput, scoreReferenceReplay, scoreTraceInsightReadiness, seatPresets, securityJudge, selectHarnessVariant, selfPreference, sentenceReorderMutator, serializeFeedbackTrajectoriesJsonl, showMeasured, signManifest, spearmanR, statusAdvanced, stopOnNoProgress, stopOnRepeatedAction, stringField, stripFencedJson, studentTCdf, subjectiveEval, summarizeAgentReceiptIntegrity, summarizeBackendIntegrity, summarizeHarnessResults, summarizePrReviewBenchmark, summarizePreferenceMemory, summaryTable, supervisorRunRolloutLines, testJudge, textInSnapshot, throwIfRunIncomplete, toAgentProfileJson, toHarborTrajectories, toHarborTrajectory, toJsonl, toLangfuseEnvelope, toOpenAiTool, toPrometheusText, toRewardRows, toSftRows, tokenizeDomainWords, toolNamesForRun, toolSpans, toolSpansToTraceAnalysisStore, traceAnalystFunctionGroup, traceAnalystOnRunComplete, traceContract, traceJudge, traceJudgeEnsemble, traceSpanKindToOpenInferenceKind, tracedAnalyzeTraces, trainingReward, trainingScore, typoMutator, unmintableReasons, urlContains, userQuestionsForKnowledgeGaps, validateAgentProfileCell, validateProductBenchmarkManifest, validateProductBenchmarkRecord, validateProductBenchmarkRun, validateRolloutLine, validateRunRecord, verbosityBias, verifyAgentProfileCell, verifyAttestation, verifyCompletion, verifyManifest, visualDiff, viteDeployRunner, vitestTestParser, weightedComposite, weightedMean, weightedRecall, welchsTTest, whitespaceCollapseMutator, wilcoxonSignedRank, wilson, withAssignedFeedbackSplit, withHeldoutBlend, withJudgeRetry, withOtelPipeline, wranglerDeployRunner, writeSupervisorRunReport };
|
|
12376
12850
|
|
|
12377
12851
|
//# sourceMappingURL=index.js.map
|