@tangle-network/agent-eval 0.144.6 → 0.144.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +24 -0
- package/README.md +2 -0
- package/dist/{benchmark-J9Qe6j2_.d.ts → agent-profile-CgDTo40f.d.ts} +91 -4
- package/dist/agent-profile-CgDTo40f.d.ts.map +1 -0
- package/dist/analyst/index.d.ts +470 -88
- package/dist/analyst/index.d.ts.map +1 -1
- package/dist/analyst/index.js +24 -5
- package/dist/analyst/index.js.map +1 -1
- package/dist/{analyze-runs-DWIvOAGk.js → analyze-runs-BNNK7irB.js} +3 -3
- package/dist/{analyze-runs-DWIvOAGk.js.map → analyze-runs-BNNK7irB.js.map} +1 -1
- package/dist/{baseline-D_fT6277.d.ts → baseline-CavEbRyH.d.ts} +2 -23
- package/dist/baseline-CavEbRyH.d.ts.map +1 -0
- package/dist/{benchmark-command-CQd78YHt.js → benchmark-command-BCafwNrf.js} +662 -605
- package/dist/benchmark-command-BCafwNrf.js.map +1 -0
- package/dist/benchmarks/index.d.ts +1 -1
- package/dist/benchmarks/index.js +1 -1
- package/dist/{benchmarks-BEOkuvIg.js → benchmarks-CDSolHq7.js} +4 -4
- package/dist/{benchmarks-BEOkuvIg.js.map → benchmarks-CDSolHq7.js.map} +1 -1
- package/dist/campaign/index.d.ts +7 -5
- package/dist/campaign/index.js +5 -3
- package/dist/{campaign-CXsdyym7.js → campaign-Tdy3h62h.js} +17 -301
- package/dist/campaign-Tdy3h62h.js.map +1 -0
- package/dist/cli.js +2 -2
- package/dist/{client-0JI64ovJ.d.ts → client-DjXROWpx.d.ts} +3 -3
- package/dist/{client-0JI64ovJ.d.ts.map → client-DjXROWpx.d.ts.map} +1 -1
- package/dist/{completion-verifier-CBiee74w.d.ts → completion-verifier-foUCLif_.d.ts} +5 -5
- package/dist/{completion-verifier-CBiee74w.d.ts.map → completion-verifier-foUCLif_.d.ts.map} +1 -1
- package/dist/contract/index.d.ts +9 -8
- package/dist/contract/index.d.ts.map +1 -1
- package/dist/contract/index.js +7 -6
- package/dist/contract/index.js.map +1 -1
- package/dist/control.d.ts +2 -2
- package/dist/control.js +1 -1
- package/dist/counterfactual-CWPTrMH7.js +126 -0
- package/dist/counterfactual-CWPTrMH7.js.map +1 -0
- package/dist/counterfactual-CxmxAONP.d.ts +72 -0
- package/dist/counterfactual-CxmxAONP.d.ts.map +1 -0
- package/dist/{default-registry-J9m-_tya.d.ts → default-registry-BZhStdjl.d.ts} +6 -5
- package/dist/default-registry-BZhStdjl.d.ts.map +1 -0
- package/dist/{default-registry-Dta70shL.js → default-registry-BaQXW1Ow.js} +2 -2
- package/dist/{default-registry-Dta70shL.js.map → default-registry-BaQXW1Ow.js.map} +1 -1
- package/dist/{dspy-rlm-engine-19FQEMBK.js → dspy-rlm-engine-BiN49gK6.js} +4 -3
- package/dist/{dspy-rlm-engine-19FQEMBK.js.map → dspy-rlm-engine-BiN49gK6.js.map} +1 -1
- package/dist/{tool-groups-CK0JCkqO.d.ts → engine-nB64f48I.d.ts} +18 -31
- package/dist/engine-nB64f48I.d.ts.map +1 -0
- package/dist/{eval-campaign-CfLQQs9B.js → eval-campaign-DNjCvAm-.js} +7 -6
- package/dist/{eval-campaign-CfLQQs9B.js.map → eval-campaign-DNjCvAm-.js.map} +1 -1
- package/dist/{exact-types-CBYF5MGd.d.ts → exact-types-Djvzosly.d.ts} +2 -2
- package/dist/{exact-types-CBYF5MGd.d.ts.map → exact-types-Djvzosly.d.ts.map} +1 -1
- package/dist/exec-BLtYZdWo.js +49 -0
- package/dist/exec-BLtYZdWo.js.map +1 -0
- package/dist/experiment/index.d.ts +802 -0
- package/dist/experiment/index.d.ts.map +1 -0
- package/dist/experiment/index.js +1108 -0
- package/dist/experiment/index.js.map +1 -0
- package/dist/experiment-tracker-CnRICnMl.js +500 -0
- package/dist/experiment-tracker-CnRICnMl.js.map +1 -0
- package/dist/experiment-tracker-IMntXr6J.d.ts +299 -0
- package/dist/experiment-tracker-IMntXr6J.d.ts.map +1 -0
- package/dist/{external-optimizer-contracts-Q9c0L3eY.d.ts → external-optimizer-contracts-CZuJNcT5.d.ts} +2 -2
- package/dist/{external-optimizer-contracts-Q9c0L3eY.d.ts.map → external-optimizer-contracts-CZuJNcT5.d.ts.map} +1 -1
- package/dist/{feedback-trajectory-GgoS0-MK.d.ts → feedback-trajectory-Rh280oXo.d.ts} +3 -3
- package/dist/{feedback-trajectory-GgoS0-MK.d.ts.map → feedback-trajectory-Rh280oXo.d.ts.map} +1 -1
- package/dist/fuzz.d.ts +1 -1
- package/dist/hosted/index.d.ts +3 -3
- package/dist/{index-BZUe-ODI.d.ts → index-BZ3-y4YL.d.ts} +3 -3
- package/dist/{index-BZUe-ODI.d.ts.map → index-BZ3-y4YL.d.ts.map} +1 -1
- package/dist/{index-DSC51roc.d.ts → index-C3ssXVLv.d.ts} +2 -2
- package/dist/{index-DSC51roc.d.ts.map → index-C3ssXVLv.d.ts.map} +1 -1
- package/dist/{index-4XwggC10.d.ts → index-C5HOo4ZF2.d.ts} +4 -4
- package/dist/index-C5HOo4ZF2.d.ts.map +1 -0
- package/dist/{index-B6-B0zTB.d.ts → index-CvXXlyz7.d.ts} +2 -2
- package/dist/{index-B6-B0zTB.d.ts.map → index-CvXXlyz7.d.ts.map} +1 -1
- package/dist/{index-Dx1kF3Ez.d.ts → index-CwDrUMe0.d.ts} +2 -2
- package/dist/{index-Dx1kF3Ez.d.ts.map → index-CwDrUMe0.d.ts.map} +1 -1
- package/dist/{index-BIL5vxxt.d.ts → index-Sh2I0DRc.d.ts} +11 -645
- package/dist/index-Sh2I0DRc.d.ts.map +1 -0
- package/dist/index.d.ts +214 -404
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +233 -649
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-DqEsugpr.d.ts → insight-report-C6h6F_4L.d.ts} +3 -3
- package/dist/{insight-report-DqEsugpr.d.ts.map → insight-report-C6h6F_4L.d.ts.map} +1 -1
- package/dist/{integrity-CNGUaGBY.d.ts → integrity-BuqEKu-x.d.ts} +2 -2
- package/dist/{integrity-CNGUaGBY.d.ts.map → integrity-BuqEKu-x.d.ts.map} +1 -1
- package/dist/integrity-MLzHOfV9.js +141 -0
- package/dist/integrity-MLzHOfV9.js.map +1 -0
- package/dist/kind-factory-BHIgPmzS.js.map +1 -1
- package/dist/{llm-client-Dv5BiKLE.js → llm-client-DzvMUsS_.js} +24 -6
- package/dist/llm-client-DzvMUsS_.js.map +1 -0
- package/dist/matrix/index.d.ts +2 -2
- package/dist/meta-eval/index.d.ts +1 -1
- package/dist/{mint-DD-0oQTA.js → mint-CGEkzPLf.js} +2 -2
- package/dist/{mint-DD-0oQTA.js.map → mint-CGEkzPLf.js.map} +1 -1
- package/dist/{multi-layer-verifier-BHY1gWAc.d.ts → multi-layer-verifier-DnAqwl0h.d.ts} +2 -2
- package/dist/{multi-layer-verifier-BHY1gWAc.d.ts.map → multi-layer-verifier-DnAqwl0h.d.ts.map} +1 -1
- package/dist/multishot/index.d.ts +2 -2
- package/dist/openapi.json +1 -1
- package/dist/paired-promotion-decision-B6zJ3gYM.d.ts +114 -0
- package/dist/paired-promotion-decision-B6zJ3gYM.d.ts.map +1 -0
- package/dist/pipelines/index.d.ts +2 -1
- package/dist/pipelines/index.d.ts.map +1 -1
- package/dist/pre-registration-DakwTRXk.js +96 -0
- package/dist/pre-registration-DakwTRXk.js.map +1 -0
- package/dist/prime-bridge-transport-6feEglLf.d.ts +49 -0
- package/dist/prime-bridge-transport-6feEglLf.d.ts.map +1 -0
- package/dist/prime-protocol-BfSalTfR.js +453 -0
- package/dist/prime-protocol-BfSalTfR.js.map +1 -0
- package/dist/profile-cell.js +242 -1
- package/dist/profile-cell.js.map +1 -0
- package/dist/promotion-policy-ChWhTDBH.d.ts +289 -0
- package/dist/promotion-policy-ChWhTDBH.d.ts.map +1 -0
- package/dist/promotion-policy-CrLrmys8.js +682 -0
- package/dist/promotion-policy-CrLrmys8.js.map +1 -0
- package/dist/{propose-review-control-BciUCZoh.js → propose-review-control-qgWLA6E9.js} +2 -2
- package/dist/{propose-review-control-BciUCZoh.js.map → propose-review-control-qgWLA6E9.js.map} +1 -1
- package/dist/{release-report-ChOgpIoQ.d.ts → release-report-CI8uisI1.d.ts} +2 -2
- package/dist/{release-report-ChOgpIoQ.d.ts.map → release-report-CI8uisI1.d.ts.map} +1 -1
- package/dist/{release-report-Sl0xfkFv.js → release-report-Dy39cbFF.js} +3 -3
- package/dist/{release-report-Sl0xfkFv.js.map → release-report-Dy39cbFF.js.map} +1 -1
- package/dist/{replay-Krvb114g.d.ts → replay-DFf-teiC.d.ts} +5 -4
- package/dist/replay-DFf-teiC.d.ts.map +1 -0
- package/dist/{replay-GW61ezMW.js → replay-DyBLaKFc.js} +3 -3
- package/dist/{replay-GW61ezMW.js.map → replay-DyBLaKFc.js.map} +1 -1
- package/dist/reporting.d.ts +3 -3
- package/dist/reporting.js +2 -2
- package/dist/{researcher-xLeNcpKX.d.ts → researcher-BoaxeCzP.d.ts} +4 -4
- package/dist/{researcher-xLeNcpKX.d.ts.map → researcher-BoaxeCzP.d.ts.map} +1 -1
- package/dist/{reward-hacking-CyuzxKly.js → reward-hacking-B2tPL9a3.js} +12 -3
- package/dist/reward-hacking-B2tPL9a3.js.map +1 -0
- package/dist/{reward-hacking-RZgnGWlx.d.ts → reward-hacking-Cf1PtEOz.d.ts} +33 -3
- package/dist/reward-hacking-Cf1PtEOz.d.ts.map +1 -0
- package/dist/rl.d.ts +17 -7
- package/dist/rl.d.ts.map +1 -1
- package/dist/rl.js +16 -7
- package/dist/rl.js.map +1 -1
- package/dist/rollout/index.d.ts +2 -2
- package/dist/rollout/index.js +2 -2
- package/dist/{rollout-C2fD1cf4.js → rollout-2ECTXb2N.js} +2 -2
- package/dist/{rollout-C2fD1cf4.js.map → rollout-2ECTXb2N.js.map} +1 -1
- package/dist/{run-evidence-C6G41MSI.d.ts → run-evidence-BDFFai9R.d.ts} +2 -2
- package/dist/{run-evidence-C6G41MSI.d.ts.map → run-evidence-BDFFai9R.d.ts.map} +1 -1
- package/dist/{run-record-CWN8-VsV.js → run-record-DqOw5X6_.js} +2 -2
- package/dist/{run-record-CWN8-VsV.js.map → run-record-DqOw5X6_.js.map} +1 -1
- package/dist/{schema-Cef2cFmb.d.ts → schema-Cef2cFmb2.d.ts} +1 -1
- package/dist/schema-Cef2cFmb2.d.ts.map +1 -0
- package/dist/{semantic-concept-judge-DwF6n05O.js → semantic-concept-judge-Bmrq6yqU.js} +2 -2
- package/dist/{semantic-concept-judge-DwF6n05O.js.map → semantic-concept-judge-Bmrq6yqU.js.map} +1 -1
- package/dist/sequential-D-BLJBKU.js +299 -0
- package/dist/sequential-D-BLJBKU.js.map +1 -0
- package/dist/{server-D6XJQHw7.js → server-iu0ede49.js} +2 -2
- package/dist/{server-D6XJQHw7.js.map → server-iu0ede49.js.map} +1 -1
- package/dist/{single-run-lock-D5iN0Xzb.js → single-run-lock-BMQEv1wG.js} +7 -161
- package/dist/single-run-lock-BMQEv1wG.js.map +1 -0
- package/dist/{skill-usage-GlOphAhX.d.ts → skill-usage-CJlWEUFt.d.ts} +10 -10
- package/dist/{skill-usage-GlOphAhX.d.ts.map → skill-usage-CJlWEUFt.d.ts.map} +1 -1
- package/dist/{skillopt-optimization-method-7S43rbDB.d.ts → skillopt-optimization-method-BO7NAl3b.d.ts} +8 -294
- package/dist/skillopt-optimization-method-BO7NAl3b.d.ts.map +1 -0
- package/dist/{skillopt-optimization-method-Bfb-vBKe.js → skillopt-optimization-method-CQwZ-ZX8.js} +8 -652
- package/dist/skillopt-optimization-method-CQwZ-ZX8.js.map +1 -0
- package/dist/statistical-heldout-Dn9ruizm.d.ts +555 -0
- package/dist/statistical-heldout-Dn9ruizm.d.ts.map +1 -0
- package/dist/{statistics-C-dm-J6H.d.ts → statistics-D6Uebe_4.d.ts} +2 -2
- package/dist/{statistics-C-dm-J6H.d.ts.map → statistics-D6Uebe_4.d.ts.map} +1 -1
- package/dist/steps-BArUxhna.d.ts +51 -0
- package/dist/steps-BArUxhna.d.ts.map +1 -0
- package/dist/{integrity-fdt8XPAv.js → store-DNe_Uv1Q.js} +2 -140
- package/dist/store-DNe_Uv1Q.js.map +1 -0
- package/dist/{summary-report-B0cAyA7N.d.ts → summary-report-DuUS_i7W.d.ts} +3 -114
- package/dist/summary-report-DuUS_i7W.d.ts.map +1 -0
- package/dist/{summary-report-9A5y7EsK.js → summary-report-Lf-5I7xh.js} +2 -2
- package/dist/{summary-report-9A5y7EsK.js.map → summary-report-Lf-5I7xh.js.map} +1 -1
- package/dist/supervisor-run/index.d.ts +2 -2
- package/dist/supervisor-run/index.js +1 -1
- package/dist/{supervisor-run-DiyQVczd.js → supervisor-run-D_sokXcO.js} +32 -8
- package/dist/{supervisor-run-DiyQVczd.js.map → supervisor-run-D_sokXcO.js.map} +1 -1
- package/dist/tool-groups-DIVBnJyl.d.ts +28 -0
- package/dist/tool-groups-DIVBnJyl.d.ts.map +1 -0
- package/dist/trace-repair/index.d.ts +2102 -0
- package/dist/trace-repair/index.d.ts.map +1 -0
- package/dist/trace-repair/index.js +3878 -0
- package/dist/trace-repair/index.js.map +1 -0
- package/dist/traces.d.ts +6 -6
- package/dist/traces.js +3 -2
- package/dist/trajectory-YC15QDYQ.d.ts +24 -0
- package/dist/trajectory-YC15QDYQ.d.ts.map +1 -0
- package/dist/trajectory-replay/index.d.ts +781 -0
- package/dist/trajectory-replay/index.d.ts.map +1 -0
- package/dist/trajectory-replay/index.js +2103 -0
- package/dist/trajectory-replay/index.js.map +1 -0
- package/dist/{types-XMVEdrE_.d.ts → types-D216SgwM.d.ts} +24 -6
- package/dist/types-D216SgwM.d.ts.map +1 -0
- package/dist/{types-yLK8gXE9.d.ts → types-D4mog56g.d.ts} +2 -2
- package/dist/{types-yLK8gXE9.d.ts.map → types-D4mog56g.d.ts.map} +1 -1
- package/dist/{types-BhP9q0Fq.d.ts → types-DF_Udrp-.d.ts} +52 -3
- package/dist/{types-BhP9q0Fq.d.ts.map → types-DF_Udrp-.d.ts.map} +1 -1
- package/dist/{types-DOZyvsFU.d.ts → types-DYuNHo9R.d.ts} +3 -3
- package/dist/{types-DOZyvsFU.d.ts.map → types-DYuNHo9R.d.ts.map} +1 -1
- package/dist/usage-receipt-EVI8B8Xu.js.map +1 -1
- package/dist/verdict-DExhxfgR.d.ts +201 -0
- package/dist/verdict-DExhxfgR.d.ts.map +1 -0
- package/dist/verdict-cache-BCcOh0kF.js +159 -0
- package/dist/verdict-cache-BCcOh0kF.js.map +1 -0
- package/dist/wire/index.d.ts +2 -2
- package/dist/wire/index.js +1 -1
- package/docs/charter.md +112 -0
- package/docs/experiment.md +104 -0
- package/docs/prime-analyst.md +1 -0
- package/docs/trace-analysis.md +26 -0
- package/docs/trace-repair-admission.md +194 -0
- package/docs/trace-repair-analyst-arms.md +121 -0
- package/docs/trace-repair-continuation.md +107 -0
- package/docs/trace-repair-grader.md +163 -0
- package/docs/trajectory-replay.md +110 -0
- package/docs/verification-strategies.md +103 -0
- package/package.json +19 -2
- package/dist/agent-profile-cell-CbfBm2g6.js +0 -335
- package/dist/agent-profile-cell-CbfBm2g6.js.map +0 -1
- package/dist/baseline-D_fT6277.d.ts.map +0 -1
- package/dist/benchmark-J9Qe6j2_.d.ts.map +0 -1
- package/dist/benchmark-command-CQd78YHt.js.map +0 -1
- package/dist/campaign-CXsdyym7.js.map +0 -1
- package/dist/default-registry-J9m-_tya.d.ts.map +0 -1
- package/dist/index-4XwggC10.d.ts.map +0 -1
- package/dist/index-BIL5vxxt.d.ts.map +0 -1
- package/dist/integrity-fdt8XPAv.js.map +0 -1
- package/dist/llm-client-Dv5BiKLE.js.map +0 -1
- package/dist/replay-Krvb114g.d.ts.map +0 -1
- package/dist/reward-hacking-CyuzxKly.js.map +0 -1
- package/dist/reward-hacking-RZgnGWlx.d.ts.map +0 -1
- package/dist/schema-Cef2cFmb.d.ts.map +0 -1
- package/dist/single-run-lock-D5iN0Xzb.js.map +0 -1
- package/dist/skillopt-optimization-method-7S43rbDB.d.ts.map +0 -1
- package/dist/skillopt-optimization-method-Bfb-vBKe.js.map +0 -1
- package/dist/summary-report-B0cAyA7N.d.ts.map +0 -1
- package/dist/tool-groups-CK0JCkqO.d.ts.map +0 -1
- package/dist/types-XMVEdrE_.d.ts.map +0 -1
- package/dist/verdict-Dps8_okt.d.ts +0 -37
- package/dist/verdict-Dps8_okt.d.ts.map +0 -1
package/dist/analyst/index.d.ts
CHANGED
|
@@ -1,14 +1,17 @@
|
|
|
1
|
-
import { a as MultiLayerVerifier, l as VerifyOptions, o as Severity } from "../multi-layer-verifier-
|
|
2
|
-
import { A as parseFindingSubject, C as FINDING_SUBJECT_KINDS, D as FindingSubjectStringSchema, E as FindingSubjectKind, F as DefineExactCustomAnalystOptions, G as SemanticConceptJudgeOptions, I as defineCustomAnalyst, L as defineTraceAnalyst, M as DspyRlmTraceEngineOptions, N as createDspyRlmTraceEngine, O as KIND_EXPECTED_SUBJECTS, P as DefineCustomAnalystOptions, S as FINDING_SUBJECT_GRAMMAR_PROMPT, T as FindingSubject, W as SemanticConceptJudgeInput, Y as RunCritic, Z as RunTrace, _ as FindingsDiff, a as SkillUsageScanConfig, b as defaultIsMaterial, c as DEFAULT_TRACE_ANALYST_KINDS, d as IMPROVEMENT_KIND_SPEC, f as FAILURE_MODE_KIND_SPEC, g as DiffPolicy, h as emitControlIntegrityFindings, i as SkillUsageReport, j as renderFindingSubject, k as findingSubjectGrammarPromptFor, l as KNOWLEDGE_POISONING_KIND_SPEC, m as ControlIntegrityAnalyst, n as SkillUsageAnalyst, o as buildSkillUsageReport, p as CONTROL_INTEGRITY_ANALYST, r as SkillUsageRecord, s as emitSkillUsageFindings, t as SKILL_USAGE_ANALYST, u as KNOWLEDGE_GAP_KIND_SPEC, v as FindingsStore, w as FINDING_SUBJECT_SYNTAX, x as diffFindings, y as PersistedFinding } from "../skill-usage-
|
|
1
|
+
import { a as MultiLayerVerifier, l as VerifyOptions, o as Severity } from "../multi-layer-verifier-DnAqwl0h.js";
|
|
2
|
+
import { A as parseFindingSubject, C as FINDING_SUBJECT_KINDS, D as FindingSubjectStringSchema, E as FindingSubjectKind, F as DefineExactCustomAnalystOptions, G as SemanticConceptJudgeOptions, I as defineCustomAnalyst, L as defineTraceAnalyst, M as DspyRlmTraceEngineOptions, N as createDspyRlmTraceEngine, O as KIND_EXPECTED_SUBJECTS, P as DefineCustomAnalystOptions, S as FINDING_SUBJECT_GRAMMAR_PROMPT, T as FindingSubject, W as SemanticConceptJudgeInput, Y as RunCritic, Z as RunTrace, _ as FindingsDiff, a as SkillUsageScanConfig, b as defaultIsMaterial, c as DEFAULT_TRACE_ANALYST_KINDS, d as IMPROVEMENT_KIND_SPEC, f as FAILURE_MODE_KIND_SPEC, g as DiffPolicy, h as emitControlIntegrityFindings, i as SkillUsageReport, j as renderFindingSubject, k as findingSubjectGrammarPromptFor, l as KNOWLEDGE_POISONING_KIND_SPEC, m as ControlIntegrityAnalyst, n as SkillUsageAnalyst, o as buildSkillUsageReport, p as CONTROL_INTEGRITY_ANALYST, r as SkillUsageRecord, s as emitSkillUsageFindings, t as SKILL_USAGE_ANALYST, u as KNOWLEDGE_GAP_KIND_SPEC, v as FindingsStore, w as FINDING_SUBJECT_SYNTAX, x as diffFindings, y as PersistedFinding } from "../skill-usage-CJlWEUFt.js";
|
|
3
3
|
import { b as CustomTokenPricing, c as CostLedgerHandle } from "../cost-ledger-Bv_e8XHY.js";
|
|
4
|
-
import { A as ChatClient, B as SandboxSdkTransportOpts, F as CreateChatClientOpts, I as CustomTransportOpts, L as DirectProviderTransportOpts, M as ChatResponse, N as ChatTransport, P as CliBridgeTransportOpts, R as MockTransportOpts, V as createChatClient, j as ChatRequest, k as ChatCallOpts, m as JudgeInput, p as JudgeFn, z as RouterTransportOpts } from "../types-
|
|
5
|
-
import { _ as
|
|
6
|
-
import { a as createTraceAnalyst, c as runTraceAnalyst, d as deriveEfficiencyFindings, i as TraceAnalystDefinition, l as BehavioralAnalystOptions, n as buildDefaultAnalystRegistry, o as renderPriorFindings, r as CreateTraceAnalystOptions, s as renderUpstreamFindings, t as DefaultAnalystRegistryOptions, u as behavioralAnalyst } from "../default-registry-
|
|
7
|
-
import { a as ExactAnalystRunPolicySnapshot, c as ExactAnalystSnapshot, d as ExactExecutionComponentSnapshot, i as ExactAnalystRunEvent, l as ExactCapableAnalyst, n as ExactAnalystExecutionPlanSnapshot, o as ExactAnalystRunResult, r as ExactAnalystRunCompletion, s as ExactAnalystRunSummary, t as ExactAnalystBudgetSnapshot, u as ExactExecutionComponentIdentity } from "../exact-types-
|
|
8
|
-
import { A as ExactAnalystRunExecutionError, D as AnalystRegistryOptions, E as AnalystRegistry, M as RegistryRunOpts, N as assertExactRegistryRunOpts, O as BudgetPolicy, T as AnalystHooks, j as ExactRegistryRunOpts, k as ExactAnalystBudgetPolicy } from "../completion-verifier-
|
|
9
|
-
import {
|
|
10
|
-
import { f as ExternalOptimizerModelExecutionObservation, h as ExternalOptimizerRunnerCommand, l as ExternalOptimizerModelCall } from "../external-optimizer-contracts-
|
|
11
|
-
import { a as
|
|
4
|
+
import { A as ChatClient, B as SandboxSdkTransportOpts, F as CreateChatClientOpts, I as CustomTransportOpts, L as DirectProviderTransportOpts, M as ChatResponse, N as ChatTransport, P as CliBridgeTransportOpts, R as MockTransportOpts, V as createChatClient, j as ChatRequest, k as ChatCallOpts, m as JudgeInput, p as JudgeFn, z as RouterTransportOpts } from "../types-D216SgwM.js";
|
|
5
|
+
import { I as TraceAnalystSpan, _ as ProposalFinding, a as AnalystInputKind, b as makeFinding, c as AnalystRunInputs, d as AnalystSeverity, f as AnalystUsageReceipt, g as ExecutionProbeRequest, h as ExecutionProbeOutcome, i as AnalystFinding, l as AnalystRunResult, m as ExecutionProbe, n as AnalystContext, o as AnalystRequirements, p as EvidenceRef, r as AnalystCost, s as AnalystRunEvent, t as Analyst, u as AnalystRunSummary, v as ProposalFindingOrigin, w as TraceAnalysisStore, x as makeProposalFinding, y as computeFindingId } from "../types-DF_Udrp-.js";
|
|
6
|
+
import { a as createTraceAnalyst, c as runTraceAnalyst, d as deriveEfficiencyFindings, i as TraceAnalystDefinition, l as BehavioralAnalystOptions, n as buildDefaultAnalystRegistry, o as renderPriorFindings, r as CreateTraceAnalystOptions, s as renderUpstreamFindings, t as DefaultAnalystRegistryOptions, u as behavioralAnalyst } from "../default-registry-BZhStdjl.js";
|
|
7
|
+
import { a as ExactAnalystRunPolicySnapshot, c as ExactAnalystSnapshot, d as ExactExecutionComponentSnapshot, i as ExactAnalystRunEvent, l as ExactCapableAnalyst, n as ExactAnalystExecutionPlanSnapshot, o as ExactAnalystRunResult, r as ExactAnalystRunCompletion, s as ExactAnalystRunSummary, t as ExactAnalystBudgetSnapshot, u as ExactExecutionComponentIdentity } from "../exact-types-Djvzosly.js";
|
|
8
|
+
import { A as ExactAnalystRunExecutionError, D as AnalystRegistryOptions, E as AnalystRegistry, M as RegistryRunOpts, N as assertExactRegistryRunOpts, O as BudgetPolicy, T as AnalystHooks, j as ExactRegistryRunOpts, k as ExactAnalystBudgetPolicy } from "../completion-verifier-foUCLif_.js";
|
|
9
|
+
import { A as registryBenchmarkRunner, C as AnalystEvidenceResolution, D as AnalystIssueExpectation, E as AnalystFindingScore, M as traceStoreEvidenceResolver, N as scoreAnalystFindings, O as AnalystLatencyDistribution, S as AnalystEvidenceExpectation, T as AnalystEvidenceResolver, _ as AnalystBenchmarkOutput, b as AnalystBenchmarkRunner, d as AnalystBenchmarkCase, f as AnalystBenchmarkDatasetRef, g as AnalystBenchmarkObservation, h as AnalystBenchmarkLabelState, j as runAnalystBenchmark, k as RunAnalystBenchmarkOptions, m as AnalystBenchmarkError, p as AnalystBenchmarkDescriptor, t as AgentProfile, v as AnalystBenchmarkProvenance, w as AnalystEvidenceResolutionError, x as AnalystBenchmarkSummary, y as AnalystBenchmarkResult } from "../agent-profile-CgDTo40f.js";
|
|
10
|
+
import { f as ExternalOptimizerModelExecutionObservation, h as ExternalOptimizerRunnerCommand, l as ExternalOptimizerModelCall } from "../external-optimizer-contracts-CZuJNcT5.js";
|
|
11
|
+
import { a as TraceAnalystLimits, c as RAW_FINDING_SCHEMA_PROMPT, d as RawAnalystFinding, f as RawAnalystFindingSchema, i as TraceAnalysisEngineResult, l as RawAnalystEvidence, m as parseRawFinding, n as TraceAnalysisEngine, o as resolveTraceAnalystLimits, p as evidenceRefsFromRawFinding, r as TraceAnalysisEngineRequest, s as ANALYST_SEVERITIES, t as DEFAULT_TRACE_ANALYST_LIMITS, u as RawAnalystEvidenceSchema } from "../engine-nB64f48I.js";
|
|
12
|
+
import { n as buildTraceToolsForGroup, t as TraceToolGroupName } from "../tool-groups-DIVBnJyl.js";
|
|
13
|
+
import { i as nodeHttpPrimeBridgeTransport, n as PrimeBridgeTransportRequest, r as PrimeBridgeTransportResult, t as PrimeBridgeTransport } from "../prime-bridge-transport-6feEglLf.js";
|
|
14
|
+
import { z } from "zod";
|
|
12
15
|
//#region src/analyst/adapters.d.ts
|
|
13
16
|
declare function liftSeverity(s: Severity): AnalystSeverity;
|
|
14
17
|
interface VerifierAdapterOpts<Env> {
|
|
@@ -485,6 +488,320 @@ declare function preparePublicAnalystBenchmark(options: {
|
|
|
485
488
|
seed: number;
|
|
486
489
|
}): Promise<PreparedPublicAnalystBenchmark>;
|
|
487
490
|
//#endregion
|
|
491
|
+
//#region src/analyst/reply-contract.d.ts
|
|
492
|
+
/**
|
|
493
|
+
* The reply grammar an analyst arm holds a model to, independent of transport.
|
|
494
|
+
*
|
|
495
|
+
* One contract serves every arm shape: the inline bridge protocol
|
|
496
|
+
* (`runPrimeExchange`) reads the base fields, and one-shot JSON arms
|
|
497
|
+
* additionally use the strict-envelope and all-rejected knobs. `PrimeReplyContract`
|
|
498
|
+
* in ./prime-protocol is a type alias of this contract, so a consumer written
|
|
499
|
+
* against the prime protocol names the same grammar object.
|
|
500
|
+
*/
|
|
501
|
+
/** Verdict of decoding one raw reply row. */
|
|
502
|
+
type ReplyRowDecoded<TRow> = {
|
|
503
|
+
ok: true;
|
|
504
|
+
row: TRow;
|
|
505
|
+
} | {
|
|
506
|
+
ok: false;
|
|
507
|
+
reason: string;
|
|
508
|
+
};
|
|
509
|
+
/** A strictly parsed reply envelope: the row array plus retained extras. */
|
|
510
|
+
interface ReplyEnvelope {
|
|
511
|
+
rows: readonly unknown[];
|
|
512
|
+
/** Envelope fields beside the rows the arm retains (e.g. a `report` string). */
|
|
513
|
+
extras: Record<string, unknown>;
|
|
514
|
+
}
|
|
515
|
+
interface ReplyContract<TRow> {
|
|
516
|
+
/** Name of the reply's row array, e.g. `blocks` or `findings`. */
|
|
517
|
+
rowsField: string;
|
|
518
|
+
/** Row-grammar lines spliced into the first-turn prompt. */
|
|
519
|
+
contractLines: readonly string[];
|
|
520
|
+
/**
|
|
521
|
+
* The same grammar restated for the repair turn, which never carries the
|
|
522
|
+
* evidence. Empty when the arm declares zero repair turns.
|
|
523
|
+
*/
|
|
524
|
+
repairContractLines: readonly string[];
|
|
525
|
+
/**
|
|
526
|
+
* Validate and convert one raw row in a single pass. One function rather than
|
|
527
|
+
* a separate validator and constructor, so the two can never disagree about
|
|
528
|
+
* what a valid row is.
|
|
529
|
+
*/
|
|
530
|
+
decodeRow(row: unknown, index: number): ReplyRowDecoded<TRow>;
|
|
531
|
+
/**
|
|
532
|
+
* Cap on ACCEPTED rows. The cap is applied after decoding, so malformed rows
|
|
533
|
+
* never consume an accepted slot; the surplus is reported as `overflow`.
|
|
534
|
+
* Omit when the consumer's own expansion owns the cap and records the drop.
|
|
535
|
+
*/
|
|
536
|
+
maxRows?: number;
|
|
537
|
+
/**
|
|
538
|
+
* Strict envelope parse for one-shot JSON arms. Throws on an envelope defect
|
|
539
|
+
* (missing report, unknown key, over-cap row count). When absent, rows are
|
|
540
|
+
* read from `rowsField` on the parsed reply object.
|
|
541
|
+
*/
|
|
542
|
+
parseEnvelope?(value: unknown): ReplyEnvelope;
|
|
543
|
+
/**
|
|
544
|
+
* What a reply whose EVERY reported row was rejected means. `fail` throws a
|
|
545
|
+
* validation error carrying each rejection; `accept-empty` (the default)
|
|
546
|
+
* keeps the honest empty result beside the recorded rejections.
|
|
547
|
+
*/
|
|
548
|
+
whenAllRowsRejected?: 'fail' | 'accept-empty';
|
|
549
|
+
/**
|
|
550
|
+
* Message prefix for the `fail` case, in the contract's own row vocabulary
|
|
551
|
+
* (e.g. "every reported failure block was malformed").
|
|
552
|
+
*/
|
|
553
|
+
allRejectedMessage?: string;
|
|
554
|
+
}
|
|
555
|
+
/** One rejected row, indexed as the model reported it. */
|
|
556
|
+
interface ReplyRowRejection {
|
|
557
|
+
index: number;
|
|
558
|
+
reason: string;
|
|
559
|
+
}
|
|
560
|
+
interface DecodedReply<TRow> {
|
|
561
|
+
rows: TRow[];
|
|
562
|
+
/** Envelope fields beside the rows (empty when the contract has no envelope). */
|
|
563
|
+
extras: Record<string, unknown>;
|
|
564
|
+
rejected: ReplyRowRejection[];
|
|
565
|
+
/** Rows the model reported, before decoding or the count cap. */
|
|
566
|
+
reportedRows: number;
|
|
567
|
+
/** Valid rows dropped by `maxRows`. */
|
|
568
|
+
overflow: number;
|
|
569
|
+
}
|
|
570
|
+
/**
|
|
571
|
+
* Decode a parsed reply value under a contract: strict envelope when declared,
|
|
572
|
+
* then per-row decoding, then the all-rejected policy. Shape before count: a
|
|
573
|
+
* malformed row never consumes an accepted slot.
|
|
574
|
+
*/
|
|
575
|
+
declare function decodeReplyRows<TRow>(contract: ReplyContract<TRow>, value: unknown): DecodedReply<TRow>;
|
|
576
|
+
//#endregion
|
|
577
|
+
//#region src/analyst/definition.d.ts
|
|
578
|
+
/**
|
|
579
|
+
* The slice of the canonical `AgentProfile` an analyst definition carries:
|
|
580
|
+
* model hints (pinned model, reasoning effort) and prompt shaping. Transport
|
|
581
|
+
* bindings that own model selection leave `model.default` unset.
|
|
582
|
+
*/
|
|
583
|
+
type AnalystProfileFragment = Pick<AgentProfile, 'model' | 'prompt'>;
|
|
584
|
+
/**
|
|
585
|
+
* How evidence reaches the model. This is the declared affordance axis of an
|
|
586
|
+
* arm: two arms answering the same question through different projections are
|
|
587
|
+
* comparable only with the difference rendered, never silently.
|
|
588
|
+
*/
|
|
589
|
+
type EvidenceProjection = {
|
|
590
|
+
readonly mode: 'inline';
|
|
591
|
+
/** Ceiling on serialized evidence characters embedded in one prompt. */
|
|
592
|
+
readonly maxInlineChars: number;
|
|
593
|
+
/**
|
|
594
|
+
* Per-attribute byte cap for the reduced refetch when the full
|
|
595
|
+
* projection is oversized. Still oversized after the refetch = refusal,
|
|
596
|
+
* never a silent truncation.
|
|
597
|
+
*/
|
|
598
|
+
readonly cappedAttributeBytes: number;
|
|
599
|
+
} | {
|
|
600
|
+
readonly mode: 'chunked';
|
|
601
|
+
/** Descending per-attribute byte caps tried until the store yields a projection. */
|
|
602
|
+
readonly attributeByteCaps: readonly number[];
|
|
603
|
+
} | {
|
|
604
|
+
/** Evidence bound as an engine REPL variable, read through bounded trace tools. */
|
|
605
|
+
readonly mode: 'repl-variable';
|
|
606
|
+
readonly toolGroup: TraceToolGroupName;
|
|
607
|
+
} | {
|
|
608
|
+
/** Evidence read through agent tool calls only; no REPL. */
|
|
609
|
+
readonly mode: 'agent-tools';
|
|
610
|
+
readonly toolGroup: TraceToolGroupName;
|
|
611
|
+
};
|
|
612
|
+
interface AnalystBudgetDeclaration {
|
|
613
|
+
/** Deadline for one model exchange. */
|
|
614
|
+
readonly timeoutMs: number;
|
|
615
|
+
/** Provider spend ceiling for one analysis, when the transport meters cost. */
|
|
616
|
+
readonly maxCostUsd?: number;
|
|
617
|
+
/** Completion-token cap per model call, when the transport enforces one. */
|
|
618
|
+
readonly maxOutputTokens?: number;
|
|
619
|
+
/** Recursive-engine iteration limits (repl-variable / agent-tools projections). */
|
|
620
|
+
readonly engineLimits?: TraceAnalystLimits;
|
|
621
|
+
}
|
|
622
|
+
interface AnalystRepairDeclaration {
|
|
623
|
+
/**
|
|
624
|
+
* Bounded retries a structurally malformed reply earns. Compared definitions
|
|
625
|
+
* must declare the same number: a retry is a second sample.
|
|
626
|
+
*/
|
|
627
|
+
readonly turns: number;
|
|
628
|
+
}
|
|
629
|
+
interface AnalystRowExpansion {
|
|
630
|
+
findings: AnalystFinding[];
|
|
631
|
+
/** Arm-specific expansion diagnostics recorded in observation metadata. */
|
|
632
|
+
diagnostics?: unknown;
|
|
633
|
+
}
|
|
634
|
+
interface ExpandRowsArgs<TRow> {
|
|
635
|
+
/** Evidence subject the case names (e.g. a trajectory id). */
|
|
636
|
+
subject: string;
|
|
637
|
+
rows: readonly TRow[];
|
|
638
|
+
store: TraceAnalysisStore;
|
|
639
|
+
analystId: string;
|
|
640
|
+
producedAt?: string;
|
|
641
|
+
/** Model the provider reported serving, when the transport captures it. */
|
|
642
|
+
providerModel?: string;
|
|
643
|
+
signal?: AbortSignal;
|
|
644
|
+
}
|
|
645
|
+
/** Ports an inline-projection arm binds: prompt framing plus row expansion. */
|
|
646
|
+
interface InlineEvidenceBinding<TRow> {
|
|
647
|
+
readonly kind: 'inline';
|
|
648
|
+
subjectFromCaseId(caseId: string): string;
|
|
649
|
+
/** Base observation metadata (analysis mode, engine label). */
|
|
650
|
+
readonly baseMetadata: Readonly<Record<string, unknown>>;
|
|
651
|
+
/**
|
|
652
|
+
* Line introducing the inlined evidence. Throws when the projected spans
|
|
653
|
+
* cannot ground the question (e.g. no assistant step spans).
|
|
654
|
+
*/
|
|
655
|
+
header(subject: string, spans: readonly TraceAnalystSpan[]): string;
|
|
656
|
+
/** Material appended after the evidence. */
|
|
657
|
+
trailer(subject: string, spans: readonly TraceAnalystSpan[]): string;
|
|
658
|
+
expandRows(args: ExpandRowsArgs<TRow>): Promise<AnalystRowExpansion>;
|
|
659
|
+
}
|
|
660
|
+
/** Ports a chunked-projection one-shot arm binds. */
|
|
661
|
+
interface ChunkedEvidenceBinding<TRow> {
|
|
662
|
+
readonly kind: 'chunked';
|
|
663
|
+
subjectFromCaseId(caseId: string): string;
|
|
664
|
+
readonly baseMetadata: Readonly<Record<string, unknown>>;
|
|
665
|
+
/** Actor name paid calls are attributed to in the cost ledger. */
|
|
666
|
+
readonly costActor: string;
|
|
667
|
+
/** Cost-ledger phase paid calls settle under. */
|
|
668
|
+
readonly costPhase: string;
|
|
669
|
+
/** Compose the user message around the rendered evidence. */
|
|
670
|
+
userMessage(rendered: string): string;
|
|
671
|
+
expandRows(args: ExpandRowsArgs<TRow>): Promise<AnalystRowExpansion>;
|
|
672
|
+
/** Ground accepted findings against the store; throws on unresolvable evidence. */
|
|
673
|
+
verifyFindings?(args: {
|
|
674
|
+
subject: string;
|
|
675
|
+
findings: readonly AnalystFinding[];
|
|
676
|
+
store: TraceAnalysisStore;
|
|
677
|
+
signal?: AbortSignal;
|
|
678
|
+
}): Promise<void>;
|
|
679
|
+
}
|
|
680
|
+
/** Majority-vote ports for a multi-sample repl-variable arm. */
|
|
681
|
+
interface ReplVariableConsensusPort<TAssignment, TBlock> {
|
|
682
|
+
/** Vote across per-sample assignments; returns voted blocks plus the decision record. */
|
|
683
|
+
vote(samples: ReadonlyArray<readonly TAssignment[]>): {
|
|
684
|
+
blocks: readonly TBlock[];
|
|
685
|
+
decision: unknown;
|
|
686
|
+
};
|
|
687
|
+
/** Expand voted blocks into findings grounded in the store. */
|
|
688
|
+
expand(args: {
|
|
689
|
+
subject: string;
|
|
690
|
+
blocks: readonly TBlock[];
|
|
691
|
+
store: TraceAnalysisStore;
|
|
692
|
+
analystId: string;
|
|
693
|
+
producedAt: string;
|
|
694
|
+
signal?: AbortSignal;
|
|
695
|
+
}): Promise<AnalystRowExpansion>;
|
|
696
|
+
/** Per-sample observation record (accepted blocks, member steps). */
|
|
697
|
+
sampleRecord(assignments: readonly TAssignment[]): Record<string, unknown>;
|
|
698
|
+
}
|
|
699
|
+
/** Ports a repl-variable (recursive engine) arm binds. */
|
|
700
|
+
interface ReplVariableEvidenceBinding<TAssignment = unknown, TBlock = unknown> {
|
|
701
|
+
readonly kind: 'repl-variable';
|
|
702
|
+
/** Identity of the trace-analyst definition the engine runs. */
|
|
703
|
+
readonly traceAnalystId: string;
|
|
704
|
+
subjectFromCaseId(caseId: string): string;
|
|
705
|
+
readonly baseMetadata: Readonly<Record<string, unknown>>;
|
|
706
|
+
/** Metadata stamped on every finding the arm emits. */
|
|
707
|
+
readonly findingBaseMetadata: Readonly<Record<string, unknown>>;
|
|
708
|
+
/** Cost-ledger phase paid calls settle under. */
|
|
709
|
+
readonly costPhase: string;
|
|
710
|
+
/** Row metadata derived from the finding's subject grammar. */
|
|
711
|
+
metadataFromSubject?(subject: string | undefined): Record<string, unknown> | undefined;
|
|
712
|
+
/** Map raw engine rows into scored findings grounded in the store. */
|
|
713
|
+
adapt(args: {
|
|
714
|
+
subject: string;
|
|
715
|
+
findings: readonly AnalystFinding[];
|
|
716
|
+
analystId: string;
|
|
717
|
+
store: TraceAnalysisStore;
|
|
718
|
+
signal?: AbortSignal;
|
|
719
|
+
}): Promise<{
|
|
720
|
+
findings: AnalystFinding[];
|
|
721
|
+
stepBlocks?: TAssignment[];
|
|
722
|
+
diagnostics?: unknown;
|
|
723
|
+
}>;
|
|
724
|
+
/** Multi-sample majority consensus; required when the arm runs samples > 1. */
|
|
725
|
+
consensus?: ReplVariableConsensusPort<TAssignment, TBlock>;
|
|
726
|
+
/** Second-opinion arm invoked when the engine submits no finding at all. */
|
|
727
|
+
abstentionFallback(config: PublicAnalystBenchmarkModelConfig): AnalystBenchmarkRunner<AnalystRunInputs>;
|
|
728
|
+
}
|
|
729
|
+
type AnalystEvidenceBinding<TRow, TAssignment = unknown, TBlock = unknown> = InlineEvidenceBinding<TRow> | ChunkedEvidenceBinding<TRow> | ReplVariableEvidenceBinding<TAssignment, TBlock>;
|
|
730
|
+
interface AnalystDefinition<TRow = unknown, TAssignment = unknown, TBlock = unknown> {
|
|
731
|
+
/** Arm identity — appears as the runner id and in every finding. */
|
|
732
|
+
readonly id: string;
|
|
733
|
+
readonly description: string;
|
|
734
|
+
readonly version: string;
|
|
735
|
+
/** Finding area the arm's expansion stamps, when uniform per arm. */
|
|
736
|
+
readonly area?: string;
|
|
737
|
+
readonly profile: AnalystProfileFragment;
|
|
738
|
+
/** User-facing question. Empty when the task text is the whole ask. */
|
|
739
|
+
readonly question: string;
|
|
740
|
+
/** Task definition / instruction text sent beside the question. */
|
|
741
|
+
readonly taskDefinition?: string;
|
|
742
|
+
readonly projection: EvidenceProjection;
|
|
743
|
+
readonly replyContract: ReplyContract<TRow>;
|
|
744
|
+
/**
|
|
745
|
+
* Numeric limits the contract states (row caps, width caps). They enter the
|
|
746
|
+
* protocol digest; insertion order is digest-bearing because the digest
|
|
747
|
+
* serializes with `JSON.stringify`.
|
|
748
|
+
*/
|
|
749
|
+
readonly contractLimits: Readonly<Record<string, number>>;
|
|
750
|
+
readonly budget: AnalystBudgetDeclaration;
|
|
751
|
+
readonly repair: AnalystRepairDeclaration;
|
|
752
|
+
/**
|
|
753
|
+
* Digest the bound arm stamps on observations. For an inline definition this
|
|
754
|
+
* equals `analystDefinitionProtocolSha256`; benchmark arms that record a
|
|
755
|
+
* shared dataset-level digest carry that digest here instead.
|
|
756
|
+
*/
|
|
757
|
+
readonly protocolSha256: string;
|
|
758
|
+
readonly binding: AnalystEvidenceBinding<TRow, TAssignment, TBlock>;
|
|
759
|
+
}
|
|
760
|
+
/**
|
|
761
|
+
* Thrown at bind time when a definition asks for something no strategy can
|
|
762
|
+
* compile — an unknown projection × transport pair, a repair-turn count the
|
|
763
|
+
* exchange machinery cannot grant, a reasoning effort the arm cannot map. The
|
|
764
|
+
* message names the construct so an expressiveness gap is a loud, attributable
|
|
765
|
+
* failure instead of a silently narrowed protocol.
|
|
766
|
+
*/
|
|
767
|
+
declare class AnalystExpressivenessError extends Error {}
|
|
768
|
+
/**
|
|
769
|
+
* Digest of everything a definition can send to its model. An inline
|
|
770
|
+
* definition hashes under the historical prime-protocol domain, so its digest
|
|
771
|
+
* equals the digest its bespoke arm always recorded; other projections hash
|
|
772
|
+
* under the definition domain.
|
|
773
|
+
*/
|
|
774
|
+
declare function analystDefinitionProtocolSha256<TRow, TAssignment, TBlock>(definition: AnalystDefinition<TRow, TAssignment, TBlock>): string;
|
|
775
|
+
/** One definition's declared difference from the compared set. */
|
|
776
|
+
interface AnalystDefinitionAsymmetry {
|
|
777
|
+
readonly id: string;
|
|
778
|
+
readonly projectionMode: EvidenceProjection['mode'];
|
|
779
|
+
readonly reasoningEffort: NonNullable<AgentProfile['model']>['reasoningEffort'] | null;
|
|
780
|
+
readonly timeoutMs: number;
|
|
781
|
+
readonly maxCostUsd: number | null;
|
|
782
|
+
readonly maxOutputTokens: number | null;
|
|
783
|
+
/** The digest the arm records on observations. */
|
|
784
|
+
readonly protocolSha256: string;
|
|
785
|
+
/** The definition's own protocol identity. */
|
|
786
|
+
readonly definitionSha256: string;
|
|
787
|
+
}
|
|
788
|
+
interface AnalystDefinitionAsymmetryReport {
|
|
789
|
+
readonly ids: readonly string[];
|
|
790
|
+
/** Repair turns every compared definition declares. */
|
|
791
|
+
readonly repairTurns: number;
|
|
792
|
+
/** The one projection mode all definitions share, or null when they differ. */
|
|
793
|
+
readonly sharedProjectionMode: EvidenceProjection['mode'] | null;
|
|
794
|
+
readonly asymmetries: readonly AnalystDefinitionAsymmetry[];
|
|
795
|
+
}
|
|
796
|
+
/**
|
|
797
|
+
* Refuse a set of definitions that cannot be compared on equal terms, and
|
|
798
|
+
* render what still differs between the ones that can. The hard rule is the
|
|
799
|
+
* repair turn: a malformed reply must earn the same number of retries in every
|
|
800
|
+
* arm, because a retry is a second sample. Projection, reasoning effort, and
|
|
801
|
+
* budget differences are declared and reported, never hidden.
|
|
802
|
+
*/
|
|
803
|
+
declare function analystDefinitionAsymmetries(definitions: ReadonlyArray<AnalystDefinition<unknown, unknown, unknown>>): AnalystDefinitionAsymmetryReport;
|
|
804
|
+
//#endregion
|
|
488
805
|
//#region src/analyst/benchmark-public-prompt.d.ts
|
|
489
806
|
/** Widest contiguous failure block a model may report. The published corpus's
|
|
490
807
|
* widest labeled block is 8 steps and its widest stage span is 9, so this bound
|
|
@@ -506,59 +823,111 @@ declare function publicBenchmarkRlmInstructions(dataset: PublicAnalystBenchmarkD
|
|
|
506
823
|
declare function publicBenchmarkProtocolSha256(dataset: PublicAnalystBenchmarkDataset): string;
|
|
507
824
|
//#endregion
|
|
508
825
|
//#region src/analyst/benchmark-public-model.d.ts
|
|
509
|
-
/**
|
|
826
|
+
/**
|
|
827
|
+
* One-shot JSON baseline arm. Not a recursive trace analyst.
|
|
828
|
+
*
|
|
829
|
+
* The arm is expressed as an `AnalystDefinition`
|
|
830
|
+
* (`publicDirectAnalystDefinition`): the task text, field and envelope
|
|
831
|
+
* contracts, the descending projection ladder, and the zero-repair declaration
|
|
832
|
+
* are definition content, and `createPublicBenchmarkDirectRunner` is a thin
|
|
833
|
+
* shell that builds the definition and runs it through the chunked strategy
|
|
834
|
+
* below — the same strategy `bindAnalyst` (./bind) dispatches to.
|
|
835
|
+
*/
|
|
836
|
+
interface PublicDirectDefinitionArgs {
|
|
837
|
+
/** Whole-analysis deadline (`config.timeoutMs`). */
|
|
838
|
+
timeoutMs: number;
|
|
839
|
+
/** Completion-token cap per call (`config.maxOutputTokens`). */
|
|
840
|
+
maxOutputTokens: number;
|
|
841
|
+
/** Per-case provider spend ceiling (`config.maxCostUsdPerAnalysis`). */
|
|
842
|
+
maxCostUsd: number;
|
|
843
|
+
}
|
|
844
|
+
/** The direct arm as a declarative unit for one public dataset. */
|
|
845
|
+
declare function publicDirectAnalystDefinition(dataset: PublicAnalystBenchmarkDataset, args: PublicDirectDefinitionArgs): AnalystDefinition<PublicBenchmarkModelPrediction>;
|
|
846
|
+
/** Thin shell: validate config, declare the definition, run the chunked strategy. */
|
|
510
847
|
declare function createPublicBenchmarkDirectRunner(dataset: PublicAnalystBenchmarkDataset, config: PublicAnalystBenchmarkModelConfig): AnalystBenchmarkRunner<AnalystRunInputs>;
|
|
848
|
+
declare const AgentRxPredictionSchema: z.ZodObject<{
|
|
849
|
+
step: z.ZodNumber;
|
|
850
|
+
severity: z.ZodEnum<{
|
|
851
|
+
critical: "critical";
|
|
852
|
+
high: "high";
|
|
853
|
+
info: "info";
|
|
854
|
+
low: "low";
|
|
855
|
+
medium: "medium";
|
|
856
|
+
}>;
|
|
857
|
+
claim: z.ZodString;
|
|
858
|
+
confidence: z.ZodNumber;
|
|
859
|
+
rationale: z.ZodOptional<z.ZodString>;
|
|
860
|
+
recommended_action: z.ZodOptional<z.ZodString>;
|
|
861
|
+
}, z.core.$strict>;
|
|
862
|
+
declare const CodeTraceBlockPredictionSchema: z.ZodObject<{
|
|
863
|
+
first_step: z.ZodNumber;
|
|
864
|
+
last_step: z.ZodNumber;
|
|
865
|
+
consequence_step: z.ZodNumber;
|
|
866
|
+
escape_status: z.ZodEnum<{
|
|
867
|
+
escaped: "escaped";
|
|
868
|
+
unescaped: "unescaped";
|
|
869
|
+
}>;
|
|
870
|
+
severity: z.ZodEnum<{
|
|
871
|
+
critical: "critical";
|
|
872
|
+
high: "high";
|
|
873
|
+
info: "info";
|
|
874
|
+
low: "low";
|
|
875
|
+
medium: "medium";
|
|
876
|
+
}>;
|
|
877
|
+
claim: z.ZodString;
|
|
878
|
+
confidence: z.ZodNumber;
|
|
879
|
+
rationale: z.ZodOptional<z.ZodString>;
|
|
880
|
+
recommended_action: z.ZodOptional<z.ZodString>;
|
|
881
|
+
}, z.core.$strict>;
|
|
882
|
+
declare const AgentRxCategorySchema: z.ZodEnum<{
|
|
883
|
+
"guardrails-triggered": "guardrails-triggered";
|
|
884
|
+
inconclusive: "inconclusive";
|
|
885
|
+
"instruction-plan-adherence-failure": "instruction-plan-adherence-failure";
|
|
886
|
+
"intent-not-supported": "intent-not-supported";
|
|
887
|
+
"intent-plan-misalignment": "intent-plan-misalignment";
|
|
888
|
+
"invalid-invocation": "invalid-invocation";
|
|
889
|
+
"invention-of-new-information": "invention-of-new-information";
|
|
890
|
+
"misinterpretation-of-tool-output-handoff-failure": "misinterpretation-of-tool-output-handoff-failure";
|
|
891
|
+
"system-failure": "system-failure";
|
|
892
|
+
"underspecified-user-intent": "underspecified-user-intent";
|
|
893
|
+
}>;
|
|
894
|
+
type AgentRxModelPrediction = z.infer<typeof AgentRxPredictionSchema> & {
|
|
895
|
+
category?: z.infer<typeof AgentRxCategorySchema>;
|
|
896
|
+
};
|
|
897
|
+
type CodeTraceModelPrediction = z.infer<typeof CodeTraceBlockPredictionSchema>;
|
|
898
|
+
type PublicBenchmarkModelPrediction = AgentRxModelPrediction | CodeTraceModelPrediction;
|
|
511
899
|
//#endregion
|
|
512
900
|
//#region src/analyst/benchmark-public-rlm.d.ts
|
|
513
|
-
/** Public benchmark candidate that runs the actual recursive trace analyst. */
|
|
514
|
-
declare function createPublicBenchmarkRlmRunner(dataset: PublicAnalystBenchmarkDataset, config: PublicAnalystBenchmarkModelConfig): AnalystBenchmarkRunner<AnalystRunInputs>;
|
|
515
|
-
//#endregion
|
|
516
|
-
//#region src/analyst/prime-bridge-transport.d.ts
|
|
517
901
|
/**
|
|
518
|
-
*
|
|
519
|
-
* OpenAI-compatible `/v1/chat/completions` endpoint, returning the raw status
|
|
520
|
-
* and body text.
|
|
902
|
+
* Public benchmark candidate that runs the actual recursive trace analyst.
|
|
521
903
|
*
|
|
522
|
-
*
|
|
523
|
-
*
|
|
524
|
-
*
|
|
525
|
-
*
|
|
526
|
-
*
|
|
527
|
-
*
|
|
528
|
-
* - it composes sampling and response-format options the bridge's CLI backends
|
|
529
|
-
* reject;
|
|
530
|
-
* - it collapses HTTP status, unparseable body, and empty content into two
|
|
531
|
-
* error classes, while the protocol classifies them as three distinct
|
|
532
|
-
* terminal reasons.
|
|
904
|
+
* The arm is expressed as an `AnalystDefinition` (`publicRlmAnalystDefinition`):
|
|
905
|
+
* the question, the recursive instructions (stock or override), the tool group,
|
|
906
|
+
* the engine iteration limits, and the budget are definition content, and
|
|
907
|
+
* `createPublicBenchmarkRlmRunner` is a thin shell that builds the definition
|
|
908
|
+
* and runs it through the repl-variable strategy below — the same strategy
|
|
909
|
+
* `bindAnalyst` (./bind) dispatches to.
|
|
533
910
|
*/
|
|
534
|
-
interface
|
|
535
|
-
|
|
536
|
-
|
|
537
|
-
|
|
538
|
-
|
|
539
|
-
|
|
540
|
-
|
|
541
|
-
|
|
542
|
-
|
|
543
|
-
/**
|
|
544
|
-
|
|
545
|
-
|
|
546
|
-
|
|
547
|
-
*/
|
|
548
|
-
signal: AbortSignal;
|
|
549
|
-
}
|
|
550
|
-
interface PrimeBridgeTransportResult {
|
|
551
|
-
status: number;
|
|
552
|
-
text: string;
|
|
911
|
+
interface PublicRlmDefinitionArgs {
|
|
912
|
+
/** Effective recursive instructions: the override text or the stock prompt. */
|
|
913
|
+
instructions: string;
|
|
914
|
+
/** Digest the arm records: the stock digest, bound to any override. */
|
|
915
|
+
protocolSha256: string;
|
|
916
|
+
/** Whole-analysis deadline (`config.timeoutMs`). */
|
|
917
|
+
timeoutMs: number;
|
|
918
|
+
/** Controller completion-token cap (`config.maxOutputTokens`). */
|
|
919
|
+
maxOutputTokens: number;
|
|
920
|
+
/** Per-case engine spend ceiling (`config.maxCostUsdPerAnalysis`). */
|
|
921
|
+
maxCostUsd: number;
|
|
922
|
+
/** Resolved recursive-engine iteration limits. */
|
|
923
|
+
engineLimits: TraceAnalystLimits;
|
|
553
924
|
}
|
|
554
|
-
/**
|
|
555
|
-
|
|
556
|
-
/**
|
|
557
|
-
|
|
558
|
-
|
|
559
|
-
|
|
560
|
-
*/
|
|
561
|
-
declare function nodeHttpPrimeBridgeTransport(): PrimeBridgeTransport;
|
|
925
|
+
/** The dspy-rlm arm as a declarative unit for one public dataset. */
|
|
926
|
+
declare function publicRlmAnalystDefinition(dataset: PublicAnalystBenchmarkDataset, args: PublicRlmDefinitionArgs): AnalystDefinition<RawAnalystFinding, CodeTraceStepAssignment>;
|
|
927
|
+
/** Thin shell: validate config, declare the definition, run the repl-variable strategy. */
|
|
928
|
+
declare function createPublicBenchmarkRlmRunner(dataset: PublicAnalystBenchmarkDataset, config: PublicAnalystBenchmarkModelConfig): AnalystBenchmarkRunner<AnalystRunInputs>;
|
|
929
|
+
/** Engine iteration limits with this arm's defaults applied. */
|
|
930
|
+
declare function rlmEngineLimits(config: PublicAnalystBenchmarkModelConfig): TraceAnalystLimits;
|
|
562
931
|
//#endregion
|
|
563
932
|
//#region src/analyst/benchmark-runner-prime.d.ts
|
|
564
933
|
interface PrimeBenchmarkRunnerOptions {
|
|
@@ -576,11 +945,22 @@ interface PrimeBenchmarkRunnerOptions {
|
|
|
576
945
|
transport?: PrimeBridgeTransport;
|
|
577
946
|
}
|
|
578
947
|
/**
|
|
579
|
-
* Digest of everything this
|
|
948
|
+
* Digest of everything this arm can send to the bridge, recorded per
|
|
580
949
|
* observation so a prime result names the exact contract that produced it.
|
|
581
950
|
*/
|
|
582
951
|
declare function primeAnalystProtocolSha256(): string;
|
|
583
|
-
|
|
952
|
+
interface PrimeCodeTraceDefinitionArgs {
|
|
953
|
+
/** Deadline for one bridge call. */
|
|
954
|
+
timeoutMs: number;
|
|
955
|
+
/** 1 grants the bounded repair turn; 0 disables it. */
|
|
956
|
+
repairTurns: number;
|
|
957
|
+
}
|
|
958
|
+
/**
|
|
959
|
+
* The prime arm as a declarative unit. CodeTraceBench-only: the question,
|
|
960
|
+
* task text, and block grammar speak its incorrect-step definition.
|
|
961
|
+
*/
|
|
962
|
+
declare function primeCodeTraceAnalystDefinition(args: PrimeCodeTraceDefinitionArgs): AnalystDefinition<CodeTraceFailureBlock>;
|
|
963
|
+
/** Thin shell: validate options, declare the definition, run the inline strategy. */
|
|
584
964
|
declare function createPrimeBenchmarkRunner(options: PrimeBenchmarkRunnerOptions): AnalystBenchmarkRunner<AnalystRunInputs>;
|
|
585
965
|
//#endregion
|
|
586
966
|
//#region src/analyst/benchmark-comparison.d.ts
|
|
@@ -894,7 +1274,7 @@ declare const ANALYST_BENCHMARK_HELP = "agent-eval analyst-benchmark\n\nRun the
|
|
|
894
1274
|
declare const ANALYST_BENCHMARK_IMPLEMENTATION_DIGEST_ALGORITHM = "sha256-canonical-source-manifest";
|
|
895
1275
|
declare const ANALYST_BENCHMARK_DEPENDENCY_LOCK_DIGEST_ALGORITHM = "sha256-canonical-file-manifest";
|
|
896
1276
|
declare const ANALYST_BENCHMARK_DEPENDENCY_LOCK_FILES: readonly string[];
|
|
897
|
-
declare const ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256 = "
|
|
1277
|
+
declare const ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256 = "96a42e1ad7cc0c9b00b2430ae09092c14bd5de385c696e849cb50507b9d77f02";
|
|
898
1278
|
/** The published benchmark evidence was produced at this package version, by
|
|
899
1279
|
* the retired one-shot direct runner, before trace analysts moved to the
|
|
900
1280
|
* recursive DSPy RLM engine. Both evidence digests below are historical facts
|
|
@@ -906,7 +1286,7 @@ declare const ANALYST_BENCHMARK_EVIDENCE_PACKAGE_VERSION = "0.137.0";
|
|
|
906
1286
|
declare const ANALYST_BENCHMARK_EVIDENCE_DEPENDENCY_LOCK_SHA256 = "1e03f2daed356d60316aabefb407ec1e437ac94d408d61eea4ae096e9c6fbb5b";
|
|
907
1287
|
declare const ANALYST_BENCHMARK_EVIDENCE_IMPLEMENTATION_SHA256 = "4dba263b6256a30d56c7fdb2d992d3a953c0035d731f359b704db806f68f75ac";
|
|
908
1288
|
declare const ANALYST_BENCHMARK_IMPLEMENTATION_FILES: readonly string[];
|
|
909
|
-
declare const ANALYST_BENCHMARK_IMPLEMENTATION_SHA256 = "
|
|
1289
|
+
declare const ANALYST_BENCHMARK_IMPLEMENTATION_SHA256 = "cc354effd79c8dfc8669c230706c67c63a06e3aff013f540806361263bae8860";
|
|
910
1290
|
declare function analystBenchmarkImplementationDigest(): string;
|
|
911
1291
|
declare function analystBenchmarkDependencyLockDigest(): string;
|
|
912
1292
|
//#endregion
|
|
@@ -933,6 +1313,26 @@ declare function renderAnalystBenchmarkMarkdown(result: AnalystBenchmarkResult,
|
|
|
933
1313
|
//#region src/analyst/benchmark-summary.d.ts
|
|
934
1314
|
declare function summarizeAnalystBenchmarkRunner(runnerId: string, observations: readonly AnalystBenchmarkObservation[]): AnalystBenchmarkSummary;
|
|
935
1315
|
//#endregion
|
|
1316
|
+
//#region src/analyst/bind.d.ts
|
|
1317
|
+
/** Execution half of a binding: who actually reaches the model. */
|
|
1318
|
+
type AnalystTransportBinding = {
|
|
1319
|
+
/** An OpenAI-compatible cli-bridge endpoint (inline projections). */
|
|
1320
|
+
readonly kind: 'prime-bridge';
|
|
1321
|
+
readonly baseUrl: string;
|
|
1322
|
+
readonly model: string;
|
|
1323
|
+
readonly transport?: PrimeBridgeTransport;
|
|
1324
|
+
readonly pricing?: CustomTokenPricing;
|
|
1325
|
+
} | {
|
|
1326
|
+
/** The caller-owned model path (chunked and repl-variable projections). */
|
|
1327
|
+
readonly kind: 'model-owner';
|
|
1328
|
+
readonly config: PublicAnalystBenchmarkModelConfig;
|
|
1329
|
+
};
|
|
1330
|
+
/**
|
|
1331
|
+
* Compile a definition plus a transport binding into a runnable arm. Dispatch
|
|
1332
|
+
* is total over the expressible pairs; everything else is a loud refusal.
|
|
1333
|
+
*/
|
|
1334
|
+
declare function bindAnalyst<TRow, TAssignment, TBlock>(definition: AnalystDefinition<TRow, TAssignment, TBlock>, transports: AnalystTransportBinding): AnalystBenchmarkRunner<AnalystRunInputs>;
|
|
1335
|
+
//#endregion
|
|
936
1336
|
//#region src/analyst/parse-tolerant.d.ts
|
|
937
1337
|
/**
|
|
938
1338
|
* Forgiving pre-parse for analyst findings. Weak models routinely emit
|
|
@@ -979,33 +1379,15 @@ declare function coerceToFindingRows(raw: unknown): unknown[];
|
|
|
979
1379
|
* nothing in this file may import a block type, a finding type, or a trace
|
|
980
1380
|
* store, and no consumer needs another copy of the protocol.
|
|
981
1381
|
*/
|
|
982
|
-
|
|
983
|
-
|
|
984
|
-
|
|
985
|
-
|
|
986
|
-
|
|
987
|
-
|
|
988
|
-
|
|
989
|
-
|
|
990
|
-
|
|
991
|
-
rowsField: string;
|
|
992
|
-
/** Row-grammar lines spliced into the first-turn prompt. */
|
|
993
|
-
contractLines: readonly string[];
|
|
994
|
-
/** The same grammar restated for the repair turn, which never carries the trajectory. */
|
|
995
|
-
repairContractLines: readonly string[];
|
|
996
|
-
/**
|
|
997
|
-
* Validate and convert one raw row in a single pass. One function rather than
|
|
998
|
-
* a separate validator and constructor, so the two can never disagree about
|
|
999
|
-
* what a valid row is.
|
|
1000
|
-
*/
|
|
1001
|
-
decodeRow(row: unknown, index: number): PrimeRowDecoded<TRow>;
|
|
1002
|
-
/**
|
|
1003
|
-
* Cap on ACCEPTED rows. The cap is applied after decoding, so malformed rows
|
|
1004
|
-
* never consume an accepted slot; the surplus is reported as `overflow`.
|
|
1005
|
-
* Omit when the consumer's own expansion owns the cap and records the drop.
|
|
1006
|
-
*/
|
|
1007
|
-
maxRows?: number;
|
|
1008
|
-
}
|
|
1382
|
+
/**
|
|
1383
|
+
* The prime reply grammar is the general analyst reply contract: these aliases
|
|
1384
|
+
* keep every prime-protocol consumer compiling against the shared type in
|
|
1385
|
+
* ./reply-contract. The exchange below reads the base fields (`rowsField`,
|
|
1386
|
+
* contract lines, `decodeRow`, `maxRows`); the strict-envelope knobs serve
|
|
1387
|
+
* one-shot JSON arms and are inert here.
|
|
1388
|
+
*/
|
|
1389
|
+
type PrimeRowDecoded<TRow> = ReplyRowDecoded<TRow>;
|
|
1390
|
+
type PrimeReplyContract<TRow> = ReplyContract<TRow>;
|
|
1009
1391
|
interface PrimePromptSpec {
|
|
1010
1392
|
question: string;
|
|
1011
1393
|
/** Task definition spliced under a `TASK DEFINITION:` heading. Omit when the question is the whole task. */
|
|
@@ -1229,5 +1611,5 @@ declare function isProposalFinding(finding: unknown): finding is ProposalFinding
|
|
|
1229
1611
|
*/
|
|
1230
1612
|
declare function assertProposalFindings(findings: unknown, context?: string): ReadonlyArray<ProposalFinding>;
|
|
1231
1613
|
//#endregion
|
|
1232
|
-
export { AGENT_RX_UPSTREAM_REVISION, ANALYST_BENCHMARK_COST_LEDGER_FILE, ANALYST_BENCHMARK_DEPENDENCY_LOCK_DIGEST_ALGORITHM, ANALYST_BENCHMARK_DEPENDENCY_LOCK_FILES, ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256, ANALYST_BENCHMARK_EVIDENCE_DEPENDENCY_LOCK_SHA256, ANALYST_BENCHMARK_EVIDENCE_IMPLEMENTATION_SHA256, ANALYST_BENCHMARK_EVIDENCE_PACKAGE_VERSION, ANALYST_BENCHMARK_HELP, ANALYST_BENCHMARK_IMPLEMENTATION_DIGEST_ALGORITHM, ANALYST_BENCHMARK_IMPLEMENTATION_FILES, ANALYST_BENCHMARK_IMPLEMENTATION_SHA256, ANALYST_BENCHMARK_LOCAL_RECEIPT_FILE, ANALYST_BENCHMARK_MANIFEST_FILE, ANALYST_BENCHMARK_OBSERVATIONS_FILE, ANALYST_SEVERITIES, type AgentRxBenchmarkCaseOptions, type AgentRxCalibrationRunnerSummary, type AgentRxCalibrationSummary, type AgentRxFailure, type AgentRxPrediction, type AgentRxPredictionReport, type AgentRxRow, type Analyst, type AnalystBenchmarkArtifact, type AnalystBenchmarkCase, type AnalystBenchmarkCommandConfig, type AnalystBenchmarkCommandDependencies, type AnalystBenchmarkDatasetRef, type AnalystBenchmarkDescriptor, type AnalystBenchmarkError, type AnalystBenchmarkLabelState, type AnalystBenchmarkLocalRunReceipt, type AnalystBenchmarkObservation, type AnalystBenchmarkOutput, type AnalystBenchmarkProgressRow, type AnalystBenchmarkProvenance, type AnalystBenchmarkResult, type AnalystBenchmarkRunIdentity, type AnalystBenchmarkRunManifest, type AnalystBenchmarkRunner, type AnalystBenchmarkSummary, type AnalystComparisonMetric, type AnalystContext, type AnalystCost, type AnalystEvidenceExpectation, type AnalystEvidenceResolution, type AnalystEvidenceResolutionError, type AnalystEvidenceResolver, type AnalystFinding, type AnalystFindingScore, type AnalystHooks, type AnalystInputKind, type AnalystInstructionsOverride, type AnalystIssueExpectation, type AnalystLatencyDistribution, type AnalystMetricComparison, AnalystRegistry, type AnalystRegistryOptions, type AnalystRequirements, type AnalystRunEvent, type AnalystRunInputs, type AnalystRunResult, type AnalystRunSummary, type AnalystRunnerComparison, type AnalystSeverity, type AnalystUsageReceipt, type BehavioralAnalystOptions, type BudgetPolicy, CODE_TRACE_BENCH_ANALYST_PROMPT, CONTROL_INTEGRITY_ANALYST, type ChatCallOpts, type ChatClient, type ChatRequest, type ChatResponse, type ChatTransport, type CliBridgeTransportOpts, type CodeTraceBenchCaseOptions, type CodeTraceBenchLabelOptions, type CodeTraceBenchLabelSet, type CodeTraceBenchRow, type CodeTraceBlockDiagnostics, type CodeTraceCalibrationRunnerSummary, type CodeTraceCalibrationSummary, type CodeTraceFailureBlock, type CodeTraceStageAnnotation, type CodeTracerLabelGroup, type CodeTracerPredictionAdapterOptions, type CodeTracerPredictions, type CodeTracerStepLabel, ControlIntegrityAnalyst, type CreateChatClientOpts, type CreateTraceAnalystOptions, type CustomTransportOpts, DEFAULT_MAX_VERIFICATION_ARTIFACT_BYTES, DEFAULT_TRACE_ANALYST_KINDS, DEFAULT_TRACE_ANALYST_LIMITS, type DefaultAnalystRegistryOptions, type DefineCustomAnalystOptions, type DefineExactCustomAnalystOptions, type DiffPolicy, type DirectProviderTransportOpts, type DspyRlmTraceEngineOptions, type EvidenceRef, type ExactAnalystBudgetPolicy, type ExactAnalystBudgetSnapshot, type ExactAnalystExecutionPlanSnapshot, type ExactAnalystRunCompletion, type ExactAnalystRunEvent, ExactAnalystRunExecutionError, type ExactAnalystRunPolicySnapshot, type ExactAnalystRunResult, type ExactAnalystRunSummary, type ExactAnalystSnapshot, type ExactCapableAnalyst, type ExactExecutionComponentIdentity, type ExactExecutionComponentSnapshot, type ExactRegistryRunOpts, FAILURE_MODE_KIND_SPEC, FINDING_SUBJECT_GRAMMAR_PROMPT, FINDING_SUBJECT_KINDS, FINDING_SUBJECT_SYNTAX, type FindingSubject, type FindingSubjectKind, FindingSubjectStringSchema, type FindingsDiff, FindingsStore, IMPROVEMENT_KIND_SPEC, type JudgeAdapterOpts, KIND_EXPECTED_SUBJECTS, KNOWLEDGE_GAP_KIND_SPEC, KNOWLEDGE_POISONING_KIND_SPEC, type LoadedVerificationArtifacts, MAX_INCORRECT_BLOCKS, MAX_INCORRECT_BLOCK_STEPS, type MockTransportOpts, type PersistedFinding, type PreparedPublicAnalystBenchmark, type PrimeBenchmarkRunnerOptions, type PrimeBridgeTransport, type PrimeBridgeTransportRequest, type PrimeBridgeTransportResult, type PrimeExchangeOptions, type PrimeExchangeOutcome, type PrimeFailure, type PrimeProjectionDelivery, type PrimeProjectionOutcome, type PrimeProjectionSource, type PrimePromptSpec, type PrimeProtocolIdentity, type PrimeRawUsage, type PrimeRejectedRow, type PrimeRepairPromptSpec, type PrimeRepairState, type PrimeReplyContract, type PrimeRowDecoded, type PrimeTurnRecord, type ProposalFinding, type ProposalFindingOrigin, type PublicAnalystBenchmarkDataset, type PublicAnalystBenchmarkModelConfig, type PublicAnalystBenchmarkModelOwner, type PublicAnalystBenchmarkModelSettings, type PublicBenchmarkDistributions, type PublicBenchmarkSelectionReport, type PublicBenchmarkValueDistribution, RAW_FINDING_SCHEMA_PROMPT, type RawAnalystEvidence, RawAnalystEvidenceSchema, type RawAnalystFinding, RawAnalystFindingSchema, type RegistryRunOpts, type RouterTransportOpts, type RunAnalystBenchmarkOptions, type RunCriticAdapterOpts, SKILL_USAGE_ANALYST, type SandboxSdkTransportOpts, type SemanticConceptJudgeAdapterOpts, SkillUsageAnalyst, type SkillUsageRecord, type SkillUsageReport, type SkillUsageScanConfig, type StepLabelAdapterOptions, type TraceAnalysisEngine, type TraceAnalysisEngineRequest, type TraceAnalysisEngineResult, type TraceAnalystDefinition, type TraceAnalystLimits, type TraceToolGroupName, type UpstreamPredictionAdapterOptions, type VerificationArtifactFile, type VerificationArtifactManifest, type VerificationArtifactRole, type VerificationAvailabilitySummary, type VerificationOutcome, type VerificationOutcomeSource, type VerificationOutcomeStatus, type VerificationResultFile, type VerifierAdapterOpts, adaptPublicBenchmarkFindings, agentRxBenchmarkCase, agentRxPredictionsToFindings, analystBenchmarkDependencyLockDigest, analystBenchmarkImplementationDigest, analystInstructionsOverrideFromText, analystUsageReceiptFromPrimeUsage, appendVerificationArtifactsToOtlp, assertExactRegistryRunOpts, assertProposalFindings, behavioralAnalyst, buildDefaultAnalystRegistry, buildPrimePrompt, buildPrimeRepairPrompt, buildSkillUsageReport, buildTraceToolsForGroup, codeTraceBenchCase, codeTracerPredictionsToFindings, coerceJson, coerceToFindingRows, compareAnalystRunners, computeFindingId, createChatClient, createDspyRlmTraceEngine, createJudgeAdapter, createPrimeBenchmarkRunner, createPublicBenchmarkDirectRunner, createPublicBenchmarkRlmRunner, createRunCriticAdapter, createSemanticConceptJudgeAdapter, createTraceAnalyst, createVerifierAdapter, defaultIsMaterial, defineCustomAnalyst, defineTraceAnalyst, deriveEfficiencyFindings, diffFindings, effectiveAnalystProtocolSha256, emitControlIntegrityFindings, emitSkillUsageFindings, emptyPrimeRawUsage, emptyPublicBenchmarkRunner, evidenceRefsFromRawFinding, expandCodeTraceFailureBlocks, extractPrimeJsonObject, findingSubjectGrammarPromptFor, isProposalFinding, liftSeverity, loadCodeTraceVerificationArtifacts, loadPublicBenchmarkRows, makeFinding, makeProposalFinding, mergePrimeRawUsage, nodeHttpPrimeBridgeTransport, normalizeAgentRxCategory, normalizeBenchmarkLabel, normalizePrimeUsage, parseFindingSubject, parseRawFinding, parseVerificationOutcome, preparePublicAnalystBenchmark, primeAnalystProtocolSha256, primeProtocolSha256, primeReplyDefect, projectPrimeTrajectory, publicBenchmarkDistributions, publicBenchmarkProtocolSha256, publicBenchmarkRlmInstructions, publicBenchmarkSelectionReport, publicBenchmarkSystemPrompt, readAnalystBenchmarkArtifact, readAnalystInstructionsOverride, registryBenchmarkRunner, renderAgentRxCalibrationMarkdown, renderAnalystBenchmarkMarkdown, renderCodeTraceCalibrationMarkdown, renderFindingSubject, renderPriorFindings, renderUpstreamFindings, resolveTraceAnalystLimits, roundAgentRxStep, runAnalystBenchmark, runAnalystBenchmarkCommand, runPrimeExchange, runTraceAnalyst, scoreAnalystFindings, selectPublicBenchmarkRows, stripCodeFences, summarizeAgentRxCalibration, summarizeAnalystBenchmarkRunner, summarizeCodeTraceCalibration, traceStoreEvidenceResolver };
|
|
1614
|
+
export { AGENT_RX_UPSTREAM_REVISION, ANALYST_BENCHMARK_COST_LEDGER_FILE, ANALYST_BENCHMARK_DEPENDENCY_LOCK_DIGEST_ALGORITHM, ANALYST_BENCHMARK_DEPENDENCY_LOCK_FILES, ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256, ANALYST_BENCHMARK_EVIDENCE_DEPENDENCY_LOCK_SHA256, ANALYST_BENCHMARK_EVIDENCE_IMPLEMENTATION_SHA256, ANALYST_BENCHMARK_EVIDENCE_PACKAGE_VERSION, ANALYST_BENCHMARK_HELP, ANALYST_BENCHMARK_IMPLEMENTATION_DIGEST_ALGORITHM, ANALYST_BENCHMARK_IMPLEMENTATION_FILES, ANALYST_BENCHMARK_IMPLEMENTATION_SHA256, ANALYST_BENCHMARK_LOCAL_RECEIPT_FILE, ANALYST_BENCHMARK_MANIFEST_FILE, ANALYST_BENCHMARK_OBSERVATIONS_FILE, ANALYST_SEVERITIES, type AgentRxBenchmarkCaseOptions, type AgentRxCalibrationRunnerSummary, type AgentRxCalibrationSummary, type AgentRxFailure, type AgentRxPrediction, type AgentRxPredictionReport, type AgentRxRow, type Analyst, type AnalystBenchmarkArtifact, type AnalystBenchmarkCase, type AnalystBenchmarkCommandConfig, type AnalystBenchmarkCommandDependencies, type AnalystBenchmarkDatasetRef, type AnalystBenchmarkDescriptor, type AnalystBenchmarkError, type AnalystBenchmarkLabelState, type AnalystBenchmarkLocalRunReceipt, type AnalystBenchmarkObservation, type AnalystBenchmarkOutput, type AnalystBenchmarkProgressRow, type AnalystBenchmarkProvenance, type AnalystBenchmarkResult, type AnalystBenchmarkRunIdentity, type AnalystBenchmarkRunManifest, type AnalystBenchmarkRunner, type AnalystBenchmarkSummary, type AnalystBudgetDeclaration, type AnalystComparisonMetric, type AnalystContext, type AnalystCost, type AnalystDefinition, type AnalystDefinitionAsymmetry, type AnalystDefinitionAsymmetryReport, type AnalystEvidenceBinding, type AnalystEvidenceExpectation, type AnalystEvidenceResolution, type AnalystEvidenceResolutionError, type AnalystEvidenceResolver, AnalystExpressivenessError, type AnalystFinding, type AnalystFindingScore, type AnalystHooks, type AnalystInputKind, type AnalystInstructionsOverride, type AnalystIssueExpectation, type AnalystLatencyDistribution, type AnalystMetricComparison, type AnalystProfileFragment, AnalystRegistry, type AnalystRegistryOptions, type AnalystRepairDeclaration, type AnalystRequirements, type AnalystRowExpansion, type AnalystRunEvent, type AnalystRunInputs, type AnalystRunResult, type AnalystRunSummary, type AnalystRunnerComparison, type AnalystSeverity, type AnalystTransportBinding, type AnalystUsageReceipt, type BehavioralAnalystOptions, type BudgetPolicy, CODE_TRACE_BENCH_ANALYST_PROMPT, CONTROL_INTEGRITY_ANALYST, type ChatCallOpts, type ChatClient, type ChatRequest, type ChatResponse, type ChatTransport, type ChunkedEvidenceBinding, type CliBridgeTransportOpts, type CodeTraceBenchCaseOptions, type CodeTraceBenchLabelOptions, type CodeTraceBenchLabelSet, type CodeTraceBenchRow, type CodeTraceBlockDiagnostics, type CodeTraceCalibrationRunnerSummary, type CodeTraceCalibrationSummary, type CodeTraceFailureBlock, type CodeTraceStageAnnotation, type CodeTracerLabelGroup, type CodeTracerPredictionAdapterOptions, type CodeTracerPredictions, type CodeTracerStepLabel, ControlIntegrityAnalyst, type CreateChatClientOpts, type CreateTraceAnalystOptions, type CustomTransportOpts, DEFAULT_MAX_VERIFICATION_ARTIFACT_BYTES, DEFAULT_TRACE_ANALYST_KINDS, DEFAULT_TRACE_ANALYST_LIMITS, type DecodedReply, type DefaultAnalystRegistryOptions, type DefineCustomAnalystOptions, type DefineExactCustomAnalystOptions, type DiffPolicy, type DirectProviderTransportOpts, type DspyRlmTraceEngineOptions, type EvidenceProjection, type EvidenceRef, type ExactAnalystBudgetPolicy, type ExactAnalystBudgetSnapshot, type ExactAnalystExecutionPlanSnapshot, type ExactAnalystRunCompletion, type ExactAnalystRunEvent, ExactAnalystRunExecutionError, type ExactAnalystRunPolicySnapshot, type ExactAnalystRunResult, type ExactAnalystRunSummary, type ExactAnalystSnapshot, type ExactCapableAnalyst, type ExactExecutionComponentIdentity, type ExactExecutionComponentSnapshot, type ExactRegistryRunOpts, type ExecutionProbe, type ExecutionProbeOutcome, type ExecutionProbeRequest, type ExpandRowsArgs, FAILURE_MODE_KIND_SPEC, FINDING_SUBJECT_GRAMMAR_PROMPT, FINDING_SUBJECT_KINDS, FINDING_SUBJECT_SYNTAX, type FindingSubject, type FindingSubjectKind, FindingSubjectStringSchema, type FindingsDiff, FindingsStore, IMPROVEMENT_KIND_SPEC, type InlineEvidenceBinding, type JudgeAdapterOpts, KIND_EXPECTED_SUBJECTS, KNOWLEDGE_GAP_KIND_SPEC, KNOWLEDGE_POISONING_KIND_SPEC, type LoadedVerificationArtifacts, MAX_INCORRECT_BLOCKS, MAX_INCORRECT_BLOCK_STEPS, type MockTransportOpts, type PersistedFinding, type PreparedPublicAnalystBenchmark, type PrimeBenchmarkRunnerOptions, type PrimeBridgeTransport, type PrimeBridgeTransportRequest, type PrimeBridgeTransportResult, type PrimeCodeTraceDefinitionArgs, type PrimeExchangeOptions, type PrimeExchangeOutcome, type PrimeFailure, type PrimeProjectionDelivery, type PrimeProjectionOutcome, type PrimeProjectionSource, type PrimePromptSpec, type PrimeProtocolIdentity, type PrimeRawUsage, type PrimeRejectedRow, type PrimeRepairPromptSpec, type PrimeRepairState, type PrimeReplyContract, type PrimeRowDecoded, type PrimeTurnRecord, type ProposalFinding, type ProposalFindingOrigin, type PublicAnalystBenchmarkDataset, type PublicAnalystBenchmarkModelConfig, type PublicAnalystBenchmarkModelOwner, type PublicAnalystBenchmarkModelSettings, type PublicBenchmarkDistributions, type PublicBenchmarkModelPrediction, type PublicBenchmarkSelectionReport, type PublicBenchmarkValueDistribution, type PublicDirectDefinitionArgs, type PublicRlmDefinitionArgs, RAW_FINDING_SCHEMA_PROMPT, type RawAnalystEvidence, RawAnalystEvidenceSchema, type RawAnalystFinding, RawAnalystFindingSchema, type RegistryRunOpts, type ReplVariableConsensusPort, type ReplVariableEvidenceBinding, type ReplyContract, type ReplyEnvelope, type ReplyRowDecoded, type ReplyRowRejection, type RouterTransportOpts, type RunAnalystBenchmarkOptions, type RunCriticAdapterOpts, SKILL_USAGE_ANALYST, type SandboxSdkTransportOpts, type SemanticConceptJudgeAdapterOpts, SkillUsageAnalyst, type SkillUsageRecord, type SkillUsageReport, type SkillUsageScanConfig, type StepLabelAdapterOptions, type TraceAnalysisEngine, type TraceAnalysisEngineRequest, type TraceAnalysisEngineResult, type TraceAnalystDefinition, type TraceAnalystLimits, type TraceToolGroupName, type UpstreamPredictionAdapterOptions, type VerificationArtifactFile, type VerificationArtifactManifest, type VerificationArtifactRole, type VerificationAvailabilitySummary, type VerificationOutcome, type VerificationOutcomeSource, type VerificationOutcomeStatus, type VerificationResultFile, type VerifierAdapterOpts, adaptPublicBenchmarkFindings, agentRxBenchmarkCase, agentRxPredictionsToFindings, analystBenchmarkDependencyLockDigest, analystBenchmarkImplementationDigest, analystDefinitionAsymmetries, analystDefinitionProtocolSha256, analystInstructionsOverrideFromText, analystUsageReceiptFromPrimeUsage, appendVerificationArtifactsToOtlp, assertExactRegistryRunOpts, assertProposalFindings, behavioralAnalyst, bindAnalyst, buildDefaultAnalystRegistry, buildPrimePrompt, buildPrimeRepairPrompt, buildSkillUsageReport, buildTraceToolsForGroup, codeTraceBenchCase, codeTracerPredictionsToFindings, coerceJson, coerceToFindingRows, compareAnalystRunners, computeFindingId, createChatClient, createDspyRlmTraceEngine, createJudgeAdapter, createPrimeBenchmarkRunner, createPublicBenchmarkDirectRunner, createPublicBenchmarkRlmRunner, createRunCriticAdapter, createSemanticConceptJudgeAdapter, createTraceAnalyst, createVerifierAdapter, decodeReplyRows, defaultIsMaterial, defineCustomAnalyst, defineTraceAnalyst, deriveEfficiencyFindings, diffFindings, effectiveAnalystProtocolSha256, emitControlIntegrityFindings, emitSkillUsageFindings, emptyPrimeRawUsage, emptyPublicBenchmarkRunner, evidenceRefsFromRawFinding, expandCodeTraceFailureBlocks, extractPrimeJsonObject, findingSubjectGrammarPromptFor, isProposalFinding, liftSeverity, loadCodeTraceVerificationArtifacts, loadPublicBenchmarkRows, makeFinding, makeProposalFinding, mergePrimeRawUsage, nodeHttpPrimeBridgeTransport, normalizeAgentRxCategory, normalizeBenchmarkLabel, normalizePrimeUsage, parseFindingSubject, parseRawFinding, parseVerificationOutcome, preparePublicAnalystBenchmark, primeAnalystProtocolSha256, primeCodeTraceAnalystDefinition, primeProtocolSha256, primeReplyDefect, projectPrimeTrajectory, publicBenchmarkDistributions, publicBenchmarkProtocolSha256, publicBenchmarkRlmInstructions, publicBenchmarkSelectionReport, publicBenchmarkSystemPrompt, publicDirectAnalystDefinition, publicRlmAnalystDefinition, readAnalystBenchmarkArtifact, readAnalystInstructionsOverride, registryBenchmarkRunner, renderAgentRxCalibrationMarkdown, renderAnalystBenchmarkMarkdown, renderCodeTraceCalibrationMarkdown, renderFindingSubject, renderPriorFindings, renderUpstreamFindings, resolveTraceAnalystLimits, rlmEngineLimits, roundAgentRxStep, runAnalystBenchmark, runAnalystBenchmarkCommand, runPrimeExchange, runTraceAnalyst, scoreAnalystFindings, selectPublicBenchmarkRows, stripCodeFences, summarizeAgentRxCalibration, summarizeAnalystBenchmarkRunner, summarizeCodeTraceCalibration, traceStoreEvidenceResolver };
|
|
1233
1615
|
//# sourceMappingURL=index.d.ts.map
|