@tangle-network/agent-eval 0.144.0 → 0.144.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +36 -0
- package/dist/analyst/index.d.ts +88 -23
- package/dist/analyst/index.d.ts.map +1 -1
- package/dist/analyst/index.js +4 -4
- package/dist/{analyze-runs-BScZvqMV.js → analyze-runs-DWIvOAGk.js} +2 -2
- package/dist/{analyze-runs-BScZvqMV.js.map → analyze-runs-DWIvOAGk.js.map} +1 -1
- package/dist/{benchmark-DxaZfy0w.d.ts → benchmark-CP6kWfj8.d.ts} +3 -3
- package/dist/{benchmark-DxaZfy0w.d.ts.map → benchmark-CP6kWfj8.d.ts.map} +1 -1
- package/dist/{benchmark-command-4c7N_rlw.js → benchmark-command-CQPKRUr-.js} +332 -111
- package/dist/benchmark-command-CQPKRUr-.js.map +1 -0
- package/dist/benchmarks/index.d.ts +1 -1
- package/dist/benchmarks/index.js +1 -1
- package/dist/{benchmarks-05UGZ8sZ.js → benchmarks-CkG1bWFa.js} +4 -4
- package/dist/{benchmarks-05UGZ8sZ.js.map → benchmarks-CkG1bWFa.js.map} +1 -1
- package/dist/campaign/index.d.ts +6 -6
- package/dist/campaign/index.js +4 -4
- package/dist/{campaign-BKOtvRAB.js → campaign-DjGFyPxH.js} +74 -40
- package/dist/campaign-DjGFyPxH.js.map +1 -0
- package/dist/cli.js +2 -2
- package/dist/{client-Cgl6KasJ.d.ts → client-Bbht4xxl.d.ts} +4 -4
- package/dist/{client-Cgl6KasJ.d.ts.map → client-Bbht4xxl.d.ts.map} +1 -1
- package/dist/{completion-verifier-D15NHYSk.d.ts → completion-verifier-EJERfFwF.d.ts} +3 -3
- package/dist/{completion-verifier-D15NHYSk.d.ts.map → completion-verifier-EJERfFwF.d.ts.map} +1 -1
- package/dist/contract/index.d.ts +8 -8
- package/dist/contract/index.js +10 -8
- package/dist/contract/index.js.map +1 -1
- package/dist/control.d.ts +2 -2
- package/dist/control.js +1 -1
- package/dist/{default-registry-DfHJEwYh.js → default-registry-SOyHB6qG.js} +2 -2
- package/dist/{default-registry-DfHJEwYh.js.map → default-registry-SOyHB6qG.js.map} +1 -1
- package/dist/{default-registry-D3uqKbo6.d.ts → default-registry-iXfu2trt.d.ts} +5 -5
- package/dist/{default-registry-D3uqKbo6.d.ts.map → default-registry-iXfu2trt.d.ts.map} +1 -1
- package/dist/{dspy-rlm-engine-CBFwlyaY.js → dspy-rlm-engine-IRCG8kdi.js} +61 -105
- package/dist/dspy-rlm-engine-IRCG8kdi.js.map +1 -0
- package/dist/{eval-campaign-YdkpWWoT.js → eval-campaign-lZcDIwQM.js} +3 -3
- package/dist/{eval-campaign-YdkpWWoT.js.map → eval-campaign-lZcDIwQM.js.map} +1 -1
- package/dist/{exact-types-B0lJV3tu.d.ts → exact-types-BygCBR4L.d.ts} +2 -2
- package/dist/{exact-types-B0lJV3tu.d.ts.map → exact-types-BygCBR4L.d.ts.map} +1 -1
- package/dist/{external-optimizer-contracts-iK0yu4AR.d.ts → external-optimizer-contracts-CdmX2K2S.d.ts} +47 -10
- package/dist/external-optimizer-contracts-CdmX2K2S.d.ts.map +1 -0
- package/dist/{feedback-trajectory-BCHqzLh3.d.ts → feedback-trajectory-CSIkRLQX.d.ts} +3 -3
- package/dist/{feedback-trajectory-BCHqzLh3.d.ts.map → feedback-trajectory-CSIkRLQX.d.ts.map} +1 -1
- package/dist/hosted/index.d.ts +3 -3
- package/dist/{index-DgbFl4cv.d.ts → index-BuHs_OnD.d.ts} +17 -12
- package/dist/{index-DgbFl4cv.d.ts.map → index-BuHs_OnD.d.ts.map} +1 -1
- package/dist/{index-D2enbqA0.d.ts → index-CNOCxBLh.d.ts} +2 -2
- package/dist/{index-D2enbqA0.d.ts.map → index-CNOCxBLh.d.ts.map} +1 -1
- package/dist/{index-DFLVtPZ9.d.ts → index-DGIzNtRv.d.ts} +2 -2
- package/dist/{index-DFLVtPZ9.d.ts.map → index-DGIzNtRv.d.ts.map} +1 -1
- package/dist/{index-DtMpBKVF.d.ts → index-D_qTihaQ.d.ts} +5 -5
- package/dist/{index-DtMpBKVF.d.ts.map → index-D_qTihaQ.d.ts.map} +1 -1
- package/dist/index.d.ts +20 -20
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +17 -17
- package/dist/{insight-report-Bh_8ksel.d.ts → insight-report-D5m1z0_n.d.ts} +3 -3
- package/dist/{insight-report-Bh_8ksel.d.ts.map → insight-report-D5m1z0_n.d.ts.map} +1 -1
- package/dist/ledger-core/index.js +1 -1
- package/dist/{ledger-core-Dxz0Rkwa.js → ledger-core-DXZIqu17.js} +3 -3
- package/dist/ledger-core-DXZIqu17.js.map +1 -0
- package/dist/{llm-client-B3WXSH5Y.js → llm-client-D3EoChAU.js} +7 -7
- package/dist/llm-client-D3EoChAU.js.map +1 -0
- package/dist/meta-eval/index.d.ts +1 -1
- package/dist/{mint-Ctwk079K.js → mint-DD-0oQTA.js} +2 -2
- package/dist/{mint-Ctwk079K.js.map → mint-DD-0oQTA.js.map} +1 -1
- package/dist/multishot/index.d.ts +1 -1
- package/dist/openapi.json +1 -1
- package/dist/{propose-review-control-DLXz4FCX.js → propose-review-control-BciUCZoh.js} +2 -2
- package/dist/{propose-review-control-DLXz4FCX.js.map → propose-review-control-BciUCZoh.js.map} +1 -1
- package/dist/{release-report-B_bQOHM-.d.ts → release-report-BZRWdq_t.d.ts} +3 -3
- package/dist/{release-report-B_bQOHM-.d.ts.map → release-report-BZRWdq_t.d.ts.map} +1 -1
- package/dist/{release-report-B5XPBvAU.js → release-report-Sl0xfkFv.js} +2 -2
- package/dist/{release-report-B5XPBvAU.js.map → release-report-Sl0xfkFv.js.map} +1 -1
- package/dist/{replay-DjUfTrHD.js → replay-CqOsGjzU.js} +2 -9
- package/dist/replay-CqOsGjzU.js.map +1 -0
- package/dist/{replay-BuJM6kLh.d.ts → replay-DQ-55DC_.d.ts} +4 -4
- package/dist/{replay-BuJM6kLh.d.ts.map → replay-DQ-55DC_.d.ts.map} +1 -1
- package/dist/reporting.d.ts +3 -3
- package/dist/reporting.js +1 -1
- package/dist/{researcher-C6lzl-rP.d.ts → researcher-BSiCoM1s.d.ts} +3 -3
- package/dist/{researcher-C6lzl-rP.d.ts.map → researcher-BSiCoM1s.d.ts.map} +1 -1
- package/dist/{reward-hacking-DjTi9HLb.js → reward-hacking-CyuzxKly.js} +2 -2
- package/dist/{reward-hacking-DjTi9HLb.js.map → reward-hacking-CyuzxKly.js.map} +1 -1
- package/dist/{reward-hacking-CEVVmy3h.d.ts → reward-hacking-DFgkEY4p.d.ts} +2 -2
- package/dist/{reward-hacking-CEVVmy3h.d.ts.map → reward-hacking-DFgkEY4p.d.ts.map} +1 -1
- package/dist/rl.d.ts +5 -5
- package/dist/rl.js +4 -4
- package/dist/rollout/index.d.ts +1 -1
- package/dist/rollout/index.js +2 -2
- package/dist/{rollout-8nj3mYvx.js → rollout-C2fD1cf4.js} +2 -2
- package/dist/{rollout-8nj3mYvx.js.map → rollout-C2fD1cf4.js.map} +1 -1
- package/dist/{rubric-predictive-validity-C2CthIfY.d.ts → rubric-predictive-validity-C4r4Y-q8.d.ts} +2 -2
- package/dist/{rubric-predictive-validity-C2CthIfY.d.ts.map → rubric-predictive-validity-C4r4Y-q8.d.ts.map} +1 -1
- package/dist/{run-evidence-8Ou28QSa.d.ts → run-evidence-H1vRpIdT.d.ts} +3 -3
- package/dist/{run-evidence-8Ou28QSa.d.ts.map → run-evidence-H1vRpIdT.d.ts.map} +1 -1
- package/dist/{run-record-vRgqWmJw.js → run-record-CWN8-VsV.js} +40 -14
- package/dist/{run-record-vRgqWmJw.js.map → run-record-CWN8-VsV.js.map} +1 -1
- package/dist/{run-record-Tb3TTtUn.d.ts → run-record-ooo9FWns.d.ts} +5 -9
- package/dist/{run-record-Tb3TTtUn.d.ts.map → run-record-ooo9FWns.d.ts.map} +1 -1
- package/dist/{semantic-concept-judge-DJQtFr95.js → semantic-concept-judge-Do5aM9wP.js} +3 -3
- package/dist/{semantic-concept-judge-DJQtFr95.js.map → semantic-concept-judge-Do5aM9wP.js.map} +1 -1
- package/dist/{server-Cu4M3NSO.js → server-Df00sdwz.js} +3 -3
- package/dist/{server-Cu4M3NSO.js.map → server-Df00sdwz.js.map} +1 -1
- package/dist/{single-run-lock-t1si1ob7.js → single-run-lock-D5iN0Xzb.js} +719 -187
- package/dist/single-run-lock-D5iN0Xzb.js.map +1 -0
- package/dist/{skill-usage-BiVEU0QY.d.ts → skill-usage-DtpLou9L.d.ts} +28 -9
- package/dist/skill-usage-DtpLou9L.d.ts.map +1 -0
- package/dist/{skillopt-optimization-method-Ds8J1_K8.js → skillopt-optimization-method-C4FX42dy.js} +42 -210
- package/dist/skillopt-optimization-method-C4FX42dy.js.map +1 -0
- package/dist/{skillopt-optimization-method-B7o01OdX.d.ts → skillopt-optimization-method-CvSJGdm3.d.ts} +12 -8
- package/dist/{skillopt-optimization-method-B7o01OdX.d.ts.map → skillopt-optimization-method-CvSJGdm3.d.ts.map} +1 -1
- package/dist/{summary-report-o3eJ3gxG.d.ts → summary-report-BOM6dfP7.d.ts} +2 -2
- package/dist/{summary-report-o3eJ3gxG.d.ts.map → summary-report-BOM6dfP7.d.ts.map} +1 -1
- package/dist/{tool-groups-Bqy4A3QB.d.ts → tool-groups-CMmsgTzj.d.ts} +7 -7
- package/dist/{tool-groups-Bqy4A3QB.d.ts.map → tool-groups-CMmsgTzj.d.ts.map} +1 -1
- package/dist/traces.d.ts +4 -4
- package/dist/traces.js +1 -1
- package/dist/types-BjMFz88h.d.ts.map +1 -1
- package/dist/{types-KEqL1pZc.d.ts → types-DcJxgsLy.d.ts} +8 -6
- package/dist/{types-KEqL1pZc.d.ts.map → types-DcJxgsLy.d.ts.map} +1 -1
- package/dist/{types-D3jh6F98.d.ts → types-y8jrxXWd.d.ts} +2 -2
- package/dist/{types-D3jh6F98.d.ts.map → types-y8jrxXWd.d.ts.map} +1 -1
- package/dist/wire/index.d.ts +1 -1
- package/dist/wire/index.js +1 -1
- package/docs/eval-surface-map.md +4 -0
- package/docs/trace-analysis.md +43 -32
- package/package.json +1 -1
- package/dist/benchmark-command-4c7N_rlw.js.map +0 -1
- package/dist/campaign-BKOtvRAB.js.map +0 -1
- package/dist/dspy-rlm-engine-CBFwlyaY.js.map +0 -1
- package/dist/external-optimizer-contracts-iK0yu4AR.d.ts.map +0 -1
- package/dist/ledger-core-Dxz0Rkwa.js.map +0 -1
- package/dist/llm-client-B3WXSH5Y.js.map +0 -1
- package/dist/replay-DjUfTrHD.js.map +0 -1
- package/dist/single-run-lock-t1si1ob7.js.map +0 -1
- package/dist/skill-usage-BiVEU0QY.d.ts.map +0 -1
- package/dist/skillopt-optimization-method-Ds8J1_K8.js.map +0 -1
package/CHANGELOG.md
CHANGED
|
@@ -4,6 +4,42 @@ All notable changes to `@tangle-network/agent-eval` and its sibling `agent-eval-
|
|
|
4
4
|
|
|
5
5
|
---
|
|
6
6
|
|
|
7
|
+
## [0.144.3] - 2026-08-03 - exact profile matrix evidence
|
|
8
|
+
|
|
9
|
+
### Changed
|
|
10
|
+
|
|
11
|
+
- Campaign cells expose every distinct agent receipt model in `resolvedModels` and expose `resolvedModel` only when all agent receipts agree.
|
|
12
|
+
- Snapshot validation accepts Router `-MMDD` model snapshots while rejecting routing selectors such as `@preset/name`.
|
|
13
|
+
|
|
14
|
+
### Fixed
|
|
15
|
+
|
|
16
|
+
- `runProfileMatrix` rejects mismatched receipt models, multiple resolved snapshots within one profile, and duplicate profile identities before they can corrupt comparisons.
|
|
17
|
+
- A failed profile campaign now cancels active sibling campaigns through their existing abort signals instead of allowing additional paid work to continue.
|
|
18
|
+
- Profile campaign cache identity now includes the caller commit, optional `dispatchRef`, profile identity, and comparison config, so changed execution cannot reuse and relabel an old cell.
|
|
19
|
+
|
|
20
|
+
## [0.144.2] - 2026-08-03 - concurrent exact profile comparison
|
|
21
|
+
|
|
22
|
+
### Changed
|
|
23
|
+
|
|
24
|
+
- `runProfileMatrix` accepts caller-controlled `maxProfileConcurrency` while preserving deterministic output order and independent per-profile run directories and cost ceilings.
|
|
25
|
+
- Profiles may keep the exact provider-facing model alias used for execution when every paid-call receipt supplies the related snapshot-bearing model written to durable records.
|
|
26
|
+
|
|
27
|
+
### Fixed
|
|
28
|
+
|
|
29
|
+
- Broad profile comparisons no longer have to serialize every profile campaign or maintain a second matrix runner just to compare exact execution profiles at practical throughput.
|
|
30
|
+
|
|
31
|
+
## [0.144.1] - 2026-08-03 - runtime-owned optimizer model calls
|
|
32
|
+
|
|
33
|
+
### Fixed
|
|
34
|
+
|
|
35
|
+
- Official GEPA, SkillOpt, and DSPy child requests now cross one OpenAI-compatible loopback boundary and invoke the caller-owned canonical model callback exactly once.
|
|
36
|
+
Provider URLs, credentials, retries, and raw HTTP responses no longer cross Agent Eval's public callback.
|
|
37
|
+
- Each invoked optimizer-model call carries a stable call ID, a deeply immutable canonical chat request, the original abort signal, and its endpoint format.
|
|
38
|
+
The owner must return a canonical chat response, an exact cost receipt, and finite JSON execution evidence.
|
|
39
|
+
- Response and receipt validation now preserves cached-input, cache-write, reasoning, output, actual, estimated, and unknown cost semantics without counting cached input twice.
|
|
40
|
+
- The public analyst benchmark records the model-owner callback identity and loads it from an explicit owner module instead of accepting provider credentials.
|
|
41
|
+
- Pinned official GEPA and SkillOpt request fixtures now cover the real library payloads, including SkillOpt's dynamic system-role task message.
|
|
42
|
+
|
|
7
43
|
## [0.144.0] - 2026-08-03 - caller-owned optimizer execution
|
|
8
44
|
|
|
9
45
|
### Changed
|
package/dist/analyst/index.d.ts
CHANGED
|
@@ -1,14 +1,14 @@
|
|
|
1
1
|
import { a as MultiLayerVerifier, l as VerifyOptions, o as Severity } from "../multi-layer-verifier-BHY1gWAc.js";
|
|
2
|
-
import { A as parseFindingSubject, C as FINDING_SUBJECT_KINDS, D as FindingSubjectStringSchema, E as FindingSubjectKind, F as DefineExactCustomAnalystOptions, G as SemanticConceptJudgeOptions, I as defineCustomAnalyst, L as defineTraceAnalyst, M as DspyRlmTraceEngineOptions, N as createDspyRlmTraceEngine, O as KIND_EXPECTED_SUBJECTS, P as DefineCustomAnalystOptions, S as FINDING_SUBJECT_GRAMMAR_PROMPT, T as FindingSubject, W as SemanticConceptJudgeInput, Y as RunCritic, Z as RunTrace, _ as FindingsDiff, a as SkillUsageScanConfig, b as defaultIsMaterial, c as DEFAULT_TRACE_ANALYST_KINDS, d as IMPROVEMENT_KIND_SPEC, f as FAILURE_MODE_KIND_SPEC, g as DiffPolicy, h as emitControlIntegrityFindings, i as SkillUsageReport, j as renderFindingSubject, k as findingSubjectGrammarPromptFor, l as KNOWLEDGE_POISONING_KIND_SPEC, m as ControlIntegrityAnalyst, n as SkillUsageAnalyst, o as buildSkillUsageReport, p as CONTROL_INTEGRITY_ANALYST, r as SkillUsageRecord, s as emitSkillUsageFindings, t as SKILL_USAGE_ANALYST, u as KNOWLEDGE_GAP_KIND_SPEC, v as FindingsStore, w as FINDING_SUBJECT_SYNTAX, x as diffFindings, y as PersistedFinding } from "../skill-usage-
|
|
2
|
+
import { A as parseFindingSubject, C as FINDING_SUBJECT_KINDS, D as FindingSubjectStringSchema, E as FindingSubjectKind, F as DefineExactCustomAnalystOptions, G as SemanticConceptJudgeOptions, I as defineCustomAnalyst, L as defineTraceAnalyst, M as DspyRlmTraceEngineOptions, N as createDspyRlmTraceEngine, O as KIND_EXPECTED_SUBJECTS, P as DefineCustomAnalystOptions, S as FINDING_SUBJECT_GRAMMAR_PROMPT, T as FindingSubject, W as SemanticConceptJudgeInput, Y as RunCritic, Z as RunTrace, _ as FindingsDiff, a as SkillUsageScanConfig, b as defaultIsMaterial, c as DEFAULT_TRACE_ANALYST_KINDS, d as IMPROVEMENT_KIND_SPEC, f as FAILURE_MODE_KIND_SPEC, g as DiffPolicy, h as emitControlIntegrityFindings, i as SkillUsageReport, j as renderFindingSubject, k as findingSubjectGrammarPromptFor, l as KNOWLEDGE_POISONING_KIND_SPEC, m as ControlIntegrityAnalyst, n as SkillUsageAnalyst, o as buildSkillUsageReport, p as CONTROL_INTEGRITY_ANALYST, r as SkillUsageRecord, s as emitSkillUsageFindings, t as SKILL_USAGE_ANALYST, u as KNOWLEDGE_GAP_KIND_SPEC, v as FindingsStore, w as FINDING_SUBJECT_SYNTAX, x as diffFindings, y as PersistedFinding } from "../skill-usage-DtpLou9L.js";
|
|
3
3
|
import { b as CustomTokenPricing, c as CostLedgerHandle } from "../cost-ledger-FuQvHxPm.js";
|
|
4
4
|
import { A as ChatClient, B as SandboxSdkTransportOpts, F as CreateChatClientOpts, I as CustomTransportOpts, L as DirectProviderTransportOpts, M as ChatResponse, N as ChatTransport, P as CliBridgeTransportOpts, R as MockTransportOpts, V as createChatClient, j as ChatRequest, k as ChatCallOpts, m as JudgeInput, p as JudgeFn, z as RouterTransportOpts } from "../types-BjMFz88h.js";
|
|
5
|
-
import { _ as makeFinding, a as AnalystInputKind, c as AnalystRunInputs, d as AnalystSeverity, f as AnalystUsageReceipt, g as computeFindingId, h as ProposalFindingOrigin, i as AnalystFinding, l as AnalystRunResult, m as ProposalFinding, n as AnalystContext, o as AnalystRequirements, p as EvidenceRef, r as AnalystCost, s as AnalystRunEvent, t as Analyst, u as AnalystRunSummary, v as makeProposalFinding, x as TraceAnalysisStore } from "../types-
|
|
6
|
-
import { a as createTraceAnalyst, c as runTraceAnalyst, d as deriveEfficiencyFindings, i as TraceAnalystDefinition, l as BehavioralAnalystOptions, n as buildDefaultAnalystRegistry, o as renderPriorFindings, r as CreateTraceAnalystOptions, s as renderUpstreamFindings, t as DefaultAnalystRegistryOptions, u as behavioralAnalyst } from "../default-registry-
|
|
7
|
-
import { a as ExactAnalystRunPolicySnapshot, c as ExactAnalystSnapshot, d as ExactExecutionComponentSnapshot, i as ExactAnalystRunEvent, l as ExactCapableAnalyst, n as ExactAnalystExecutionPlanSnapshot, o as ExactAnalystRunResult, r as ExactAnalystRunCompletion, s as ExactAnalystRunSummary, t as ExactAnalystBudgetSnapshot, u as ExactExecutionComponentIdentity } from "../exact-types-
|
|
8
|
-
import { A as ExactAnalystRunExecutionError, D as AnalystRegistryOptions, E as AnalystRegistry, M as RegistryRunOpts, N as assertExactRegistryRunOpts, O as BudgetPolicy, T as AnalystHooks, j as ExactRegistryRunOpts, k as ExactAnalystBudgetPolicy } from "../completion-verifier-
|
|
9
|
-
import { C as scoreAnalystFindings, S as traceStoreEvidenceResolver, _ as AnalystIssueExpectation, a as AnalystBenchmarkLabelState, b as registryBenchmarkRunner, c as AnalystBenchmarkProvenance, d as AnalystBenchmarkSummary, f as AnalystEvidenceExpectation, g as AnalystFindingScore, h as AnalystEvidenceResolver, i as AnalystBenchmarkError, l as AnalystBenchmarkResult, m as AnalystEvidenceResolutionError, n as AnalystBenchmarkDatasetRef, o as AnalystBenchmarkObservation, p as AnalystEvidenceResolution, r as AnalystBenchmarkDescriptor, s as AnalystBenchmarkOutput, t as AnalystBenchmarkCase, u as AnalystBenchmarkRunner, v as AnalystLatencyDistribution, x as runAnalystBenchmark, y as RunAnalystBenchmarkOptions } from "../benchmark-
|
|
10
|
-
import {
|
|
11
|
-
import { a as TraceAnalysisEngineRequest, c as resolveTraceAnalystLimits, d as RawAnalystEvidence, f as RawAnalystEvidenceSchema, g as parseRawFinding, h as evidenceRefsFromRawFinding, i as TraceAnalysisEngine, l as ANALYST_SEVERITIES, m as RawAnalystFindingSchema, n as buildTraceToolsForGroup, o as TraceAnalysisEngineResult, p as RawAnalystFinding, r as DEFAULT_TRACE_ANALYST_LIMITS, s as TraceAnalystLimits, t as TraceToolGroupName, u as RAW_FINDING_SCHEMA_PROMPT } from "../tool-groups-
|
|
5
|
+
import { _ as makeFinding, a as AnalystInputKind, c as AnalystRunInputs, d as AnalystSeverity, f as AnalystUsageReceipt, g as computeFindingId, h as ProposalFindingOrigin, i as AnalystFinding, l as AnalystRunResult, m as ProposalFinding, n as AnalystContext, o as AnalystRequirements, p as EvidenceRef, r as AnalystCost, s as AnalystRunEvent, t as Analyst, u as AnalystRunSummary, v as makeProposalFinding, x as TraceAnalysisStore } from "../types-y8jrxXWd.js";
|
|
6
|
+
import { a as createTraceAnalyst, c as runTraceAnalyst, d as deriveEfficiencyFindings, i as TraceAnalystDefinition, l as BehavioralAnalystOptions, n as buildDefaultAnalystRegistry, o as renderPriorFindings, r as CreateTraceAnalystOptions, s as renderUpstreamFindings, t as DefaultAnalystRegistryOptions, u as behavioralAnalyst } from "../default-registry-iXfu2trt.js";
|
|
7
|
+
import { a as ExactAnalystRunPolicySnapshot, c as ExactAnalystSnapshot, d as ExactExecutionComponentSnapshot, i as ExactAnalystRunEvent, l as ExactCapableAnalyst, n as ExactAnalystExecutionPlanSnapshot, o as ExactAnalystRunResult, r as ExactAnalystRunCompletion, s as ExactAnalystRunSummary, t as ExactAnalystBudgetSnapshot, u as ExactExecutionComponentIdentity } from "../exact-types-BygCBR4L.js";
|
|
8
|
+
import { A as ExactAnalystRunExecutionError, D as AnalystRegistryOptions, E as AnalystRegistry, M as RegistryRunOpts, N as assertExactRegistryRunOpts, O as BudgetPolicy, T as AnalystHooks, j as ExactRegistryRunOpts, k as ExactAnalystBudgetPolicy } from "../completion-verifier-EJERfFwF.js";
|
|
9
|
+
import { C as scoreAnalystFindings, S as traceStoreEvidenceResolver, _ as AnalystIssueExpectation, a as AnalystBenchmarkLabelState, b as registryBenchmarkRunner, c as AnalystBenchmarkProvenance, d as AnalystBenchmarkSummary, f as AnalystEvidenceExpectation, g as AnalystFindingScore, h as AnalystEvidenceResolver, i as AnalystBenchmarkError, l as AnalystBenchmarkResult, m as AnalystEvidenceResolutionError, n as AnalystBenchmarkDatasetRef, o as AnalystBenchmarkObservation, p as AnalystEvidenceResolution, r as AnalystBenchmarkDescriptor, s as AnalystBenchmarkOutput, t as AnalystBenchmarkCase, u as AnalystBenchmarkRunner, v as AnalystLatencyDistribution, x as runAnalystBenchmark, y as RunAnalystBenchmarkOptions } from "../benchmark-CP6kWfj8.js";
|
|
10
|
+
import { f as ExternalOptimizerModelExecutionObservation, h as ExternalOptimizerRunnerCommand, l as ExternalOptimizerModelCall } from "../external-optimizer-contracts-CdmX2K2S.js";
|
|
11
|
+
import { a as TraceAnalysisEngineRequest, c as resolveTraceAnalystLimits, d as RawAnalystEvidence, f as RawAnalystEvidenceSchema, g as parseRawFinding, h as evidenceRefsFromRawFinding, i as TraceAnalysisEngine, l as ANALYST_SEVERITIES, m as RawAnalystFindingSchema, n as buildTraceToolsForGroup, o as TraceAnalysisEngineResult, p as RawAnalystFinding, r as DEFAULT_TRACE_ANALYST_LIMITS, s as TraceAnalystLimits, t as TraceToolGroupName, u as RAW_FINDING_SCHEMA_PROMPT } from "../tool-groups-CMmsgTzj.js";
|
|
12
12
|
//#region src/analyst/adapters.d.ts
|
|
13
13
|
declare function liftSeverity(s: Severity): AnalystSeverity;
|
|
14
14
|
interface VerifierAdapterOpts<Env> {
|
|
@@ -282,12 +282,32 @@ interface AnalystInstructionsOverride {
|
|
|
282
282
|
/** SHA-256 hex digest of `text`. */
|
|
283
283
|
readonly sha256: string;
|
|
284
284
|
}
|
|
285
|
+
/** Model execution supplied by the package that owns credentials and provider policy. */
|
|
286
|
+
interface PublicAnalystBenchmarkModelOwner {
|
|
287
|
+
call: ExternalOptimizerModelCall;
|
|
288
|
+
callRef: string;
|
|
289
|
+
recordExecution: (observation: ExternalOptimizerModelExecutionObservation) => void;
|
|
290
|
+
/** Exact rates when the selected model is absent from Agent Eval's catalog. */
|
|
291
|
+
pricing?: CustomTokenPricing;
|
|
292
|
+
}
|
|
285
293
|
interface PublicAnalystBenchmarkModelConfig {
|
|
286
|
-
|
|
287
|
-
|
|
294
|
+
/** Caller-owned execution path. Agent Eval never receives provider credentials. */
|
|
295
|
+
call: ExternalOptimizerModelCall;
|
|
296
|
+
/** Stable public identity for the caller-owned execution path. */
|
|
297
|
+
callRef: string;
|
|
298
|
+
/** Persist every finite execution record returned by the caller-owned path. */
|
|
299
|
+
recordExecution: (observation: ExternalOptimizerModelExecutionObservation) => void;
|
|
288
300
|
model: string;
|
|
289
301
|
maxOutputTokens: number;
|
|
290
302
|
timeoutMs: number;
|
|
303
|
+
/** Model request bytes per call. Default: 16 MiB. */
|
|
304
|
+
maxModelRequestBytes?: number;
|
|
305
|
+
/** Model response bytes per call. Default: 4 MiB. */
|
|
306
|
+
maxModelResponseBytes?: number;
|
|
307
|
+
/** Reasoning tokens billed beyond completion tokens. Default: four times output. */
|
|
308
|
+
maxReasoningTokens?: number;
|
|
309
|
+
/** Deadline for one caller-owned model invocation. Default: timeoutMs. */
|
|
310
|
+
modelRequestTimeoutMs?: number;
|
|
291
311
|
/** Required when the model is absent from agent-eval's pricing table. */
|
|
292
312
|
pricing?: CustomTokenPricing;
|
|
293
313
|
/** Independent per-case recursive-engine spend limit. Default: 1 USD. */
|
|
@@ -300,6 +320,10 @@ interface PublicAnalystBenchmarkModelConfig {
|
|
|
300
320
|
maxLlmCalls?: number;
|
|
301
321
|
maxToolCalls?: number;
|
|
302
322
|
maxOutputChars?: number;
|
|
323
|
+
maxModelRequests?: number;
|
|
324
|
+
traceToolRequestBytes?: number;
|
|
325
|
+
traceToolResponseBytes?: number;
|
|
326
|
+
traceToolTimeoutMs?: number;
|
|
303
327
|
/**
|
|
304
328
|
* Independent engine runs per case. Above 1 (CodeTraceBench only), the
|
|
305
329
|
* runner scores the step-level majority consensus across all runs instead
|
|
@@ -312,8 +336,6 @@ interface PublicAnalystBenchmarkModelConfig {
|
|
|
312
336
|
runIdentitySha256: string;
|
|
313
337
|
responseCacheDir: string;
|
|
314
338
|
};
|
|
315
|
-
/** Test-only transport injection. */
|
|
316
|
-
fetchImpl?: typeof fetch;
|
|
317
339
|
}
|
|
318
340
|
interface PreparedPublicAnalystBenchmark {
|
|
319
341
|
cases: AnalystBenchmarkCase<AnalystRunInputs>[];
|
|
@@ -586,8 +608,30 @@ interface AnalystBenchmarkArtifact {
|
|
|
586
608
|
/** Absent on artifacts produced before consensus sampling existed. */
|
|
587
609
|
rlmSamples?: number;
|
|
588
610
|
model: string;
|
|
611
|
+
/** These fields are absent only on immutable evidence produced before model owners existed. */
|
|
612
|
+
modelOwnerCallRef?: string;
|
|
589
613
|
maxOutputTokens: number;
|
|
614
|
+
maxReasoningTokens?: number;
|
|
615
|
+
maxModelRequestBytes?: number;
|
|
616
|
+
maxModelResponseBytes?: number;
|
|
617
|
+
modelRequestTimeoutMs?: number;
|
|
590
618
|
timeoutMs: number;
|
|
619
|
+
pricing?: CustomTokenPricing;
|
|
620
|
+
recursiveLimits?: {
|
|
621
|
+
maxIterations: number;
|
|
622
|
+
maxLlmCalls: number;
|
|
623
|
+
maxToolCalls: number;
|
|
624
|
+
maxOutputChars: number;
|
|
625
|
+
maxModelRequests: number | null;
|
|
626
|
+
traceToolRequestBytes: number;
|
|
627
|
+
traceToolResponseBytes: number;
|
|
628
|
+
traceToolTimeoutMs: number;
|
|
629
|
+
};
|
|
630
|
+
processLimits?: {
|
|
631
|
+
maxInputBytes: number;
|
|
632
|
+
maxResultBytes: number;
|
|
633
|
+
maxOutputChars: number;
|
|
634
|
+
};
|
|
591
635
|
maxCostUsd: number;
|
|
592
636
|
maxArtifactBytes: number;
|
|
593
637
|
analystProtocolSha256: string;
|
|
@@ -619,8 +663,29 @@ interface AnalystBenchmarkRunIdentity {
|
|
|
619
663
|
datasetSplit: string;
|
|
620
664
|
model: {
|
|
621
665
|
id: string;
|
|
666
|
+
ownerCallRef: string;
|
|
622
667
|
maxOutputTokens: number;
|
|
668
|
+
maxReasoningTokens: number;
|
|
669
|
+
maxRequestBytes: number;
|
|
670
|
+
maxResponseBytes: number;
|
|
671
|
+
requestTimeoutMs: number;
|
|
623
672
|
timeoutMs: number;
|
|
673
|
+
pricing: CustomTokenPricing;
|
|
674
|
+
recursiveLimits: {
|
|
675
|
+
maxIterations: number;
|
|
676
|
+
maxLlmCalls: number;
|
|
677
|
+
maxToolCalls: number;
|
|
678
|
+
maxOutputChars: number;
|
|
679
|
+
maxModelRequests: number | null;
|
|
680
|
+
traceToolRequestBytes: number;
|
|
681
|
+
traceToolResponseBytes: number;
|
|
682
|
+
traceToolTimeoutMs: number;
|
|
683
|
+
};
|
|
684
|
+
processLimits: {
|
|
685
|
+
maxInputBytes: number;
|
|
686
|
+
maxResultBytes: number;
|
|
687
|
+
maxOutputChars: number;
|
|
688
|
+
};
|
|
624
689
|
};
|
|
625
690
|
limit: number;
|
|
626
691
|
seed: number;
|
|
@@ -666,8 +731,7 @@ interface AnalystBenchmarkLocalRunReceipt {
|
|
|
666
731
|
traceDir: string;
|
|
667
732
|
artifactDir?: string;
|
|
668
733
|
outputDir: string;
|
|
669
|
-
|
|
670
|
-
apiKeyEnvironment: string;
|
|
734
|
+
modelOwnerModule: string;
|
|
671
735
|
};
|
|
672
736
|
command: string;
|
|
673
737
|
environment: {
|
|
@@ -702,14 +766,15 @@ declare function readAnalystBenchmarkArtifact(path: string): Promise<AnalystBenc
|
|
|
702
766
|
//#region src/analyst/benchmark-command.d.ts
|
|
703
767
|
interface AnalystBenchmarkCommandDependencies {
|
|
704
768
|
createAnalystRunner?: (dataset: PublicAnalystBenchmarkDataset, config: PublicAnalystBenchmarkModelConfig) => AnalystBenchmarkRunner<AnalystRunInputs>;
|
|
769
|
+
loadModelExecutionOwner?: (moduleRef: string, context: {
|
|
770
|
+
model: string;
|
|
771
|
+
environment: Readonly<NodeJS.ProcessEnv>;
|
|
772
|
+
}) => Promise<PublicAnalystBenchmarkModelOwner>;
|
|
705
773
|
}
|
|
706
774
|
/**
|
|
707
775
|
* Which analyst produces the scored arm.
|
|
708
776
|
*
|
|
709
|
-
* `dspy-rlm` is the recursive engine. `direct` is the
|
|
710
|
-
* kept reachable because the published evidence was produced by it: a
|
|
711
|
-
* comparison against those numbers is only sound when the same runner can be
|
|
712
|
-
* re-run over the same inputs.
|
|
777
|
+
* `dspy-rlm` is the recursive engine. `direct` is the one-shot comparison arm.
|
|
713
778
|
*/
|
|
714
779
|
type AnalystBenchmarkRunnerKind = 'dspy-rlm' | 'direct';
|
|
715
780
|
interface AnalystBenchmarkCommandConfig {
|
|
@@ -730,18 +795,18 @@ interface AnalystBenchmarkCommandConfig {
|
|
|
730
795
|
rlmSamples: number;
|
|
731
796
|
maxCostUsd: number;
|
|
732
797
|
maxArtifactBytes: number;
|
|
733
|
-
|
|
798
|
+
modelOwnerModule: string;
|
|
734
799
|
command: string;
|
|
735
800
|
resume: boolean;
|
|
736
801
|
}
|
|
737
802
|
declare function runAnalystBenchmarkCommand(argv: readonly string[], env?: NodeJS.ProcessEnv, dependencies?: AnalystBenchmarkCommandDependencies): Promise<number>;
|
|
738
|
-
declare const ANALYST_BENCHMARK_HELP = "agent-eval analyst-benchmark\n\nRun the recursive DSPy trace analyst against public AgentRx or CodeTraceBench labels.\n\nRequired:\n --dataset agentrx|codetracebench\n --analyst dspy-rlm|direct Scored analyst. Default: dspy-rlm.\n 'direct' is the
|
|
803
|
+
declare const ANALYST_BENCHMARK_HELP = "agent-eval analyst-benchmark\n\nRun the recursive DSPy trace analyst against public AgentRx or CodeTraceBench labels.\n\nRequired:\n --dataset agentrx|codetracebench\n --analyst dspy-rlm|direct Scored analyst. Default: dspy-rlm.\n 'direct' is the one-shot comparison arm.\n --labels <dataset.json|dataset.jsonl>\n --trace-dir <one-trace-per-file OTLP JSONL directory>\n --artifact-dir <extracted artifact root> Required for CodeTraceBench\n --out <new output directory>\n --revision <full 40- or 64-character hex digest>\n --split <dataset split>\n --model-owner-module <module> Module exporting createModelExecutionOwner;\n the owner keeps provider credentials and policy\n --model <provider model id>\n --limit <positive case count>\n\nControls:\n --resume Continue an interrupted run in --out\n --seed <integer> Case-selection and comparison seed. Default: 0\n --concurrency <positive integer> Parallel benchmark jobs. Default: 1\n --repetitions <positive integer> Runs per case and runner. Default: 1\n --rlm-samples <positive integer> Recursive-engine runs per case; above 1 the\n step-level majority consensus is scored\n (CodeTraceBench + dspy-rlm only). Default: 1\n --instructions-file <path> Replace the recursive analyst instructions\n with this file's text (dspy-rlm only). The\n recorded protocol digest binds the stock\n protocol to the override text, and\n result.json records instructionsOverrideSha256.\n --max-output-tokens <positive> Model output limit per call. Default: 16384\n --max-reasoning-tokens <integer> Reasoning-token limit per call. Default: 65536\n --max-model-requests <positive> Caller-owned model calls per analysis.\n Default: max iterations + model calls + 1\n --max-model-request-bytes <positive> Default: 16777216\n --max-model-response-bytes <positive> Default: 4194304\n --model-request-timeout-ms <positive> Default: --timeout-ms\n --max-iterations <positive> Recursive iterations per analysis. Default: 14\n --max-llm-calls <positive> DSPy model calls per analysis. Default: 8\n --max-tool-calls <positive> Trace-tool calls per analysis. Default: 80\n --max-analysis-output-chars <positive> Default: 8000\n --trace-tool-request-bytes <positive> Default: 1000000\n --trace-tool-response-bytes <positive> Default: 4000000\n --trace-tool-timeout-ms <positive> Default: 60000\n --max-process-input-bytes <positive> Default: 67108864\n --max-process-result-bytes <positive> Default: 4194304\n --max-process-output-chars <positive> Default: 64000\n --python <executable> Python with agent-eval-rpc[dspy]. Default: python\n --timeout-ms <positive> Model analyst deadline per case. Default: 300000\n --max-cost-usd <positive> Run-wide spend limit. Default: 5\n --max-artifact-bytes <positive> Final evidence bytes per case. Default: 8388608\n\nWrites result.json with every observation, metric, usage field, error, comparison,\ninput digest, artifact digest, case distribution, selected case id, and explicit\nunknown cost. Limited deterministic-hash subsets are marked non-representative.\nCompleted observations are fsynced to observations.jsonl. Shareable output is in\nresult.json and report.md. Machine-local paths, execution-owner module, and command\nare isolated in run.local.json. Provider credentials never enter this command.";
|
|
739
804
|
//#endregion
|
|
740
805
|
//#region src/analyst/benchmark-implementation.d.ts
|
|
741
806
|
declare const ANALYST_BENCHMARK_IMPLEMENTATION_DIGEST_ALGORITHM = "sha256-canonical-source-manifest";
|
|
742
807
|
declare const ANALYST_BENCHMARK_DEPENDENCY_LOCK_DIGEST_ALGORITHM = "sha256-canonical-file-manifest";
|
|
743
808
|
declare const ANALYST_BENCHMARK_DEPENDENCY_LOCK_FILES: readonly string[];
|
|
744
|
-
declare const ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256 = "
|
|
809
|
+
declare const ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256 = "f0d788f72ea83b6bd485f2ae61e58dda7e1d87ec79284ce245b3a7f0cc7ca812";
|
|
745
810
|
/** The published benchmark evidence was produced at this package version, by
|
|
746
811
|
* the retired one-shot direct runner, before trace analysts moved to the
|
|
747
812
|
* recursive DSPy RLM engine. Both evidence digests below are historical facts
|
|
@@ -753,7 +818,7 @@ declare const ANALYST_BENCHMARK_EVIDENCE_PACKAGE_VERSION = "0.137.0";
|
|
|
753
818
|
declare const ANALYST_BENCHMARK_EVIDENCE_DEPENDENCY_LOCK_SHA256 = "1e03f2daed356d60316aabefb407ec1e437ac94d408d61eea4ae096e9c6fbb5b";
|
|
754
819
|
declare const ANALYST_BENCHMARK_EVIDENCE_IMPLEMENTATION_SHA256 = "4dba263b6256a30d56c7fdb2d992d3a953c0035d731f359b704db806f68f75ac";
|
|
755
820
|
declare const ANALYST_BENCHMARK_IMPLEMENTATION_FILES: readonly string[];
|
|
756
|
-
declare const ANALYST_BENCHMARK_IMPLEMENTATION_SHA256 = "
|
|
821
|
+
declare const ANALYST_BENCHMARK_IMPLEMENTATION_SHA256 = "67cd83aaaa158d44a9fa32e172cbe68fa97f01ae673ccb117469688b628bb9b1";
|
|
757
822
|
declare function analystBenchmarkImplementationDigest(): string;
|
|
758
823
|
declare function analystBenchmarkDependencyLockDigest(): string;
|
|
759
824
|
//#endregion
|
|
@@ -817,5 +882,5 @@ declare function isProposalFinding(finding: unknown): finding is ProposalFinding
|
|
|
817
882
|
*/
|
|
818
883
|
declare function assertProposalFindings(findings: unknown, context?: string): ReadonlyArray<ProposalFinding>;
|
|
819
884
|
//#endregion
|
|
820
|
-
export { AGENT_RX_UPSTREAM_REVISION, ANALYST_BENCHMARK_COST_LEDGER_FILE, ANALYST_BENCHMARK_DEPENDENCY_LOCK_DIGEST_ALGORITHM, ANALYST_BENCHMARK_DEPENDENCY_LOCK_FILES, ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256, ANALYST_BENCHMARK_EVIDENCE_DEPENDENCY_LOCK_SHA256, ANALYST_BENCHMARK_EVIDENCE_IMPLEMENTATION_SHA256, ANALYST_BENCHMARK_EVIDENCE_PACKAGE_VERSION, ANALYST_BENCHMARK_HELP, ANALYST_BENCHMARK_IMPLEMENTATION_DIGEST_ALGORITHM, ANALYST_BENCHMARK_IMPLEMENTATION_FILES, ANALYST_BENCHMARK_IMPLEMENTATION_SHA256, ANALYST_BENCHMARK_LOCAL_RECEIPT_FILE, ANALYST_BENCHMARK_MANIFEST_FILE, ANALYST_BENCHMARK_OBSERVATIONS_FILE, ANALYST_SEVERITIES, type AgentRxBenchmarkCaseOptions, type AgentRxCalibrationRunnerSummary, type AgentRxCalibrationSummary, type AgentRxFailure, type AgentRxPrediction, type AgentRxPredictionReport, type AgentRxRow, type Analyst, type AnalystBenchmarkArtifact, type AnalystBenchmarkCase, type AnalystBenchmarkCommandConfig, type AnalystBenchmarkCommandDependencies, type AnalystBenchmarkDatasetRef, type AnalystBenchmarkDescriptor, type AnalystBenchmarkError, type AnalystBenchmarkLabelState, type AnalystBenchmarkLocalRunReceipt, type AnalystBenchmarkObservation, type AnalystBenchmarkOutput, type AnalystBenchmarkProgressRow, type AnalystBenchmarkProvenance, type AnalystBenchmarkResult, type AnalystBenchmarkRunIdentity, type AnalystBenchmarkRunManifest, type AnalystBenchmarkRunner, type AnalystBenchmarkSummary, type AnalystComparisonMetric, type AnalystContext, type AnalystCost, type AnalystEvidenceExpectation, type AnalystEvidenceResolution, type AnalystEvidenceResolutionError, type AnalystEvidenceResolver, type AnalystFinding, type AnalystFindingScore, type AnalystHooks, type AnalystInputKind, type AnalystInstructionsOverride, type AnalystIssueExpectation, type AnalystLatencyDistribution, type AnalystMetricComparison, AnalystRegistry, type AnalystRegistryOptions, type AnalystRequirements, type AnalystRunEvent, type AnalystRunInputs, type AnalystRunResult, type AnalystRunSummary, type AnalystRunnerComparison, type AnalystSeverity, type AnalystUsageReceipt, type BehavioralAnalystOptions, type BudgetPolicy, CODE_TRACE_BENCH_ANALYST_PROMPT, CONTROL_INTEGRITY_ANALYST, type ChatCallOpts, type ChatClient, type ChatRequest, type ChatResponse, type ChatTransport, type CliBridgeTransportOpts, type CodeTraceBenchCaseOptions, type CodeTraceBenchLabelOptions, type CodeTraceBenchLabelSet, type CodeTraceBenchRow, type CodeTraceBlockDiagnostics, type CodeTraceCalibrationRunnerSummary, type CodeTraceCalibrationSummary, type CodeTraceFailureBlock, type CodeTraceStageAnnotation, type CodeTracerLabelGroup, type CodeTracerPredictionAdapterOptions, type CodeTracerPredictions, type CodeTracerStepLabel, ControlIntegrityAnalyst, type CreateChatClientOpts, type CreateTraceAnalystOptions, type CustomTransportOpts, DEFAULT_MAX_VERIFICATION_ARTIFACT_BYTES, DEFAULT_TRACE_ANALYST_KINDS, DEFAULT_TRACE_ANALYST_LIMITS, type DefaultAnalystRegistryOptions, type DefineCustomAnalystOptions, type DefineExactCustomAnalystOptions, type DiffPolicy, type DirectProviderTransportOpts, type DspyRlmTraceEngineOptions, type EvidenceRef, type ExactAnalystBudgetPolicy, type ExactAnalystBudgetSnapshot, type ExactAnalystExecutionPlanSnapshot, type ExactAnalystRunCompletion, type ExactAnalystRunEvent, ExactAnalystRunExecutionError, type ExactAnalystRunPolicySnapshot, type ExactAnalystRunResult, type ExactAnalystRunSummary, type ExactAnalystSnapshot, type ExactCapableAnalyst, type ExactExecutionComponentIdentity, type ExactExecutionComponentSnapshot, type ExactRegistryRunOpts, FAILURE_MODE_KIND_SPEC, FINDING_SUBJECT_GRAMMAR_PROMPT, FINDING_SUBJECT_KINDS, FINDING_SUBJECT_SYNTAX, type FindingSubject, type FindingSubjectKind, FindingSubjectStringSchema, type FindingsDiff, FindingsStore, IMPROVEMENT_KIND_SPEC, type JudgeAdapterOpts, KIND_EXPECTED_SUBJECTS, KNOWLEDGE_GAP_KIND_SPEC, KNOWLEDGE_POISONING_KIND_SPEC, type LoadedVerificationArtifacts, MAX_INCORRECT_BLOCKS, MAX_INCORRECT_BLOCK_STEPS, type MockTransportOpts, type PersistedFinding, type PreparedPublicAnalystBenchmark, type ProposalFinding, type ProposalFindingOrigin, type PublicAnalystBenchmarkDataset, type PublicAnalystBenchmarkModelConfig, type PublicBenchmarkDistributions, type PublicBenchmarkSelectionReport, type PublicBenchmarkValueDistribution, RAW_FINDING_SCHEMA_PROMPT, type RawAnalystEvidence, RawAnalystEvidenceSchema, type RawAnalystFinding, RawAnalystFindingSchema, type RegistryRunOpts, type RouterTransportOpts, type RunAnalystBenchmarkOptions, type RunCriticAdapterOpts, SKILL_USAGE_ANALYST, type SandboxSdkTransportOpts, type SemanticConceptJudgeAdapterOpts, SkillUsageAnalyst, type SkillUsageRecord, type SkillUsageReport, type SkillUsageScanConfig, type StepLabelAdapterOptions, type TraceAnalysisEngine, type TraceAnalysisEngineRequest, type TraceAnalysisEngineResult, type TraceAnalystDefinition, type TraceAnalystLimits, type TraceToolGroupName, type UpstreamPredictionAdapterOptions, type VerificationArtifactFile, type VerificationArtifactManifest, type VerificationArtifactRole, type VerificationAvailabilitySummary, type VerificationOutcome, type VerificationOutcomeSource, type VerificationOutcomeStatus, type VerificationResultFile, type VerifierAdapterOpts, adaptPublicBenchmarkFindings, agentRxBenchmarkCase, agentRxPredictionsToFindings, analystBenchmarkDependencyLockDigest, analystBenchmarkImplementationDigest, analystInstructionsOverrideFromText, appendVerificationArtifactsToOtlp, assertExactRegistryRunOpts, assertProposalFindings, behavioralAnalyst, buildDefaultAnalystRegistry, buildSkillUsageReport, buildTraceToolsForGroup, codeTraceBenchCase, codeTracerPredictionsToFindings, coerceJson, coerceToFindingRows, compareAnalystRunners, computeFindingId, createChatClient, createDspyRlmTraceEngine, createJudgeAdapter, createPublicBenchmarkDirectRunner, createPublicBenchmarkRlmRunner, createRunCriticAdapter, createSemanticConceptJudgeAdapter, createTraceAnalyst, createVerifierAdapter, defaultIsMaterial, defineCustomAnalyst, defineTraceAnalyst, deriveEfficiencyFindings, diffFindings, effectiveAnalystProtocolSha256, emitControlIntegrityFindings, emitSkillUsageFindings, emptyPublicBenchmarkRunner, evidenceRefsFromRawFinding, expandCodeTraceFailureBlocks, findingSubjectGrammarPromptFor, isProposalFinding, liftSeverity, loadCodeTraceVerificationArtifacts, loadPublicBenchmarkRows, makeFinding, makeProposalFinding, normalizeAgentRxCategory, normalizeBenchmarkLabel, parseFindingSubject, parseRawFinding, parseVerificationOutcome, preparePublicAnalystBenchmark, publicBenchmarkDistributions, publicBenchmarkProtocolSha256, publicBenchmarkRlmInstructions, publicBenchmarkSelectionReport, publicBenchmarkSystemPrompt, readAnalystBenchmarkArtifact, readAnalystInstructionsOverride, registryBenchmarkRunner, renderAgentRxCalibrationMarkdown, renderAnalystBenchmarkMarkdown, renderCodeTraceCalibrationMarkdown, renderFindingSubject, renderPriorFindings, renderUpstreamFindings, resolveTraceAnalystLimits, roundAgentRxStep, runAnalystBenchmark, runAnalystBenchmarkCommand, runTraceAnalyst, scoreAnalystFindings, selectPublicBenchmarkRows, stripCodeFences, summarizeAgentRxCalibration, summarizeAnalystBenchmarkRunner, summarizeCodeTraceCalibration, traceStoreEvidenceResolver };
|
|
885
|
+
export { AGENT_RX_UPSTREAM_REVISION, ANALYST_BENCHMARK_COST_LEDGER_FILE, ANALYST_BENCHMARK_DEPENDENCY_LOCK_DIGEST_ALGORITHM, ANALYST_BENCHMARK_DEPENDENCY_LOCK_FILES, ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256, ANALYST_BENCHMARK_EVIDENCE_DEPENDENCY_LOCK_SHA256, ANALYST_BENCHMARK_EVIDENCE_IMPLEMENTATION_SHA256, ANALYST_BENCHMARK_EVIDENCE_PACKAGE_VERSION, ANALYST_BENCHMARK_HELP, ANALYST_BENCHMARK_IMPLEMENTATION_DIGEST_ALGORITHM, ANALYST_BENCHMARK_IMPLEMENTATION_FILES, ANALYST_BENCHMARK_IMPLEMENTATION_SHA256, ANALYST_BENCHMARK_LOCAL_RECEIPT_FILE, ANALYST_BENCHMARK_MANIFEST_FILE, ANALYST_BENCHMARK_OBSERVATIONS_FILE, ANALYST_SEVERITIES, type AgentRxBenchmarkCaseOptions, type AgentRxCalibrationRunnerSummary, type AgentRxCalibrationSummary, type AgentRxFailure, type AgentRxPrediction, type AgentRxPredictionReport, type AgentRxRow, type Analyst, type AnalystBenchmarkArtifact, type AnalystBenchmarkCase, type AnalystBenchmarkCommandConfig, type AnalystBenchmarkCommandDependencies, type AnalystBenchmarkDatasetRef, type AnalystBenchmarkDescriptor, type AnalystBenchmarkError, type AnalystBenchmarkLabelState, type AnalystBenchmarkLocalRunReceipt, type AnalystBenchmarkObservation, type AnalystBenchmarkOutput, type AnalystBenchmarkProgressRow, type AnalystBenchmarkProvenance, type AnalystBenchmarkResult, type AnalystBenchmarkRunIdentity, type AnalystBenchmarkRunManifest, type AnalystBenchmarkRunner, type AnalystBenchmarkSummary, type AnalystComparisonMetric, type AnalystContext, type AnalystCost, type AnalystEvidenceExpectation, type AnalystEvidenceResolution, type AnalystEvidenceResolutionError, type AnalystEvidenceResolver, type AnalystFinding, type AnalystFindingScore, type AnalystHooks, type AnalystInputKind, type AnalystInstructionsOverride, type AnalystIssueExpectation, type AnalystLatencyDistribution, type AnalystMetricComparison, AnalystRegistry, type AnalystRegistryOptions, type AnalystRequirements, type AnalystRunEvent, type AnalystRunInputs, type AnalystRunResult, type AnalystRunSummary, type AnalystRunnerComparison, type AnalystSeverity, type AnalystUsageReceipt, type BehavioralAnalystOptions, type BudgetPolicy, CODE_TRACE_BENCH_ANALYST_PROMPT, CONTROL_INTEGRITY_ANALYST, type ChatCallOpts, type ChatClient, type ChatRequest, type ChatResponse, type ChatTransport, type CliBridgeTransportOpts, type CodeTraceBenchCaseOptions, type CodeTraceBenchLabelOptions, type CodeTraceBenchLabelSet, type CodeTraceBenchRow, type CodeTraceBlockDiagnostics, type CodeTraceCalibrationRunnerSummary, type CodeTraceCalibrationSummary, type CodeTraceFailureBlock, type CodeTraceStageAnnotation, type CodeTracerLabelGroup, type CodeTracerPredictionAdapterOptions, type CodeTracerPredictions, type CodeTracerStepLabel, ControlIntegrityAnalyst, type CreateChatClientOpts, type CreateTraceAnalystOptions, type CustomTransportOpts, DEFAULT_MAX_VERIFICATION_ARTIFACT_BYTES, DEFAULT_TRACE_ANALYST_KINDS, DEFAULT_TRACE_ANALYST_LIMITS, type DefaultAnalystRegistryOptions, type DefineCustomAnalystOptions, type DefineExactCustomAnalystOptions, type DiffPolicy, type DirectProviderTransportOpts, type DspyRlmTraceEngineOptions, type EvidenceRef, type ExactAnalystBudgetPolicy, type ExactAnalystBudgetSnapshot, type ExactAnalystExecutionPlanSnapshot, type ExactAnalystRunCompletion, type ExactAnalystRunEvent, ExactAnalystRunExecutionError, type ExactAnalystRunPolicySnapshot, type ExactAnalystRunResult, type ExactAnalystRunSummary, type ExactAnalystSnapshot, type ExactCapableAnalyst, type ExactExecutionComponentIdentity, type ExactExecutionComponentSnapshot, type ExactRegistryRunOpts, FAILURE_MODE_KIND_SPEC, FINDING_SUBJECT_GRAMMAR_PROMPT, FINDING_SUBJECT_KINDS, FINDING_SUBJECT_SYNTAX, type FindingSubject, type FindingSubjectKind, FindingSubjectStringSchema, type FindingsDiff, FindingsStore, IMPROVEMENT_KIND_SPEC, type JudgeAdapterOpts, KIND_EXPECTED_SUBJECTS, KNOWLEDGE_GAP_KIND_SPEC, KNOWLEDGE_POISONING_KIND_SPEC, type LoadedVerificationArtifacts, MAX_INCORRECT_BLOCKS, MAX_INCORRECT_BLOCK_STEPS, type MockTransportOpts, type PersistedFinding, type PreparedPublicAnalystBenchmark, type ProposalFinding, type ProposalFindingOrigin, type PublicAnalystBenchmarkDataset, type PublicAnalystBenchmarkModelConfig, type PublicAnalystBenchmarkModelOwner, type PublicBenchmarkDistributions, type PublicBenchmarkSelectionReport, type PublicBenchmarkValueDistribution, RAW_FINDING_SCHEMA_PROMPT, type RawAnalystEvidence, RawAnalystEvidenceSchema, type RawAnalystFinding, RawAnalystFindingSchema, type RegistryRunOpts, type RouterTransportOpts, type RunAnalystBenchmarkOptions, type RunCriticAdapterOpts, SKILL_USAGE_ANALYST, type SandboxSdkTransportOpts, type SemanticConceptJudgeAdapterOpts, SkillUsageAnalyst, type SkillUsageRecord, type SkillUsageReport, type SkillUsageScanConfig, type StepLabelAdapterOptions, type TraceAnalysisEngine, type TraceAnalysisEngineRequest, type TraceAnalysisEngineResult, type TraceAnalystDefinition, type TraceAnalystLimits, type TraceToolGroupName, type UpstreamPredictionAdapterOptions, type VerificationArtifactFile, type VerificationArtifactManifest, type VerificationArtifactRole, type VerificationAvailabilitySummary, type VerificationOutcome, type VerificationOutcomeSource, type VerificationOutcomeStatus, type VerificationResultFile, type VerifierAdapterOpts, adaptPublicBenchmarkFindings, agentRxBenchmarkCase, agentRxPredictionsToFindings, analystBenchmarkDependencyLockDigest, analystBenchmarkImplementationDigest, analystInstructionsOverrideFromText, appendVerificationArtifactsToOtlp, assertExactRegistryRunOpts, assertProposalFindings, behavioralAnalyst, buildDefaultAnalystRegistry, buildSkillUsageReport, buildTraceToolsForGroup, codeTraceBenchCase, codeTracerPredictionsToFindings, coerceJson, coerceToFindingRows, compareAnalystRunners, computeFindingId, createChatClient, createDspyRlmTraceEngine, createJudgeAdapter, createPublicBenchmarkDirectRunner, createPublicBenchmarkRlmRunner, createRunCriticAdapter, createSemanticConceptJudgeAdapter, createTraceAnalyst, createVerifierAdapter, defaultIsMaterial, defineCustomAnalyst, defineTraceAnalyst, deriveEfficiencyFindings, diffFindings, effectiveAnalystProtocolSha256, emitControlIntegrityFindings, emitSkillUsageFindings, emptyPublicBenchmarkRunner, evidenceRefsFromRawFinding, expandCodeTraceFailureBlocks, findingSubjectGrammarPromptFor, isProposalFinding, liftSeverity, loadCodeTraceVerificationArtifacts, loadPublicBenchmarkRows, makeFinding, makeProposalFinding, normalizeAgentRxCategory, normalizeBenchmarkLabel, parseFindingSubject, parseRawFinding, parseVerificationOutcome, preparePublicAnalystBenchmark, publicBenchmarkDistributions, publicBenchmarkProtocolSha256, publicBenchmarkRlmInstructions, publicBenchmarkSelectionReport, publicBenchmarkSystemPrompt, readAnalystBenchmarkArtifact, readAnalystInstructionsOverride, registryBenchmarkRunner, renderAgentRxCalibrationMarkdown, renderAnalystBenchmarkMarkdown, renderCodeTraceCalibrationMarkdown, renderFindingSubject, renderPriorFindings, renderUpstreamFindings, resolveTraceAnalystLimits, roundAgentRxStep, runAnalystBenchmark, runAnalystBenchmarkCommand, runTraceAnalyst, scoreAnalystFindings, selectPublicBenchmarkRows, stripCodeFences, summarizeAgentRxCalibration, summarizeAnalystBenchmarkRunner, summarizeCodeTraceCalibration, traceStoreEvidenceResolver };
|
|
821
886
|
//# sourceMappingURL=index.d.ts.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"index.d.ts","names":[],"sources":["../../src/analyst/adapters.ts","../../src/analyst/benchmark-agentrx-calibration.ts","../../src/analyst/benchmark-dataset-types.ts","../../src/analyst/benchmark-dataset-agentrx.ts","../../src/analyst/benchmark-dataset-codetrace.ts","../../src/analyst/benchmark-dataset-utils.ts","../../src/analyst/benchmark-verification-outcome.ts","../../src/analyst/benchmark-verification-artifacts.ts","../../src/analyst/benchmark-public-types.ts","../../src/analyst/benchmark-public-adapters.ts","../../src/analyst/benchmark-public-data.ts","../../src/analyst/benchmark-public-prompt.ts","../../src/analyst/benchmark-public-model.ts","../../src/analyst/benchmark-public-rlm.ts","../../src/analyst/benchmark-comparison.ts","../../src/analyst/benchmark-public-calibration.ts","../../src/analyst/benchmark-command-artifact.ts","../../src/analyst/benchmark-command-result.ts","../../src/analyst/benchmark-command.ts","../../src/analyst/benchmark-implementation.ts","../../src/analyst/benchmark-instructions-override.ts","../../src/analyst/benchmark-report.ts","../../src/analyst/benchmark-summary.ts","../../src/analyst/parse-tolerant.ts","../../src/analyst/proposal-findings.ts"],"mappings":";;;;;;;;;;;;iBA4CgB,aAAa,GAAG,WAAgB;UAe/B,oBAAoB;EACnC;EACA;EACA,UAAU,mBAAmB;;;;;EAK7B,UAAU,KAAK,cAAc;;iBAGf,sBAAsB,KAAK,MAAM,oBAAoB,OAAO,QAAQ;UAwEnE;EACf;EACA;EACA,SAAS;;EAET;;iBAGc,uBAAuB,OAAM,uBAA4B,QAAQ;UAiEhE;EACf;EACA;EACA,OAAO;;EAEP,MAAM;;EAEN,OAAO;;EAEP;;iBAGc,mBAAmB,MAAM,mBAAmB,QAAQ;UAiDnD;EACf;EACA;;EAEA,UAAU,KAAK;;EAEf;;iBAGc,kCACd,OAAM,kCACL,QAAQ;;;cC5RE;UAEI;EACf;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;;UAGe;EACf;EACA;EACA;EACA,SAAS;;iBAGK,4BACd,QAAQ,wBACR,2BACC;iBAkBa,iCAAiC,SAAS;;;KCtD9C;UAEK;EACf,YAAY;EACZ;EACA;EACA;EACA;EACA;;UAGe;EACf,eAAe;EACf,mBAAmB;EACnB;IAAe,YAAY;IAAY;;EACvC,wBAAwB;EACxB;EACA;EACA;;UAGe;EACf,UAAU;EACV;EACA;EACA;EACA;;UAGe;EACf,UAAU;EACV,mBAAmB;EACnB;EACA;EACA;EACA;EACA;;UAGe;EACf;EACA;EACA;EACA;;UAGe;EACf;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;;UAGe;EACf;EACA;EACA;EACA,iBAAiB;;KAGP,0CAEC,sCACA,iCACA;UAEI;EACf;EACA;EACA;EACA;;EAEA;EACA;EACA;EACA;EACA;EACA;EACA,oCAAoC;;UAGrB;EACf,eAAe;EACf,WAAW,sBAAsB;;KAGvB;UAEK;;;;;EAKf,WAAW;;UAGI,kCACP,yBACN;UAEa,oCAAoC;EACnD;;EAEA;;UAGe,yCAAyC;EACxD;EACA;EACA;EACA;;UAGe,2CACP,kCACN;;;iBCtGY,qBAAqB,QACnC,KAAK,YACL,OAAO,QACP,UAAS,8BACR,qBAAqB;;iBAoGR,6BACd,mBAAmB,YACnB,iBACA,UAAS,mCACR;iBA6Ea,yBAAyB;;iBAoNzB,iBAAiB;;;iBC7YjB,mBAAmB,QACjC,KAAK,mBACL,OAAO,QACP,UAAS,4BACR,qBAAqB;;iBAwFR,gCACd,2BACA,aAAa,uBACb,UAAS,qCACR;;;iBClHa,wBAAwB;;;KCA5B;UAEK;EACf;EACA;EACA,QAAQ;;UAGO;EACf,QAAQ;EACR;EAKA;IAAe;IAAe;;EAC9B,SAAS;EACT;EACA;EACA;EACA;;UAGe;EACf;EACA;;iBA8Ec,yBACd,gBAAgB,2BACf;;;cChGU;KAED;UAEK;EACf,MAAM;EACN;EACA;EACA;EACA;EACA;;UAGe;EACf;EACA;EACA,SAAS;EACT;EACA;EACA;EACA;EACA;EACA,OAAO;EACP,cAAc;EACd,UAAU,OAAO;;UAGF;EACf,UAAU;EACV,SAAS;EACT,OAAO,MAAM;IAA6B;;;iBAatB,mCAAmC;EACvD;EACA,KAAK;EACL;IACE,QAAQ;iBAwHI,kCACd,kBACA,iBACA,WAAW,6BACX;;;
|
|
1
|
+
{"version":3,"file":"index.d.ts","names":[],"sources":["../../src/analyst/adapters.ts","../../src/analyst/benchmark-agentrx-calibration.ts","../../src/analyst/benchmark-dataset-types.ts","../../src/analyst/benchmark-dataset-agentrx.ts","../../src/analyst/benchmark-dataset-codetrace.ts","../../src/analyst/benchmark-dataset-utils.ts","../../src/analyst/benchmark-verification-outcome.ts","../../src/analyst/benchmark-verification-artifacts.ts","../../src/analyst/benchmark-public-types.ts","../../src/analyst/benchmark-public-adapters.ts","../../src/analyst/benchmark-public-data.ts","../../src/analyst/benchmark-public-prompt.ts","../../src/analyst/benchmark-public-model.ts","../../src/analyst/benchmark-public-rlm.ts","../../src/analyst/benchmark-comparison.ts","../../src/analyst/benchmark-public-calibration.ts","../../src/analyst/benchmark-command-artifact.ts","../../src/analyst/benchmark-command-result.ts","../../src/analyst/benchmark-command.ts","../../src/analyst/benchmark-implementation.ts","../../src/analyst/benchmark-instructions-override.ts","../../src/analyst/benchmark-report.ts","../../src/analyst/benchmark-summary.ts","../../src/analyst/parse-tolerant.ts","../../src/analyst/proposal-findings.ts"],"mappings":";;;;;;;;;;;;iBA4CgB,aAAa,GAAG,WAAgB;UAe/B,oBAAoB;EACnC;EACA;EACA,UAAU,mBAAmB;;;;;EAK7B,UAAU,KAAK,cAAc;;iBAGf,sBAAsB,KAAK,MAAM,oBAAoB,OAAO,QAAQ;UAwEnE;EACf;EACA;EACA,SAAS;;EAET;;iBAGc,uBAAuB,OAAM,uBAA4B,QAAQ;UAiEhE;EACf;EACA;EACA,OAAO;;EAEP,MAAM;;EAEN,OAAO;;EAEP;;iBAGc,mBAAmB,MAAM,mBAAmB,QAAQ;UAiDnD;EACf;EACA;;EAEA,UAAU,KAAK;;EAEf;;iBAGc,kCACd,OAAM,kCACL,QAAQ;;;cC5RE;UAEI;EACf;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;;UAGe;EACf;EACA;EACA;EACA,SAAS;;iBAGK,4BACd,QAAQ,wBACR,2BACC;iBAkBa,iCAAiC,SAAS;;;KCtD9C;UAEK;EACf,YAAY;EACZ;EACA;EACA;EACA;EACA;;UAGe;EACf,eAAe;EACf,mBAAmB;EACnB;IAAe,YAAY;IAAY;;EACvC,wBAAwB;EACxB;EACA;EACA;;UAGe;EACf,UAAU;EACV;EACA;EACA;EACA;;UAGe;EACf,UAAU;EACV,mBAAmB;EACnB;EACA;EACA;EACA;EACA;;UAGe;EACf;EACA;EACA;EACA;;UAGe;EACf;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;;UAGe;EACf;EACA;EACA;EACA,iBAAiB;;KAGP,0CAEC,sCACA,iCACA;UAEI;EACf;EACA;EACA;EACA;;EAEA;EACA;EACA;EACA;EACA;EACA;EACA,oCAAoC;;UAGrB;EACf,eAAe;EACf,WAAW,sBAAsB;;KAGvB;UAEK;;;;;EAKf,WAAW;;UAGI,kCACP,yBACN;UAEa,oCAAoC;EACnD;;EAEA;;UAGe,yCAAyC;EACxD;EACA;EACA;EACA;;UAGe,2CACP,kCACN;;;iBCtGY,qBAAqB,QACnC,KAAK,YACL,OAAO,QACP,UAAS,8BACR,qBAAqB;;iBAoGR,6BACd,mBAAmB,YACnB,iBACA,UAAS,mCACR;iBA6Ea,yBAAyB;;iBAoNzB,iBAAiB;;;iBC7YjB,mBAAmB,QACjC,KAAK,mBACL,OAAO,QACP,UAAS,4BACR,qBAAqB;;iBAwFR,gCACd,2BACA,aAAa,uBACb,UAAS,qCACR;;;iBClHa,wBAAwB;;;KCA5B;UAEK;EACf;EACA;EACA,QAAQ;;UAGO;EACf,QAAQ;EACR;EAKA;IAAe;IAAe;;EAC9B,SAAS;EACT;EACA;EACA;EACA;;UAGe;EACf;EACA;;iBA8Ec,yBACd,gBAAgB,2BACf;;;cChGU;KAED;UAEK;EACf,MAAM;EACN;EACA;EACA;EACA;EACA;;UAGe;EACf;EACA;EACA,SAAS;EACT;EACA;EACA;EACA;EACA;EACA,OAAO;EACP,cAAc;EACd,UAAU,OAAO;;UAGF;EACf,UAAU;EACV,SAAS;EACT,OAAO,MAAM;IAA6B;;;iBAatB,mCAAmC;EACvD;EACA,KAAK;EACL;IACE,QAAQ;iBAwHI,kCACd,kBACA,iBACA,WAAW,6BACX;;;KC5KU;;;;;;;;UASK;;WAEN;;WAEA;;;UAIM;EACf,MAAM;EACN;EACA,kBAAkB,aAAa;;EAE/B,UAAU;;UAGK;;EAEf,MAAM;;EAEN;;EAEA,kBAAkB,aAAa;EAC/B;EACA;EACA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA,UAAU;;EAEV;;EAEA,uBAAuB;EACvB;IACE,SAAS;IACT;IACA;IACA;IACA;IACA;IACA;IACA;IACA;;;;;;IAMA;;EAEF,aAAa;EACb;IACE;IACA;;;UAIa;EACf,OAAO,qBAAqB;EAC5B;EACA;EACA;EACA,YAAY;IACV;IACA;IACA;;EAEF,uBAAuB;EACvB,WAAW;;UAGI;EACf;EACA;EACA,QAAQ;;UAGO;EACf,OAAO;EACP,OAAO;EACP,OAAO;EACP,YAAY;EACZ,QAAQ;;UAGO;EACf;EACA;EACA;EACA;EACA;EACA;EACA,QAAQ;EACR,UAAU;;;;;;;;;;UCpGK;EACf;EACA;;EAEA;EACA;EACA,UAAU;EACV;EACA;EACA;EACA;EACA,WAAW;;;;;;;;;UAUI;EACf;EACA;;EAEA,kCAAkC;;EAElC;;EAEA;;EAEA;;EAEA;;EAEA;;;;;;;;UASe;EACf;EACA,OAAO;;iBAGO,8BAA8B,uBAAuB;iBAiB/C,6BAA6B;EACjD,SAAS;EACT;EACA,mBAAmB;EACnB;EACA,OAAO;EACP,SAAS;IACP;EACF,UAAU;EACV,aAAa;;EAEb,aAAa;;;;;;;;;iBA4MO,6BAA6B;EACjD;EACA,iBAAiB;EACjB,OAAO;EACP;EACA;EACA,SAAS;IACP;EACF,UAAU;EACV,aAAa;EACb,YAAY;;;;iBCtQQ,wBACpB,eACC,QAAQ,MAAM;iBA2BD,0BACd,SAAS,+BACT,eAAe,2BACf;EAAW;EAAe;IACzB,MAAM;iBAwBO,6BACd,SAAS,+BACT,eAAe,4BACd;iBAyCa,+BACd,SAAS,+BACT,iBAAiB,2BACjB,mBAAmB,2BACnB,eACC;iBAcmB,8BAA8B;EAClD,SAAS;EACT;EACA;EACA;EACA;EACA;EACA;IACE,QAAQ;;;;;;;cCxKC;;;;cAKA;cAMA;;iBAiGG,4BAA4B,SAAS;;iBAerC,+BAA+B,SAAS;;;;iBAaxC,8BAA8B,SAAS;;;;iBCjFvC,kCACd,SAAS,+BACT,QAAQ,oCACP,uBAAuB;;;;iBC/BV,+BACd,SAAS,+BACT,QAAQ,oCACP,uBAAuB;;;KCnCd;UAoBK;EACf,QAAQ;EACR;;EAEA;;EAEA;;EAEA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;;UAGe;EACf;EACA;EACA,SAAS;;iBASK,sBACd,QAAQ,wBACR;EACE;EACA;EACA;EACA;EACA;IAED;;;UCnEc;EACf;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;;EAEA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;;UAGe;EACf;EACA;EACA,SAAS;;iBAGK,8BACd,QAAQ,yBACP;iBAca,mCAAmC,SAAS;;;UCpC3C;EACf;EACA;EACA;IACE,SAAS;IACT;IACA;IACA;IACA;IACA,YAAY;MAAQ;MAAiB;MAAsB;;IAC3D,uBAAuB;IACvB,0BAA0B;IAC1B;MACE;MACA;MACA;MACA,QAAQ;;IAEV;MACE;MACA;;MAEA;MACA;;MAEA;MACA;MACA;MACA;MACA;MACA;MACA;MACA,UAAU;MACV;QACE;QACA;QACA;QACA;QACA;QACA;QACA;QACA;;MAEF;QACE;QACA;QACA;;MAEF;MACA;MACA;;MAEA;MACA;MACA;;;EAGJ,QAAQ;EACR,aAAa;EACb,uBAAuB;EACvB,qBAAqB;;UAGN;EACf;EACA;EACA;EACA;IACE;IACA;IACA;;;UAIa;EACf;IACE,SAAS;IACT;IACA;IACA;MACE;MACA;MACA;MACA;MACA;MACA;MACA;MACA;MACA,SAAS;MACT;QACE;QACA;QACA;QACA;QACA;QACA;QACA;QACA;;MAEF;QACE;QACA;QACA;;;IAGJ;IACA;IACA;IACA;;IAEA;IACA;IACA;IACA;;IAEA;IACA;IACA;IACA;;EAEF;IACE;IACA;IACA;IACA,YAAY;MAAQ;MAAiB;MAAsB;;IAC3D;IACA;;;UAIa;EACf;EACA;EACA;EACA;EACA,UAAU;;UAGK;EACf;EACA;EACA;EACA;IACE;IACA;IACA;IACA;IACA;;EAEF;EACA;IACE;IACA;IACA;;EAEF;IACE;IACA;IACA;IACA;IACA;IACA;;;UAIa;EACf;EACA;EACA;EACA,aAAa;EACb;;cAGW;cACA;cACA;cACA;;;iBC3KS,6BACpB,eACC,QAAQ;;;UCqEM;EACf,uBACE,SAAS,+BACT,QAAQ,sCACL,uBAAuB;EAC5B,2BACE,mBACA;IACE;IACA,aAAa,SAAS,OAAO;QAE5B,QAAQ;;;;;;;KAQH;UAEK;EACf,SAAS;EACT,SAAS;EACT;EACA;EACA;EACA;EACA;EACA;EACA,OAAO;EACP;EACA;EACA;EACA;;EAEA;EACA;EACA;EACA;EACA;EACA;;iBAKoB,2BACpB,yBACA,MAAK,OAAO,YACZ,eAAc,sCACb;cA0TU;;;cCtcA;cAEA;cAEA;cAOA;;;;;;;;cAUA;cAEA;cAGA;cAGA;cAyFA;iBAGG;iBAIA;;;;iBCpHA,oCAAoC,eAAe;;iBAQnD,gCAAgC,eAAe;;;;;;;;;;;iBAyB/C,+BACd,SAAS,+BACT,WAAW,KAAK;;;iBCzCF,+BACd,QAAQ,wBACR,uBAAsB;;;iBCER,gCACd,kBACA,uBAAuB,gCACtB;;;;;;;;;;;;;;;iBCGa,gBAAgB;;;;;iBAgBhB,WAAW;;;;;;;iBAeX,oBAAoB;;;;iBCXpB,kBAAkB,mBAAmB,WAAW;;;;;;iBAShD,uBACd,mBACA,mBACC,cAAc"}
|
package/dist/analyst/index.js
CHANGED
|
@@ -1,12 +1,12 @@
|
|
|
1
1
|
import { i as CostLedger } from "../cost-ledger-DMFxsLKr.js";
|
|
2
|
-
import { C as createChatClient, _ as CONTROL_INTEGRITY_ANALYST, b as behavioralAnalyst, f as DEFAULT_TRACE_ANALYST_KINDS, g as FAILURE_MODE_KIND_SPEC, h as IMPROVEMENT_KIND_SPEC, i as assertExactRegistryRunOpts, m as KNOWLEDGE_GAP_KIND_SPEC, n as AnalystRegistry, p as KNOWLEDGE_POISONING_KIND_SPEC, r as ExactAnalystRunExecutionError, t as buildDefaultAnalystRegistry, v as ControlIntegrityAnalyst, x as deriveEfficiencyFindings, y as emitControlIntegrityFindings } from "../default-registry-
|
|
2
|
+
import { C as createChatClient, _ as CONTROL_INTEGRITY_ANALYST, b as behavioralAnalyst, f as DEFAULT_TRACE_ANALYST_KINDS, g as FAILURE_MODE_KIND_SPEC, h as IMPROVEMENT_KIND_SPEC, i as assertExactRegistryRunOpts, m as KNOWLEDGE_GAP_KIND_SPEC, n as AnalystRegistry, p as KNOWLEDGE_POISONING_KIND_SPEC, r as ExactAnalystRunExecutionError, t as buildDefaultAnalystRegistry, v as ControlIntegrityAnalyst, x as deriveEfficiencyFindings, y as emitControlIntegrityFindings } from "../default-registry-SOyHB6qG.js";
|
|
3
3
|
import { A as coerceJson, B as renderFindingSubject, D as RawAnalystFindingSchema, E as RawAnalystEvidenceSchema, F as FINDING_SUBJECT_SYNTAX, G as resolveTraceAnalystLimits, I as FindingSubjectStringSchema, L as KIND_EXPECTED_SUBJECTS, M as stripCodeFences, N as FINDING_SUBJECT_GRAMMAR_PROMPT, O as evidenceRefsFromRawFinding, P as FINDING_SUBJECT_KINDS, R as findingSubjectGrammarPromptFor, T as RAW_FINDING_SCHEMA_PROMPT, W as DEFAULT_TRACE_ANALYST_LIMITS, a as buildTraceToolsForGroup, i as runTraceAnalyst, j as coerceToFindingRows, k as parseRawFinding, n as renderPriorFindings, r as renderUpstreamFindings, t as createTraceAnalyst, w as ANALYST_SEVERITIES, z as parseFindingSubject } from "../kind-factory-Bvwe3pup.js";
|
|
4
4
|
import { a as computeFindingId, i as validateUsageSettlementTimeout, n as settleUsageReceiptFromCostLedger, o as makeFinding, s as makeProposalFinding } from "../usage-receipt-CgxMEBZq.js";
|
|
5
5
|
import { n as isProposalFinding, t as assertProposalFindings } from "../proposal-findings-2GIUo1et.js";
|
|
6
|
-
import { a as RunCritic, c as buildSkillUsageReport, d as defaultIsMaterial, f as diffFindings, h as defineTraceAnalyst, i as runSemanticConceptJudge, l as emitSkillUsageFindings, m as defineCustomAnalyst, n as SEMANTIC_CONCEPT_JUDGE_VERSION, o as SKILL_USAGE_ANALYST, s as SkillUsageAnalyst, u as FindingsStore } from "../semantic-concept-judge-
|
|
7
|
-
import { t as createDspyRlmTraceEngine } from "../dspy-rlm-engine-
|
|
6
|
+
import { a as RunCritic, c as buildSkillUsageReport, d as defaultIsMaterial, f as diffFindings, h as defineTraceAnalyst, i as runSemanticConceptJudge, l as emitSkillUsageFindings, m as defineCustomAnalyst, n as SEMANTIC_CONCEPT_JUDGE_VERSION, o as SKILL_USAGE_ANALYST, s as SkillUsageAnalyst, u as FindingsStore } from "../semantic-concept-judge-Do5aM9wP.js";
|
|
7
|
+
import { t as createDspyRlmTraceEngine } from "../dspy-rlm-engine-IRCG8kdi.js";
|
|
8
8
|
import { a as scoreAnalystFindings, i as summarizeAnalystBenchmarkRunner, n as runAnalystBenchmark, r as traceStoreEvidenceResolver, t as registryBenchmarkRunner } from "../benchmark-CYtcIF2V.js";
|
|
9
|
-
import { $ as normalizeAgentRxCategory, A as parseVerificationOutcome, B as analystBenchmarkDependencyLockDigest, C as MAX_INCORRECT_BLOCK_STEPS, D as DEFAULT_MAX_VERIFICATION_ARTIFACT_BYTES, E as publicBenchmarkSystemPrompt, F as ANALYST_BENCHMARK_EVIDENCE_IMPLEMENTATION_SHA256, G as ANALYST_BENCHMARK_OBSERVATIONS_FILE, H as ANALYST_BENCHMARK_COST_LEDGER_FILE, I as ANALYST_BENCHMARK_EVIDENCE_PACKAGE_VERSION, J as summarizeAgentRxCalibration, K as AGENT_RX_UPSTREAM_REVISION, L as ANALYST_BENCHMARK_IMPLEMENTATION_DIGEST_ALGORITHM, M as ANALYST_BENCHMARK_DEPENDENCY_LOCK_FILES, N as ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256, O as appendVerificationArtifactsToOtlp, P as ANALYST_BENCHMARK_EVIDENCE_DEPENDENCY_LOCK_SHA256, Q as agentRxPredictionsToFindings, R as ANALYST_BENCHMARK_IMPLEMENTATION_FILES, S as MAX_INCORRECT_BLOCKS, T as publicBenchmarkRlmInstructions, U as ANALYST_BENCHMARK_LOCAL_RECEIPT_FILE, V as analystBenchmarkImplementationDigest, W as ANALYST_BENCHMARK_MANIFEST_FILE, X as codeTracerPredictionsToFindings, Y as codeTraceBenchCase, Z as agentRxBenchmarkCase, _ as compareAnalystRunners, a as preparePublicAnalystBenchmark, b as readAnalystInstructionsOverride, c as selectPublicBenchmarkRows, d as adaptPublicBenchmarkFindings, et as roundAgentRxStep, f as emptyPublicBenchmarkRunner, g as summarizeCodeTraceCalibration, h as renderCodeTraceCalibrationMarkdown, i as loadPublicBenchmarkRows, j as ANALYST_BENCHMARK_DEPENDENCY_LOCK_DIGEST_ALGORITHM, k as loadCodeTraceVerificationArtifacts, l as createPublicBenchmarkRlmRunner, m as readAnalystBenchmarkArtifact, n as runAnalystBenchmarkCommand, o as publicBenchmarkDistributions, p as expandCodeTraceFailureBlocks, q as renderAgentRxCalibrationMarkdown, r as renderAnalystBenchmarkMarkdown, s as publicBenchmarkSelectionReport, t as ANALYST_BENCHMARK_HELP, tt as normalizeBenchmarkLabel, u as createPublicBenchmarkDirectRunner, v as analystInstructionsOverrideFromText, w as publicBenchmarkProtocolSha256, x as CODE_TRACE_BENCH_ANALYST_PROMPT, y as effectiveAnalystProtocolSha256, z as ANALYST_BENCHMARK_IMPLEMENTATION_SHA256 } from "../benchmark-command-
|
|
9
|
+
import { $ as normalizeAgentRxCategory, A as parseVerificationOutcome, B as analystBenchmarkDependencyLockDigest, C as MAX_INCORRECT_BLOCK_STEPS, D as DEFAULT_MAX_VERIFICATION_ARTIFACT_BYTES, E as publicBenchmarkSystemPrompt, F as ANALYST_BENCHMARK_EVIDENCE_IMPLEMENTATION_SHA256, G as ANALYST_BENCHMARK_OBSERVATIONS_FILE, H as ANALYST_BENCHMARK_COST_LEDGER_FILE, I as ANALYST_BENCHMARK_EVIDENCE_PACKAGE_VERSION, J as summarizeAgentRxCalibration, K as AGENT_RX_UPSTREAM_REVISION, L as ANALYST_BENCHMARK_IMPLEMENTATION_DIGEST_ALGORITHM, M as ANALYST_BENCHMARK_DEPENDENCY_LOCK_FILES, N as ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256, O as appendVerificationArtifactsToOtlp, P as ANALYST_BENCHMARK_EVIDENCE_DEPENDENCY_LOCK_SHA256, Q as agentRxPredictionsToFindings, R as ANALYST_BENCHMARK_IMPLEMENTATION_FILES, S as MAX_INCORRECT_BLOCKS, T as publicBenchmarkRlmInstructions, U as ANALYST_BENCHMARK_LOCAL_RECEIPT_FILE, V as analystBenchmarkImplementationDigest, W as ANALYST_BENCHMARK_MANIFEST_FILE, X as codeTracerPredictionsToFindings, Y as codeTraceBenchCase, Z as agentRxBenchmarkCase, _ as compareAnalystRunners, a as preparePublicAnalystBenchmark, b as readAnalystInstructionsOverride, c as selectPublicBenchmarkRows, d as adaptPublicBenchmarkFindings, et as roundAgentRxStep, f as emptyPublicBenchmarkRunner, g as summarizeCodeTraceCalibration, h as renderCodeTraceCalibrationMarkdown, i as loadPublicBenchmarkRows, j as ANALYST_BENCHMARK_DEPENDENCY_LOCK_DIGEST_ALGORITHM, k as loadCodeTraceVerificationArtifacts, l as createPublicBenchmarkRlmRunner, m as readAnalystBenchmarkArtifact, n as runAnalystBenchmarkCommand, o as publicBenchmarkDistributions, p as expandCodeTraceFailureBlocks, q as renderAgentRxCalibrationMarkdown, r as renderAnalystBenchmarkMarkdown, s as publicBenchmarkSelectionReport, t as ANALYST_BENCHMARK_HELP, tt as normalizeBenchmarkLabel, u as createPublicBenchmarkDirectRunner, v as analystInstructionsOverrideFromText, w as publicBenchmarkProtocolSha256, x as CODE_TRACE_BENCH_ANALYST_PROMPT, y as effectiveAnalystProtocolSha256, z as ANALYST_BENCHMARK_IMPLEMENTATION_SHA256 } from "../benchmark-command-CQPKRUr-.js";
|
|
10
10
|
//#region src/analyst/adapters.ts
|
|
11
11
|
/**
|
|
12
12
|
* Adapter factories — lift each existing agent-eval primitive into the
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
import { r as observedSplitScore } from "./reward-nw2xZGZG.js";
|
|
2
2
|
import { D as pairedCohensDz, E as pairedBootstrap, L as pearsonR, P as pairedTTest, V as spearmanR, Z as continuousAgreement, k as pairedMde, z as requiredPairedSampleSize } from "./statistics-ByxzSiOM.js";
|
|
3
3
|
import { r as pairRunRecords } from "./paired-arms-iZ08VFMN.js";
|
|
4
|
-
import { s as validateRunRecord } from "./run-record-
|
|
4
|
+
import { s as validateRunRecord } from "./run-record-CWN8-VsV.js";
|
|
5
5
|
import { r as welchsTTest } from "./baseline-C-GocmIW.js";
|
|
6
6
|
import { o as llmSpans } from "./query-Di7eEQ79.js";
|
|
7
7
|
import { r as paretoChart } from "./summary-report-9A5y7EsK.js";
|
|
@@ -1079,4 +1079,4 @@ function buildRecommendations(ctx) {
|
|
|
1079
1079
|
//#endregion
|
|
1080
1080
|
export { checkBehavioralCanary as a, canaryLeakView as i, summarizeExecution as n, checkCanaries as o, HoldoutAuditor as r, runBehavioralCanaries as s, analyzeRuns as t };
|
|
1081
1081
|
|
|
1082
|
-
//# sourceMappingURL=analyze-runs-
|
|
1082
|
+
//# sourceMappingURL=analyze-runs-DWIvOAGk.js.map
|