@tangle-network/agent-eval 0.138.0 → 0.139.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +30 -0
- package/README.md +2 -1
- package/dist/analyst/index.d.ts +41 -94
- package/dist/analyst/index.d.ts.map +1 -1
- package/dist/analyst/index.js +9 -24
- package/dist/analyst/index.js.map +1 -1
- package/dist/{benchmark-D8dkki-J.js → benchmark-CYtcIF2V.js} +2 -2
- package/dist/{benchmark-D8dkki-J.js.map → benchmark-CYtcIF2V.js.map} +1 -1
- package/dist/{benchmark-DlQgU_XI.d.ts → benchmark-DDVdWcwA.d.ts} +3 -3
- package/dist/{benchmark-DlQgU_XI.d.ts.map → benchmark-DDVdWcwA.d.ts.map} +1 -1
- package/dist/{benchmark-command-CMqVqReF.js → benchmark-command-BKfjOBJ5.js} +243 -38
- package/dist/benchmark-command-BKfjOBJ5.js.map +1 -0
- package/dist/benchmarks/index.d.ts +1 -1
- package/dist/benchmarks/index.js +1 -1
- package/dist/{benchmarks-BJ_xK5rQ.js → benchmarks-zxhy1QV3.js} +4 -4
- package/dist/{benchmarks-BJ_xK5rQ.js.map → benchmarks-zxhy1QV3.js.map} +1 -1
- package/dist/campaign/index.d.ts +5 -5
- package/dist/campaign/index.js +4 -3
- package/dist/{campaign-BIBS-NHV.js → campaign-DrS6_hLd.js} +10 -9
- package/dist/campaign-DrS6_hLd.js.map +1 -0
- package/dist/canonical-D011XM8r.js +86 -0
- package/dist/canonical-D011XM8r.js.map +1 -0
- package/dist/cli.js +3 -3
- package/dist/{client-BwPKohkJ.d.ts → client-BohnDFBq.d.ts} +4 -4
- package/dist/{client-BwPKohkJ.d.ts.map → client-BohnDFBq.d.ts.map} +1 -1
- package/dist/{completion-verifier-B4-IMYcS.d.ts → completion-verifier-IPoP4fQO.d.ts} +178 -4
- package/dist/completion-verifier-IPoP4fQO.d.ts.map +1 -0
- package/dist/contract/index.d.ts +10 -10
- package/dist/contract/index.js +8 -7
- package/dist/contract/index.js.map +1 -1
- package/dist/control.d.ts +2 -2
- package/dist/{cost-ledger-CHDLA0Ss.js → cost-ledger-CZ9diLxY.js} +7 -7
- package/dist/cost-ledger-CZ9diLxY.js.map +1 -0
- package/dist/{cost-ledger-B1D3COAc.d.ts → cost-ledger-DKgyIWRj.d.ts} +5 -2
- package/dist/cost-ledger-DKgyIWRj.d.ts.map +1 -0
- package/dist/default-registry-B8vf7Rmf.d.ts +118 -0
- package/dist/default-registry-B8vf7Rmf.d.ts.map +1 -0
- package/dist/{default-registry-lp5R0lve.js → default-registry-BgJJItGr.js} +57 -1532
- package/dist/default-registry-BgJJItGr.js.map +1 -0
- package/dist/dspy-rlm-engine-DTkVyDX-.js +344 -0
- package/dist/dspy-rlm-engine-DTkVyDX-.js.map +1 -0
- package/dist/{eval-campaign-9MozgKL7.js → eval-campaign-BmptJj50.js} +2 -2
- package/dist/{eval-campaign-9MozgKL7.js.map → eval-campaign-BmptJj50.js.map} +1 -1
- package/dist/{exact-types-Dpw2LeHA.d.ts → exact-types-MaaFcllV.d.ts} +2 -2
- package/dist/{exact-types-Dpw2LeHA.d.ts.map → exact-types-MaaFcllV.d.ts.map} +1 -1
- package/dist/external-optimizer-contracts-BrxY2Sli.d.ts +32 -0
- package/dist/external-optimizer-contracts-BrxY2Sli.d.ts.map +1 -0
- package/dist/{extract-usage-CS391dOE.js → extract-usage-DZs601Va.js} +2 -2
- package/dist/{extract-usage-CS391dOE.js.map → extract-usage-DZs601Va.js.map} +1 -1
- package/dist/{feedback-trajectory-CoNep7rl.d.ts → feedback-trajectory-BJUWOkJM.d.ts} +3 -3
- package/dist/{feedback-trajectory-CoNep7rl.d.ts.map → feedback-trajectory-BJUWOkJM.d.ts.map} +1 -1
- package/dist/fuzz.d.ts +1 -1
- package/dist/fuzz.js +1 -1
- package/dist/{hf-dataset-DBJXXoY1.js → hf-dataset-XggBupCr.js} +2 -2
- package/dist/{hf-dataset-DBJXXoY1.js.map → hf-dataset-XggBupCr.js.map} +1 -1
- package/dist/hosted/index.d.ts +3 -3
- package/dist/{index-D0cxAdaV.d.ts → index-BTm_P9aC.d.ts} +11 -11
- package/dist/{index-D0cxAdaV.d.ts.map → index-BTm_P9aC.d.ts.map} +1 -1
- package/dist/{index-B2-IxCMB.d.ts → index-CWOPCJiw.d.ts} +2 -2
- package/dist/{index-B2-IxCMB.d.ts.map → index-CWOPCJiw.d.ts.map} +1 -1
- package/dist/{index-sMN_hI4E.d.ts → index-CtR1xh4V.d.ts} +3 -3
- package/dist/{index-sMN_hI4E.d.ts.map → index-CtR1xh4V.d.ts.map} +1 -1
- package/dist/{index-CjVYlVBK.d.ts → index-_66rVpwN.d.ts} +5 -5
- package/dist/{index-CjVYlVBK.d.ts.map → index-_66rVpwN.d.ts.map} +1 -1
- package/dist/index.d.ts +35 -56
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +51 -176
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-CXd8VBDR.d.ts → insight-report-Bu5Wi9tG.d.ts} +4 -4
- package/dist/{insight-report-CXd8VBDR.d.ts.map → insight-report-Bu5Wi9tG.d.ts.map} +1 -1
- package/dist/{integrity-B-MLFz0I.d.ts → integrity-COTh3DTH.d.ts} +2 -2
- package/dist/{integrity-B-MLFz0I.d.ts.map → integrity-COTh3DTH.d.ts.map} +1 -1
- package/dist/kind-factory-CFxA0JQX.js +2133 -0
- package/dist/kind-factory-CFxA0JQX.js.map +1 -0
- package/dist/ledger-core/index.js +2 -1
- package/dist/{ledger-core-C0Yx1I14.js → ledger-core-Dxz0Rkwa.js} +3 -85
- package/dist/ledger-core-Dxz0Rkwa.js.map +1 -0
- package/dist/{llm-client-Cj3c7PEm.js → llm-client-bkztEfIx.js} +2 -2
- package/dist/{llm-client-Cj3c7PEm.js.map → llm-client-bkztEfIx.js.map} +1 -1
- package/dist/meta-eval/index.d.ts +2 -2
- package/dist/multishot/index.d.ts +2 -2
- package/dist/openapi.json +1 -1
- package/dist/{release-report-CoyvyLBs.d.ts → release-report-fZarvIm-.d.ts} +3 -3
- package/dist/{release-report-CoyvyLBs.d.ts.map → release-report-fZarvIm-.d.ts.map} +1 -1
- package/dist/{replay-DbIYwso6.d.ts → replay-DjG4IG60.d.ts} +34 -143
- package/dist/replay-DjG4IG60.d.ts.map +1 -0
- package/dist/{replay-Cb-4Vf0k.js → replay-SA4OB7O7.js} +48 -137
- package/dist/replay-SA4OB7O7.js.map +1 -0
- package/dist/reporting.d.ts +4 -4
- package/dist/{researcher-BCeOEjtR.d.ts → researcher-BxhtGfKa.d.ts} +5 -5
- package/dist/{researcher-BCeOEjtR.d.ts.map → researcher-BxhtGfKa.d.ts.map} +1 -1
- package/dist/{reward-hacking-sE2l_NV6.d.ts → reward-hacking-CqSLiV51.d.ts} +2 -2
- package/dist/{reward-hacking-sE2l_NV6.d.ts.map → reward-hacking-CqSLiV51.d.ts.map} +1 -1
- package/dist/rl.d.ts +5 -5
- package/dist/rl.js +1 -1
- package/dist/rollout/index.d.ts +1 -1
- package/dist/rollout/index.js +2 -2
- package/dist/{rollout-DQFl0UXA.js → rollout-8nj3mYvx.js} +2 -2
- package/dist/{rollout-DQFl0UXA.js.map → rollout-8nj3mYvx.js.map} +1 -1
- package/dist/{rubric-predictive-validity-w2klGv1u.d.ts → rubric-predictive-validity-DQBQj6uV.d.ts} +2 -2
- package/dist/{rubric-predictive-validity-w2klGv1u.d.ts.map → rubric-predictive-validity-DQBQj6uV.d.ts.map} +1 -1
- package/dist/{run-evidence-CbE0A8Xg.d.ts → run-evidence-C4RcRQT5.d.ts} +3 -3
- package/dist/{run-evidence-CbE0A8Xg.d.ts.map → run-evidence-C4RcRQT5.d.ts.map} +1 -1
- package/dist/{run-record-DwHMk1Ai.d.ts → run-record-CztDMXVF.d.ts} +2 -2
- package/dist/{run-record-DwHMk1Ai.d.ts.map → run-record-CztDMXVF.d.ts.map} +1 -1
- package/dist/{semantic-concept-judge-DYXDPZW0.js → semantic-concept-judge-BuIJ9IfB.js} +43 -6
- package/dist/semantic-concept-judge-BuIJ9IfB.js.map +1 -0
- package/dist/{server-DLEvyW2z.js → server-DaCpLfi0.js} +3 -3
- package/dist/{server-DLEvyW2z.js.map → server-DaCpLfi0.js.map} +1 -1
- package/dist/single-run-lock-BTTtPZ9N.js +989 -0
- package/dist/single-run-lock-BTTtPZ9N.js.map +1 -0
- package/dist/{skill-usage-Bv3G4VkA.d.ts → skill-usage-B-BFS8M2.d.ts} +54 -39
- package/dist/skill-usage-B-BFS8M2.d.ts.map +1 -0
- package/dist/{skillopt-optimization-method-CjKMZy0d.js → skillopt-optimization-method-BbGnCC53.js} +18 -802
- package/dist/skillopt-optimization-method-BbGnCC53.js.map +1 -0
- package/dist/{skillopt-optimization-method-CzfnA8O-.d.ts → skillopt-optimization-method-_s0Tub7Y.d.ts} +11 -39
- package/dist/skillopt-optimization-method-_s0Tub7Y.d.ts.map +1 -0
- package/dist/{statistics-mf70aXKp.d.ts → statistics-B5d0Zd-z.d.ts} +2 -2
- package/dist/{statistics-mf70aXKp.d.ts.map → statistics-B5d0Zd-z.d.ts.map} +1 -1
- package/dist/store-otlp-DX4fGIcf.js +757 -0
- package/dist/store-otlp-DX4fGIcf.js.map +1 -0
- package/dist/{summary-report-BKinV4yD.d.ts → summary-report-Cg7BifAM.d.ts} +3 -3
- package/dist/{summary-report-BKinV4yD.d.ts.map → summary-report-Cg7BifAM.d.ts.map} +1 -1
- package/dist/tool-groups-CdYq22lX.d.ts +258 -0
- package/dist/tool-groups-CdYq22lX.d.ts.map +1 -0
- package/dist/traces.d.ts +7 -6
- package/dist/traces.js +5 -5
- package/dist/{types-zFYez3PK.d.ts → types-BBFNHxSK.d.ts} +5 -5
- package/dist/{types-zFYez3PK.d.ts.map → types-BBFNHxSK.d.ts.map} +1 -1
- package/dist/{types-BtJhn8v6.d.ts → types-DoEYskCd.d.ts} +5 -5
- package/dist/{types-BtJhn8v6.d.ts.map → types-DoEYskCd.d.ts.map} +1 -1
- package/dist/{types-5q2T25iW.d.ts → types-uPrS6mD-.d.ts} +2 -2
- package/dist/{types-5q2T25iW.d.ts.map → types-uPrS6mD-.d.ts.map} +1 -1
- package/dist/usage-receipt-CgxMEBZq.js +134 -0
- package/dist/usage-receipt-CgxMEBZq.js.map +1 -0
- package/dist/wire/index.d.ts +3 -3
- package/dist/wire/index.js +1 -1
- package/docs/trace-analysis.md +170 -484
- package/package.json +1 -2
- package/dist/analyze-runs-CPYxfPWT.d.ts +0 -72
- package/dist/analyze-runs-CPYxfPWT.d.ts.map +0 -1
- package/dist/benchmark-command-CMqVqReF.js.map +0 -1
- package/dist/campaign-BIBS-NHV.js.map +0 -1
- package/dist/completion-verifier-B4-IMYcS.d.ts.map +0 -1
- package/dist/cost-ledger-B1D3COAc.d.ts.map +0 -1
- package/dist/cost-ledger-CHDLA0Ss.js.map +0 -1
- package/dist/default-registry-PUhIVRWz.d.ts +0 -215
- package/dist/default-registry-PUhIVRWz.d.ts.map +0 -1
- package/dist/default-registry-lp5R0lve.js.map +0 -1
- package/dist/ledger-core-C0Yx1I14.js.map +0 -1
- package/dist/registry-C4yJTza7.d.ts +0 -178
- package/dist/registry-C4yJTza7.d.ts.map +0 -1
- package/dist/replay-Cb-4Vf0k.js.map +0 -1
- package/dist/replay-DbIYwso6.d.ts.map +0 -1
- package/dist/semantic-concept-judge-DYXDPZW0.js.map +0 -1
- package/dist/single-run-lock-D_bS5xhj.js +0 -318
- package/dist/single-run-lock-D_bS5xhj.js.map +0 -1
- package/dist/skill-usage-Bv3G4VkA.d.ts.map +0 -1
- package/dist/skillopt-optimization-method-CjKMZy0d.js.map +0 -1
- package/dist/skillopt-optimization-method-CzfnA8O-.d.ts.map +0 -1
- package/dist/store-otlp-BenKynPE.js +0 -1688
- package/dist/store-otlp-BenKynPE.js.map +0 -1
- package/dist/tools-DZGdROtG.js +0 -255
- package/dist/tools-DZGdROtG.js.map +0 -1
package/CHANGELOG.md
CHANGED
|
@@ -4,6 +4,16 @@ All notable changes to `@tangle-network/agent-eval` and its sibling `agent-eval-
|
|
|
4
4
|
|
|
5
5
|
---
|
|
6
6
|
|
|
7
|
+
## [0.139.0] - 2026-07-31 - recursive RLM trace analysts and caller failure reasons
|
|
8
|
+
|
|
9
|
+
### Added
|
|
10
|
+
|
|
11
|
+
- Trace analysts run official DSPy RLMs; the Ax stack is retired (#495).
|
|
12
|
+
The published CodeTraceBench evidence remains bound to the retired direct runner via the evidence digests; a fresh certified run must replace it before any accuracy number is attributed to the new engine.
|
|
13
|
+
- `CostLedger.reconcile` accepts a caller-supplied failure reason: `reconcile(callId, observed, { error })` settles a failed receipt carrying that reason, and supplying a reason implies failure.
|
|
14
|
+
0.138.0 had narrowed `CostReceipt.error` to the ledger's own `'paid-call-failed'`, silently discarding caller reasons — a crash orphan settled as a successful $0 call.
|
|
15
|
+
The receipt schema accepts any non-empty reason again, so ledgers persisted before 0.138.0 parse.
|
|
16
|
+
|
|
7
17
|
## [0.138.0] - 2026-07-30 - exact analyst runs with sealed receipts
|
|
8
18
|
|
|
9
19
|
### Fixed
|
|
@@ -38,6 +48,26 @@ All notable changes to `@tangle-network/agent-eval` and its sibling `agent-eval-
|
|
|
38
48
|
Completed results retain a digest of every behavior-defining source file and are read through one strict recursive schema.
|
|
39
49
|
The repository includes one pinned 32-case CodeTraceBench input, two complete 64-call GLM-5.2 Agent Eval runs, and a failure-inclusive run of pinned CodeTracer on the same trajectories.
|
|
40
50
|
Exact source, input, result, resume, usage, cost, and secret-scan checks are committed with the results.
|
|
51
|
+
- The Python package gains a `dspy` extra: `agent-eval-rpc[dspy]` runs official `dspy.RLM` in a sandboxed Deno/Pyodide child process with seven allowlisted trace tools and strict JSON I/O.
|
|
52
|
+
|
|
53
|
+
### Changed
|
|
54
|
+
|
|
55
|
+
- **Breaking:** trace analysts are recursive research programs run through an explicit analysis engine.
|
|
56
|
+
`analyzeTraces` requires an `engine` (the DSPy RLM engine is the primary implementation) and reports engine iterations; `maxTurns`, `maxSubqueries`, `onTurn`, and `AnalyzeTracesTurnSnapshot` are gone.
|
|
57
|
+
`callLlmJson` remains only as the one-shot `direct` benchmark baseline.
|
|
58
|
+
- **Breaking:** `defineTraceAnalyst()` returns an inert `TraceAnalystDefinition` for registration instead of a registrable analyst, and no longer takes a `cost` declaration — analyst cost is always metered LLM usage.
|
|
59
|
+
`createTraceAnalystKind` is now `createTraceAnalyst`; `TraceAnalystKindSpec` and `CreateTraceAnalystKindOpts` are replaced by `TraceAnalystDefinition`.
|
|
60
|
+
- **Breaking:** `BuildTraceAnalystSurfaceDispatchOptions.analyze` receives `instructions` instead of `actorDescription`.
|
|
61
|
+
- **Breaking:** `SteeringOptimizerBackend` narrows to `'pairwise'`.
|
|
62
|
+
- Citation verification is store-backed: cited trace and span ids must resolve in the trace analysis store, and encoded or foreign ids are rejected.
|
|
63
|
+
- External-optimizer subprocess calls fail on HTTP 200 responses with zero input and output usage, and on output over-reservation after recording the actual charge.
|
|
64
|
+
- Relative external-optimizer runner commands resolve against the caller's working directory instead of the child's temporary directory.
|
|
65
|
+
|
|
66
|
+
### Removed
|
|
67
|
+
|
|
68
|
+
- **Breaking:** the Ax analyst stack and the `@ax-llm/ax` dependency: `createAnalystAi`, `CreateAnalystAiConfig`, `structureFindings`, `StructureFindingsOptions`, `StructureFindingsResult`, `AxGepaSteeringOptimizer`, and `AxSteeringOptimizerConfig`.
|
|
69
|
+
- **Breaking:** `createPublicBenchmarkModelRunner`; the analyst benchmark CLI defaults to the DSPy RLM runner and keeps the one-shot runner as the explicit `direct` baseline.
|
|
70
|
+
- `buildTraceAnalystTools`; trace tools are built from the transport-neutral descriptors.
|
|
41
71
|
|
|
42
72
|
### Fixed
|
|
43
73
|
|
package/README.md
CHANGED
|
@@ -357,7 +357,8 @@ pnpm tsx examples/selfimprove-quickstart/index.ts
|
|
|
357
357
|
You do not need a runnable agent to analyze data you already captured.
|
|
358
358
|
Use `analyzeRuns()` for `RunRecord[]`.
|
|
359
359
|
For traces, run a registry of built-in or custom analysts, measure it on labeled issues and exact span locations, then turn only reviewed findings into eval data.
|
|
360
|
-
For a public quality check, convert CodeTraceBench with `traces import-codetracebench`, then run `agent-eval analyst-benchmark` against
|
|
360
|
+
For a public quality check, convert CodeTraceBench with `traces import-codetracebench`, then run `agent-eval analyst-benchmark` against pinned labels.
|
|
361
|
+
The command compares an empty baseline with the official DSPy `RLM` trace analyst and records its trace reads, model calls, tokens, cost, runtime, and cited findings.
|
|
361
362
|
|
|
362
363
|
Use `AnalystRegistry.runExact()` when the caller, rather than registry defaults, must own every execution choice.
|
|
363
364
|
The ordered `analystIds` array is the execution order, and `null` explicitly disables optional budget, timeout, cancellation, cost, tag, or prior-finding channels.
|
package/dist/analyst/index.d.ts
CHANGED
|
@@ -1,13 +1,14 @@
|
|
|
1
1
|
import { a as MultiLayerVerifier, l as VerifyOptions, o as Severity } from "../multi-layer-verifier-BHY1gWAc.js";
|
|
2
|
-
import { A as parseFindingSubject,
|
|
3
|
-
import { c as CostLedgerHandle } from "../cost-ledger-
|
|
4
|
-
import { A as ChatClient, B as SandboxSdkTransportOpts, F as CreateChatClientOpts, I as CustomTransportOpts, L as DirectProviderTransportOpts, M as ChatResponse, N as ChatTransport, P as CliBridgeTransportOpts, R as MockTransportOpts, V as createChatClient, j as ChatRequest, k as ChatCallOpts, m as JudgeInput, p as JudgeFn,
|
|
5
|
-
import { _ as makeFinding, a as AnalystInputKind, c as AnalystRunInputs, d as AnalystSeverity, f as AnalystUsageReceipt, g as computeFindingId, h as ProposalFindingOrigin, i as AnalystFinding, l as AnalystRunResult, m as ProposalFinding, n as AnalystContext, o as AnalystRequirements, p as EvidenceRef, r as AnalystCost, s as AnalystRunEvent, t as Analyst, u as AnalystRunSummary, v as makeProposalFinding
|
|
6
|
-
import {
|
|
7
|
-
import { a as ExactAnalystRunPolicySnapshot, c as ExactAnalystSnapshot, d as ExactExecutionComponentSnapshot, i as ExactAnalystRunEvent, l as ExactCapableAnalyst, n as ExactAnalystExecutionPlanSnapshot, o as ExactAnalystRunResult, r as ExactAnalystRunCompletion, s as ExactAnalystRunSummary, t as ExactAnalystBudgetSnapshot, u as ExactExecutionComponentIdentity } from "../exact-types-
|
|
8
|
-
import {
|
|
9
|
-
import { C as scoreAnalystFindings, S as traceStoreEvidenceResolver, _ as AnalystIssueExpectation, a as AnalystBenchmarkLabelState, b as registryBenchmarkRunner, c as AnalystBenchmarkProvenance, d as AnalystBenchmarkSummary, f as AnalystEvidenceExpectation, g as AnalystFindingScore, h as AnalystEvidenceResolver, i as AnalystBenchmarkError, l as AnalystBenchmarkResult, m as AnalystEvidenceResolutionError, n as AnalystBenchmarkDatasetRef, o as AnalystBenchmarkObservation, p as AnalystEvidenceResolution, r as AnalystBenchmarkDescriptor, s as AnalystBenchmarkOutput, t as AnalystBenchmarkCase, u as AnalystBenchmarkRunner, v as AnalystLatencyDistribution, x as runAnalystBenchmark, y as RunAnalystBenchmarkOptions } from "../benchmark-
|
|
10
|
-
import {
|
|
2
|
+
import { A as parseFindingSubject, C as FINDING_SUBJECT_KINDS, D as FindingSubjectStringSchema, E as FindingSubjectKind, F as DefineExactCustomAnalystOptions, G as SemanticConceptJudgeOptions, I as defineCustomAnalyst, L as defineTraceAnalyst, M as DspyRlmTraceEngineOptions, N as createDspyRlmTraceEngine, O as KIND_EXPECTED_SUBJECTS, P as DefineCustomAnalystOptions, S as FINDING_SUBJECT_GRAMMAR_PROMPT, T as FindingSubject, W as SemanticConceptJudgeInput, Y as RunCritic, Z as RunTrace, _ as FindingsDiff, a as SkillUsageScanConfig, b as defaultIsMaterial, c as DEFAULT_TRACE_ANALYST_KINDS, d as IMPROVEMENT_KIND_SPEC, f as FAILURE_MODE_KIND_SPEC, g as DiffPolicy, h as emitControlIntegrityFindings, i as SkillUsageReport, j as renderFindingSubject, k as findingSubjectGrammarPromptFor, l as KNOWLEDGE_POISONING_KIND_SPEC, m as ControlIntegrityAnalyst, n as SkillUsageAnalyst, o as buildSkillUsageReport, p as CONTROL_INTEGRITY_ANALYST, r as SkillUsageRecord, s as emitSkillUsageFindings, t as SKILL_USAGE_ANALYST, u as KNOWLEDGE_GAP_KIND_SPEC, v as FindingsStore, w as FINDING_SUBJECT_SYNTAX, x as diffFindings, y as PersistedFinding } from "../skill-usage-B-BFS8M2.js";
|
|
3
|
+
import { b as CustomTokenPricing, c as CostLedgerHandle } from "../cost-ledger-DKgyIWRj.js";
|
|
4
|
+
import { A as ChatClient, B as SandboxSdkTransportOpts, F as CreateChatClientOpts, I as CustomTransportOpts, L as DirectProviderTransportOpts, M as ChatResponse, N as ChatTransport, P as CliBridgeTransportOpts, R as MockTransportOpts, V as createChatClient, j as ChatRequest, k as ChatCallOpts, m as JudgeInput, p as JudgeFn, z as RouterTransportOpts } from "../types-uPrS6mD-.js";
|
|
5
|
+
import { _ as makeFinding, a as AnalystInputKind, c as AnalystRunInputs, d as AnalystSeverity, f as AnalystUsageReceipt, g as computeFindingId, h as ProposalFindingOrigin, i as AnalystFinding, l as AnalystRunResult, m as ProposalFinding, n as AnalystContext, o as AnalystRequirements, p as EvidenceRef, r as AnalystCost, s as AnalystRunEvent, t as Analyst, u as AnalystRunSummary, v as makeProposalFinding } from "../types-DoEYskCd.js";
|
|
6
|
+
import { a as createTraceAnalyst, c as runTraceAnalyst, d as deriveEfficiencyFindings, i as TraceAnalystDefinition, l as BehavioralAnalystOptions, n as buildDefaultAnalystRegistry, o as renderPriorFindings, r as CreateTraceAnalystOptions, s as renderUpstreamFindings, t as DefaultAnalystRegistryOptions, u as behavioralAnalyst } from "../default-registry-B8vf7Rmf.js";
|
|
7
|
+
import { a as ExactAnalystRunPolicySnapshot, c as ExactAnalystSnapshot, d as ExactExecutionComponentSnapshot, i as ExactAnalystRunEvent, l as ExactCapableAnalyst, n as ExactAnalystExecutionPlanSnapshot, o as ExactAnalystRunResult, r as ExactAnalystRunCompletion, s as ExactAnalystRunSummary, t as ExactAnalystBudgetSnapshot, u as ExactExecutionComponentIdentity } from "../exact-types-MaaFcllV.js";
|
|
8
|
+
import { A as ExactAnalystRunExecutionError, D as AnalystRegistryOptions, E as AnalystRegistry, M as RegistryRunOpts, N as assertExactRegistryRunOpts, O as BudgetPolicy, T as AnalystHooks, j as ExactRegistryRunOpts, k as ExactAnalystBudgetPolicy } from "../completion-verifier-IPoP4fQO.js";
|
|
9
|
+
import { C as scoreAnalystFindings, S as traceStoreEvidenceResolver, _ as AnalystIssueExpectation, a as AnalystBenchmarkLabelState, b as registryBenchmarkRunner, c as AnalystBenchmarkProvenance, d as AnalystBenchmarkSummary, f as AnalystEvidenceExpectation, g as AnalystFindingScore, h as AnalystEvidenceResolver, i as AnalystBenchmarkError, l as AnalystBenchmarkResult, m as AnalystEvidenceResolutionError, n as AnalystBenchmarkDatasetRef, o as AnalystBenchmarkObservation, p as AnalystEvidenceResolution, r as AnalystBenchmarkDescriptor, s as AnalystBenchmarkOutput, t as AnalystBenchmarkCase, u as AnalystBenchmarkRunner, v as AnalystLatencyDistribution, x as runAnalystBenchmark, y as RunAnalystBenchmarkOptions } from "../benchmark-DDVdWcwA.js";
|
|
10
|
+
import { r as ExternalOptimizerRunnerCommand } from "../external-optimizer-contracts-BrxY2Sli.js";
|
|
11
|
+
import { a as TraceAnalysisEngineRequest, c as resolveTraceAnalystLimits, d as RawAnalystEvidence, f as RawAnalystEvidenceSchema, g as parseRawFinding, h as evidenceRefsFromRawFinding, i as TraceAnalysisEngine, l as ANALYST_SEVERITIES, m as RawAnalystFindingSchema, n as buildTraceToolsForGroup, o as TraceAnalysisEngineResult, p as RawAnalystFinding, r as DEFAULT_TRACE_ANALYST_LIMITS, s as TraceAnalystLimits, t as TraceToolGroupName, u as RAW_FINDING_SCHEMA_PROMPT } from "../tool-groups-CdYq22lX.js";
|
|
11
12
|
//#region src/analyst/adapters.d.ts
|
|
12
13
|
declare function liftSeverity(s: Severity): AnalystSeverity;
|
|
13
14
|
interface VerifierAdapterOpts<Env> {
|
|
@@ -274,6 +275,17 @@ interface PublicAnalystBenchmarkModelConfig {
|
|
|
274
275
|
model: string;
|
|
275
276
|
maxOutputTokens: number;
|
|
276
277
|
timeoutMs: number;
|
|
278
|
+
/** Required when the model is absent from agent-eval's pricing table. */
|
|
279
|
+
pricing?: CustomTokenPricing;
|
|
280
|
+
/** Independent per-case recursive-engine spend limit. Default: 1 USD. */
|
|
281
|
+
maxCostUsdPerAnalysis?: number;
|
|
282
|
+
dspyRlm?: {
|
|
283
|
+
runner?: ExternalOptimizerRunnerCommand;
|
|
284
|
+
maxIterations?: number;
|
|
285
|
+
maxLlmCalls?: number;
|
|
286
|
+
maxToolCalls?: number;
|
|
287
|
+
maxOutputChars?: number;
|
|
288
|
+
};
|
|
277
289
|
costLedger?: CostLedgerHandle;
|
|
278
290
|
durability?: {
|
|
279
291
|
runIdentitySha256: string;
|
|
@@ -341,10 +353,15 @@ declare function preparePublicAnalystBenchmark(options: {
|
|
|
341
353
|
}): Promise<PreparedPublicAnalystBenchmark>;
|
|
342
354
|
//#endregion
|
|
343
355
|
//#region src/analyst/benchmark-public-model.d.ts
|
|
344
|
-
|
|
356
|
+
/** One-shot JSON baseline. This is not a recursive trace analyst. */
|
|
357
|
+
declare function createPublicBenchmarkDirectRunner(dataset: PublicAnalystBenchmarkDataset, config: PublicAnalystBenchmarkModelConfig): AnalystBenchmarkRunner<AnalystRunInputs>;
|
|
345
358
|
declare function publicBenchmarkProtocolSha256(dataset: PublicAnalystBenchmarkDataset): string;
|
|
346
359
|
declare const CODE_TRACE_BENCH_ANALYST_PROMPT = "Analyze exactly one coding-agent trajectory and its attached final verification.\nYour task is the CodeTraceBench incorrect-step task: identify every wrong state-changing action, bad hypothesis that drives an action, and regression.\nAn incorrect step remains incorrect when the agent later recovers or the final verification passes.\nInspect the complete supplied trace data.\nUse the final-verification outcome as evidence about the final state, not as a rule for whether earlier steps were incorrect.\nFor each candidate, inspect the assistant action and its following observation.\nLabel a failed command when the assistant caused it through a wrong action or unsupported hypothesis.\nLabel the later corrective action only when that action is itself wrong.\nDo not label a diagnostic probe merely because it exposes an earlier defect.\nDo not label a redundant but correct read or search; CodeTraceBench scores unuseful steps separately, and this run scores incorrect steps only.\nDo not label a step solely because final verification failed.\nWhen final verification is unavailable, use only directly observed trajectory evidence.\nEmit one finding per incorrect assistant step.\nEach finding's step MUST be the positive integer n from an existing assistant LLM span named step-<n>.\nNever select an EVALUATOR, TOOL, CHAIN, final-verification, benchmark-verification, or message-<n> span.\nBefore emitting a finding, inspect its candidate span's attributes.content and describe only the action shown there.\nWhen the trajectory has no incorrect steps, return an empty findings array.";
|
|
347
360
|
//#endregion
|
|
361
|
+
//#region src/analyst/benchmark-public-rlm.d.ts
|
|
362
|
+
/** Public benchmark candidate that runs the actual recursive trace analyst. */
|
|
363
|
+
declare function createPublicBenchmarkRlmRunner(dataset: PublicAnalystBenchmarkDataset, config: PublicAnalystBenchmarkModelConfig): AnalystBenchmarkRunner<AnalystRunInputs>;
|
|
364
|
+
//#endregion
|
|
348
365
|
//#region src/analyst/benchmark-comparison.d.ts
|
|
349
366
|
type AnalystComparisonMetric = 'completion' | 'issueRecall' | 'findingPrecision' | 'f1' | 'criticalStepAccuracy' | 'citationCoverage' | 'citationExcerptCoverage' | 'citationLabelAgreement' | 'citationResolution' | 'trustedNegativeAccuracy' | 'latencyMs' | 'calls' | 'inputTokens' | 'outputTokens' | 'reasoningTokens' | 'cachedTokens' | 'cacheWriteTokens' | 'costUsd';
|
|
350
367
|
interface AnalystMetricComparison {
|
|
@@ -488,7 +505,7 @@ interface AnalystBenchmarkRunIdentity {
|
|
|
488
505
|
analystProtocolSha256: string;
|
|
489
506
|
implementationSha256: string;
|
|
490
507
|
dependencyLockSha256: string;
|
|
491
|
-
runnerIds: readonly ['empty', '
|
|
508
|
+
runnerIds: readonly ['empty', 'dspy-rlm'];
|
|
492
509
|
};
|
|
493
510
|
inputs: {
|
|
494
511
|
labelsSha256: string;
|
|
@@ -554,7 +571,7 @@ declare function readAnalystBenchmarkArtifact(path: string): Promise<AnalystBenc
|
|
|
554
571
|
//#endregion
|
|
555
572
|
//#region src/analyst/benchmark-command.d.ts
|
|
556
573
|
interface AnalystBenchmarkCommandDependencies {
|
|
557
|
-
|
|
574
|
+
createAnalystRunner?: (dataset: PublicAnalystBenchmarkDataset, config: PublicAnalystBenchmarkModelConfig) => AnalystBenchmarkRunner<AnalystRunInputs>;
|
|
558
575
|
}
|
|
559
576
|
interface AnalystBenchmarkCommandConfig {
|
|
560
577
|
dataset: PublicAnalystBenchmarkDataset;
|
|
@@ -576,46 +593,31 @@ interface AnalystBenchmarkCommandConfig {
|
|
|
576
593
|
resume: boolean;
|
|
577
594
|
}
|
|
578
595
|
declare function runAnalystBenchmarkCommand(argv: readonly string[], env?: NodeJS.ProcessEnv, dependencies?: AnalystBenchmarkCommandDependencies): Promise<number>;
|
|
579
|
-
declare const ANALYST_BENCHMARK_HELP = "agent-eval analyst-benchmark\n\nRun
|
|
596
|
+
declare const ANALYST_BENCHMARK_HELP = "agent-eval analyst-benchmark\n\nRun the recursive DSPy trace analyst against public AgentRx or CodeTraceBench labels.\n\nRequired:\n --dataset agentrx|codetracebench\n --labels <dataset.json|dataset.jsonl>\n --trace-dir <one-trace-per-file OTLP JSONL directory>\n --artifact-dir <extracted artifact root> Required for CodeTraceBench\n --out <new output directory>\n --revision <full 40- or 64-character hex digest>\n --split <dataset split>\n --base-url <OpenAI-compatible /v1 URL>\n --api-key-env <environment variable containing the bearer>\n --model <provider model id>\n --limit <positive case count>\n\nControls:\n --resume Continue an interrupted run in --out\n --seed <integer> Case-selection and comparison seed. Default: 0\n --concurrency <positive integer> Parallel benchmark jobs. Default: 1\n --repetitions <positive integer> Runs per case and runner. Default: 1\n --max-output-tokens <positive> Model output limit per call. Default: 4096\n --python <executable> Python with agent-eval-rpc[dspy]. Default: python\n --timeout-ms <positive> Model analyst deadline per case. Default: 300000\n --max-cost-usd <positive> Run-wide spend limit. Default: 5\n --max-artifact-bytes <positive> Final evidence bytes per case. Default: 8388608\n\nWrites result.json with every observation, metric, usage field, error, comparison,\ninput digest, artifact digest, case distribution, selected case id, and explicit\nunknown cost. Limited deterministic-hash subsets are marked non-representative.\nCompleted observations are fsynced to observations.jsonl. Shareable output is in\nresult.json and report.md. Machine-local paths, endpoint, and command are isolated\nin run.local.json.\nThe key is read from the named environment variable and is never written.";
|
|
580
597
|
//#endregion
|
|
581
598
|
//#region src/analyst/benchmark-implementation.d.ts
|
|
582
599
|
declare const ANALYST_BENCHMARK_IMPLEMENTATION_DIGEST_ALGORITHM = "sha256-canonical-source-manifest";
|
|
583
600
|
declare const ANALYST_BENCHMARK_DEPENDENCY_LOCK_DIGEST_ALGORITHM = "sha256-canonical-file-manifest";
|
|
584
601
|
declare const ANALYST_BENCHMARK_DEPENDENCY_LOCK_FILES: readonly string[];
|
|
585
|
-
declare const ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256 = "
|
|
586
|
-
/** The published benchmark evidence was produced at this package version
|
|
587
|
-
*
|
|
588
|
-
*
|
|
589
|
-
*
|
|
590
|
-
*
|
|
591
|
-
*
|
|
602
|
+
declare const ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256 = "6f3dc59755ff2a26fba4c2c1ad670f436cdd183faf252a5b9c2d50e6b190b460";
|
|
603
|
+
/** The published benchmark evidence was produced at this package version, by
|
|
604
|
+
* the retired one-shot direct runner, before trace analysts moved to the
|
|
605
|
+
* recursive DSPy RLM engine. Both evidence digests below are historical facts
|
|
606
|
+
* about that artifact: the current implementation and dependency manifest have
|
|
607
|
+
* since changed, so they cannot describe the current engine. A fresh certified
|
|
608
|
+
* run must replace the published evidence before any accuracy number is
|
|
609
|
+
* attributed to the engine that ships today. */
|
|
592
610
|
declare const ANALYST_BENCHMARK_EVIDENCE_PACKAGE_VERSION = "0.137.0";
|
|
593
611
|
declare const ANALYST_BENCHMARK_EVIDENCE_DEPENDENCY_LOCK_SHA256 = "1e03f2daed356d60316aabefb407ec1e437ac94d408d61eea4ae096e9c6fbb5b";
|
|
612
|
+
declare const ANALYST_BENCHMARK_EVIDENCE_IMPLEMENTATION_SHA256 = "4dba263b6256a30d56c7fdb2d992d3a953c0035d731f359b704db806f68f75ac";
|
|
594
613
|
declare const ANALYST_BENCHMARK_IMPLEMENTATION_FILES: readonly string[];
|
|
595
|
-
declare const ANALYST_BENCHMARK_IMPLEMENTATION_SHA256 = "
|
|
614
|
+
declare const ANALYST_BENCHMARK_IMPLEMENTATION_SHA256 = "5e47fc9d9f49c468d3ef06c5d552925e6a026f9e22a75056a9c8c5a2879744c0";
|
|
596
615
|
declare function analystBenchmarkImplementationDigest(): string;
|
|
597
616
|
declare function analystBenchmarkDependencyLockDigest(): string;
|
|
598
617
|
//#endregion
|
|
599
618
|
//#region src/analyst/benchmark-report.d.ts
|
|
600
619
|
declare function renderAnalystBenchmarkMarkdown(result: AnalystBenchmarkResult, comparisons?: readonly AnalystRunnerComparison[]): string;
|
|
601
620
|
//#endregion
|
|
602
|
-
//#region src/analyst/define.d.ts
|
|
603
|
-
interface DefineTraceAnalystOptions {
|
|
604
|
-
id: string;
|
|
605
|
-
description: string;
|
|
606
|
-
version?: string;
|
|
607
|
-
cost: AnalystCost;
|
|
608
|
-
analyze: Analyst<TraceAnalysisStore>['analyze'];
|
|
609
|
-
}
|
|
610
|
-
interface DefineExactTraceAnalystOptions extends DefineTraceAnalystOptions {
|
|
611
|
-
/** Canonical JSON for behavior knobs not already bound by `version`. */
|
|
612
|
-
executionConfig: Readonly<Record<string, unknown>>;
|
|
613
|
-
}
|
|
614
|
-
/** Define a custom trace analyst without repeating fixed registry fields. */
|
|
615
|
-
declare function defineTraceAnalyst(options: DefineExactTraceAnalystOptions): ExactCapableAnalyst<TraceAnalysisStore>;
|
|
616
|
-
declare function defineTraceAnalyst(options: DefineTraceAnalystOptions): Analyst<TraceAnalysisStore>;
|
|
617
|
-
type TraceAnalystAnalyze = (store: TraceAnalysisStore, context: Parameters<Analyst<TraceAnalysisStore>['analyze']>[1]) => Promise<AnalystFinding[]>;
|
|
618
|
-
//#endregion
|
|
619
621
|
//#region src/analyst/parse-tolerant.d.ts
|
|
620
622
|
/**
|
|
621
623
|
* Forgiving pre-parse for analyst findings. Weak models routinely emit
|
|
@@ -653,60 +655,5 @@ declare function isProposalFinding(finding: unknown): finding is ProposalFinding
|
|
|
653
655
|
*/
|
|
654
656
|
declare function assertProposalFindings(findings: unknown, context?: string): ReadonlyArray<ProposalFinding>;
|
|
655
657
|
//#endregion
|
|
656
|
-
|
|
657
|
-
interface StructureFindingsOptions {
|
|
658
|
-
/** The actor's free-form diagnosis prose. */
|
|
659
|
-
report: string;
|
|
660
|
-
analystId: string;
|
|
661
|
-
/** Coarse classification stamped on every extracted finding. */
|
|
662
|
-
area: string;
|
|
663
|
-
model: string;
|
|
664
|
-
baseUrl: string;
|
|
665
|
-
apiKey?: string;
|
|
666
|
-
/** Optional ledger for direct use. */
|
|
667
|
-
costLedger?: CostLedgerHandle;
|
|
668
|
-
costPhase?: string;
|
|
669
|
-
costTags?: Record<string, string>;
|
|
670
|
-
maxTokens?: number;
|
|
671
|
-
signal?: AbortSignal;
|
|
672
|
-
/** Max reask attempts after a zero/invalid extraction. Default 1. */
|
|
673
|
-
maxReasks?: number;
|
|
674
|
-
/** Apply the caller's normal finding rules before a recovered row is lifted. */
|
|
675
|
-
processRow?: (row: RawAnalystFinding) => RawAnalystFinding | null;
|
|
676
|
-
/** Provenance copied onto every recovered finding. */
|
|
677
|
-
findingMetadata?: Record<string, unknown>;
|
|
678
|
-
/** Test seam: inject a fetch (no network in unit tests). */
|
|
679
|
-
fetchImpl?: LlmClientOptions['fetch'];
|
|
680
|
-
}
|
|
681
|
-
interface StructureFindingsResult {
|
|
682
|
-
findings: AnalystFinding[];
|
|
683
|
-
outcome: 'ok' | 'extraction_failed';
|
|
684
|
-
}
|
|
685
|
-
declare function structureFindings(opts: StructureFindingsOptions): Promise<StructureFindingsResult>;
|
|
686
|
-
//#endregion
|
|
687
|
-
//#region src/analyst/tool-groups.d.ts
|
|
688
|
-
/** Named tool sets. Kinds pass `tools: TRACE_TOOL_GROUPS.failureForensics` etc. */
|
|
689
|
-
type TraceToolGroupName =
|
|
690
|
-
/** All seven tools. Use for open-ended discovery kinds. */
|
|
691
|
-
'all' |
|
|
692
|
-
/** Overview + paginated query + count. No deep reads. Cheap. */
|
|
693
|
-
'discovery' |
|
|
694
|
-
/** Discovery + viewTrace + viewSpans. Deep-read but no regex search. */
|
|
695
|
-
'discoveryAndRead' |
|
|
696
|
-
/** Discovery + search tools. For pattern-matching across many traces. */
|
|
697
|
-
'discoveryAndSearch' |
|
|
698
|
-
/** Discovery + viewSpans + searchSpan. Targeted-span work after another kind narrows down. */
|
|
699
|
-
'targeted' |
|
|
700
|
-
/** One known-small trace: overview, bounded reads, and in-trace search. */
|
|
701
|
-
'singleTrace';
|
|
702
|
-
/**
|
|
703
|
-
* Build the tool set for a named group bound to a specific trace store.
|
|
704
|
-
*
|
|
705
|
-
* `all` returns every tool. Other groups filter `buildTraceAnalystTools`
|
|
706
|
-
* by name to the documented subset. An unrecognised group name throws —
|
|
707
|
-
* silently returning all tools would defeat the cost-control point.
|
|
708
|
-
*/
|
|
709
|
-
declare function buildTraceToolsForGroup(group: TraceToolGroupName, store: TraceAnalysisStore): AxFunction[];
|
|
710
|
-
//#endregion
|
|
711
|
-
export { AGENT_RX_UPSTREAM_REVISION, ANALYST_BENCHMARK_COST_LEDGER_FILE, ANALYST_BENCHMARK_DEPENDENCY_LOCK_DIGEST_ALGORITHM, ANALYST_BENCHMARK_DEPENDENCY_LOCK_FILES, ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256, ANALYST_BENCHMARK_EVIDENCE_DEPENDENCY_LOCK_SHA256, ANALYST_BENCHMARK_EVIDENCE_PACKAGE_VERSION, ANALYST_BENCHMARK_HELP, ANALYST_BENCHMARK_IMPLEMENTATION_DIGEST_ALGORITHM, ANALYST_BENCHMARK_IMPLEMENTATION_FILES, ANALYST_BENCHMARK_IMPLEMENTATION_SHA256, ANALYST_BENCHMARK_LOCAL_RECEIPT_FILE, ANALYST_BENCHMARK_MANIFEST_FILE, ANALYST_BENCHMARK_OBSERVATIONS_FILE, ANALYST_SEVERITIES, type AgentRxBenchmarkCaseOptions, type AgentRxCalibrationRunnerSummary, type AgentRxCalibrationSummary, type AgentRxFailure, type AgentRxPrediction, type AgentRxPredictionReport, type AgentRxRow, type Analyst, type AnalystBenchmarkArtifact, type AnalystBenchmarkCase, type AnalystBenchmarkCommandConfig, type AnalystBenchmarkCommandDependencies, type AnalystBenchmarkDatasetRef, type AnalystBenchmarkDescriptor, type AnalystBenchmarkError, type AnalystBenchmarkLabelState, type AnalystBenchmarkLocalRunReceipt, type AnalystBenchmarkObservation, type AnalystBenchmarkOutput, type AnalystBenchmarkProgressRow, type AnalystBenchmarkProvenance, type AnalystBenchmarkResult, type AnalystBenchmarkRunIdentity, type AnalystBenchmarkRunManifest, type AnalystBenchmarkRunner, type AnalystBenchmarkSummary, type AnalystComparisonMetric, type AnalystContext, type AnalystCost, type AnalystEvidenceExpectation, type AnalystEvidenceResolution, type AnalystEvidenceResolutionError, type AnalystEvidenceResolver, type AnalystFinding, type AnalystFindingScore, type AnalystHooks, type AnalystInputKind, type AnalystIssueExpectation, type AnalystLatencyDistribution, type AnalystMetricComparison, AnalystRegistry, type AnalystRegistryOptions, type AnalystRequirements, type AnalystRunEvent, type AnalystRunInputs, type AnalystRunResult, type AnalystRunSummary, type AnalystRunnerComparison, type AnalystSeverity, type AnalystUsageReceipt, type BehavioralAnalystOptions, type BudgetPolicy, CODE_TRACE_BENCH_ANALYST_PROMPT, CONTROL_INTEGRITY_ANALYST, type ChatCallOpts, type ChatClient, type ChatRequest, type ChatResponse, type ChatTransport, type CliBridgeTransportOpts, type CodeTraceBenchCaseOptions, type CodeTraceBenchLabelOptions, type CodeTraceBenchLabelSet, type CodeTraceBenchRow, type CodeTraceCalibrationRunnerSummary, type CodeTraceCalibrationSummary, type CodeTraceStageAnnotation, type CodeTracerLabelGroup, type CodeTracerPredictionAdapterOptions, type CodeTracerPredictions, type CodeTracerStepLabel, ControlIntegrityAnalyst, type CreateAnalystAiConfig, type CreateChatClientOpts, type CreateTraceAnalystKindOpts, type CustomTransportOpts, DEFAULT_MAX_VERIFICATION_ARTIFACT_BYTES, DEFAULT_TRACE_ANALYST_KINDS, type DefaultAnalystRegistryOptions, type DefineExactTraceAnalystOptions, type DefineTraceAnalystOptions, type DiffPolicy, type DirectProviderTransportOpts, type EvidenceRef, type ExactAnalystBudgetPolicy, type ExactAnalystBudgetSnapshot, type ExactAnalystExecutionPlanSnapshot, type ExactAnalystRunCompletion, type ExactAnalystRunEvent, ExactAnalystRunExecutionError, type ExactAnalystRunPolicySnapshot, type ExactAnalystRunResult, type ExactAnalystRunSummary, type ExactAnalystSnapshot, type ExactCapableAnalyst, type ExactExecutionComponentIdentity, type ExactExecutionComponentSnapshot, type ExactRegistryRunOpts, FAILURE_MODE_KIND_SPEC, FINDING_SUBJECT_GRAMMAR_PROMPT, FINDING_SUBJECT_KINDS, FINDING_SUBJECT_SYNTAX, type FindingSubject, type FindingSubjectKind, FindingSubjectStringSchema, type FindingsDiff, FindingsStore, IMPROVEMENT_KIND_SPEC, type JudgeAdapterOpts, KIND_EXPECTED_SUBJECTS, KNOWLEDGE_GAP_KIND_SPEC, KNOWLEDGE_POISONING_KIND_SPEC, type LoadedVerificationArtifacts, type MockTransportOpts, type PersistedFinding, type PreparedPublicAnalystBenchmark, type ProposalFinding, type ProposalFindingOrigin, type PublicAnalystBenchmarkDataset, type PublicAnalystBenchmarkModelConfig, type PublicBenchmarkDistributions, type PublicBenchmarkSelectionReport, type PublicBenchmarkValueDistribution, RAW_FINDING_SCHEMA_PROMPT, type RawAnalystEvidence, RawAnalystEvidenceSchema, type RawAnalystFinding, RawAnalystFindingSchema, type RegistryRunOpts, type RouterTransportOpts, type RunAnalystBenchmarkOptions, type RunCriticAdapterOpts, SKILL_USAGE_ANALYST, type SandboxSdkTransportOpts, type SemanticConceptJudgeAdapterOpts, SkillUsageAnalyst, type SkillUsageRecord, type SkillUsageReport, type SkillUsageScanConfig, type StepLabelAdapterOptions, type StructureFindingsOptions, type StructureFindingsResult, type TraceAnalystAnalyze, type TraceAnalystKindSpec, type TraceToolGroupName, type UpstreamPredictionAdapterOptions, type VerificationArtifactFile, type VerificationArtifactManifest, type VerificationArtifactRole, type VerificationAvailabilitySummary, type VerificationOutcome, type VerificationOutcomeSource, type VerificationOutcomeStatus, type VerificationResultFile, type VerifierAdapterOpts, adaptPublicBenchmarkFindings, agentRxBenchmarkCase, agentRxPredictionsToFindings, analystBenchmarkDependencyLockDigest, analystBenchmarkImplementationDigest, appendVerificationArtifactsToOtlp, assertExactRegistryRunOpts, assertProposalFindings, behavioralAnalyst, buildDefaultAnalystRegistry, buildSkillUsageReport, buildTraceToolsForGroup, codeTraceBenchCase, codeTracerPredictionsToFindings, coerceJson, coerceToFindingRows, compareAnalystRunners, computeFindingId, createAnalystAi, createChatClient, createJudgeAdapter, createPublicBenchmarkModelRunner, createRunCriticAdapter, createSemanticConceptJudgeAdapter, createTraceAnalystKind, createVerifierAdapter, defaultIsMaterial, defineTraceAnalyst, deriveEfficiencyFindings, diffFindings, emitControlIntegrityFindings, emitSkillUsageFindings, emptyPublicBenchmarkRunner, evidenceRefsFromRawFinding, findingSubjectGrammarPromptFor, isProposalFinding, liftSeverity, loadCodeTraceVerificationArtifacts, loadPublicBenchmarkRows, makeFinding, makeProposalFinding, normalizeAgentRxCategory, normalizeBenchmarkLabel, parseFindingSubject, parseRawFinding, parseVerificationOutcome, preparePublicAnalystBenchmark, publicBenchmarkDistributions, publicBenchmarkProtocolSha256, publicBenchmarkSelectionReport, readAnalystBenchmarkArtifact, registryBenchmarkRunner, renderAgentRxCalibrationMarkdown, renderAnalystBenchmarkMarkdown, renderCodeTraceCalibrationMarkdown, renderFindingSubject, renderPriorFindings, renderUpstreamFindings, roundAgentRxStep, runAnalystBenchmark, runAnalystBenchmarkCommand, scoreAnalystFindings, selectPublicBenchmarkRows, stripCodeFences, structureFindings, summarizeAgentRxCalibration, summarizeCodeTraceCalibration, traceStoreEvidenceResolver };
|
|
658
|
+
export { AGENT_RX_UPSTREAM_REVISION, ANALYST_BENCHMARK_COST_LEDGER_FILE, ANALYST_BENCHMARK_DEPENDENCY_LOCK_DIGEST_ALGORITHM, ANALYST_BENCHMARK_DEPENDENCY_LOCK_FILES, ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256, ANALYST_BENCHMARK_EVIDENCE_DEPENDENCY_LOCK_SHA256, ANALYST_BENCHMARK_EVIDENCE_IMPLEMENTATION_SHA256, ANALYST_BENCHMARK_EVIDENCE_PACKAGE_VERSION, ANALYST_BENCHMARK_HELP, ANALYST_BENCHMARK_IMPLEMENTATION_DIGEST_ALGORITHM, ANALYST_BENCHMARK_IMPLEMENTATION_FILES, ANALYST_BENCHMARK_IMPLEMENTATION_SHA256, ANALYST_BENCHMARK_LOCAL_RECEIPT_FILE, ANALYST_BENCHMARK_MANIFEST_FILE, ANALYST_BENCHMARK_OBSERVATIONS_FILE, ANALYST_SEVERITIES, type AgentRxBenchmarkCaseOptions, type AgentRxCalibrationRunnerSummary, type AgentRxCalibrationSummary, type AgentRxFailure, type AgentRxPrediction, type AgentRxPredictionReport, type AgentRxRow, type Analyst, type AnalystBenchmarkArtifact, type AnalystBenchmarkCase, type AnalystBenchmarkCommandConfig, type AnalystBenchmarkCommandDependencies, type AnalystBenchmarkDatasetRef, type AnalystBenchmarkDescriptor, type AnalystBenchmarkError, type AnalystBenchmarkLabelState, type AnalystBenchmarkLocalRunReceipt, type AnalystBenchmarkObservation, type AnalystBenchmarkOutput, type AnalystBenchmarkProgressRow, type AnalystBenchmarkProvenance, type AnalystBenchmarkResult, type AnalystBenchmarkRunIdentity, type AnalystBenchmarkRunManifest, type AnalystBenchmarkRunner, type AnalystBenchmarkSummary, type AnalystComparisonMetric, type AnalystContext, type AnalystCost, type AnalystEvidenceExpectation, type AnalystEvidenceResolution, type AnalystEvidenceResolutionError, type AnalystEvidenceResolver, type AnalystFinding, type AnalystFindingScore, type AnalystHooks, type AnalystInputKind, type AnalystIssueExpectation, type AnalystLatencyDistribution, type AnalystMetricComparison, AnalystRegistry, type AnalystRegistryOptions, type AnalystRequirements, type AnalystRunEvent, type AnalystRunInputs, type AnalystRunResult, type AnalystRunSummary, type AnalystRunnerComparison, type AnalystSeverity, type AnalystUsageReceipt, type BehavioralAnalystOptions, type BudgetPolicy, CODE_TRACE_BENCH_ANALYST_PROMPT, CONTROL_INTEGRITY_ANALYST, type ChatCallOpts, type ChatClient, type ChatRequest, type ChatResponse, type ChatTransport, type CliBridgeTransportOpts, type CodeTraceBenchCaseOptions, type CodeTraceBenchLabelOptions, type CodeTraceBenchLabelSet, type CodeTraceBenchRow, type CodeTraceCalibrationRunnerSummary, type CodeTraceCalibrationSummary, type CodeTraceStageAnnotation, type CodeTracerLabelGroup, type CodeTracerPredictionAdapterOptions, type CodeTracerPredictions, type CodeTracerStepLabel, ControlIntegrityAnalyst, type CreateChatClientOpts, type CreateTraceAnalystOptions, type CustomTransportOpts, DEFAULT_MAX_VERIFICATION_ARTIFACT_BYTES, DEFAULT_TRACE_ANALYST_KINDS, DEFAULT_TRACE_ANALYST_LIMITS, type DefaultAnalystRegistryOptions, type DefineCustomAnalystOptions, type DefineExactCustomAnalystOptions, type DiffPolicy, type DirectProviderTransportOpts, type DspyRlmTraceEngineOptions, type EvidenceRef, type ExactAnalystBudgetPolicy, type ExactAnalystBudgetSnapshot, type ExactAnalystExecutionPlanSnapshot, type ExactAnalystRunCompletion, type ExactAnalystRunEvent, ExactAnalystRunExecutionError, type ExactAnalystRunPolicySnapshot, type ExactAnalystRunResult, type ExactAnalystRunSummary, type ExactAnalystSnapshot, type ExactCapableAnalyst, type ExactExecutionComponentIdentity, type ExactExecutionComponentSnapshot, type ExactRegistryRunOpts, FAILURE_MODE_KIND_SPEC, FINDING_SUBJECT_GRAMMAR_PROMPT, FINDING_SUBJECT_KINDS, FINDING_SUBJECT_SYNTAX, type FindingSubject, type FindingSubjectKind, FindingSubjectStringSchema, type FindingsDiff, FindingsStore, IMPROVEMENT_KIND_SPEC, type JudgeAdapterOpts, KIND_EXPECTED_SUBJECTS, KNOWLEDGE_GAP_KIND_SPEC, KNOWLEDGE_POISONING_KIND_SPEC, type LoadedVerificationArtifacts, type MockTransportOpts, type PersistedFinding, type PreparedPublicAnalystBenchmark, type ProposalFinding, type ProposalFindingOrigin, type PublicAnalystBenchmarkDataset, type PublicAnalystBenchmarkModelConfig, type PublicBenchmarkDistributions, type PublicBenchmarkSelectionReport, type PublicBenchmarkValueDistribution, RAW_FINDING_SCHEMA_PROMPT, type RawAnalystEvidence, RawAnalystEvidenceSchema, type RawAnalystFinding, RawAnalystFindingSchema, type RegistryRunOpts, type RouterTransportOpts, type RunAnalystBenchmarkOptions, type RunCriticAdapterOpts, SKILL_USAGE_ANALYST, type SandboxSdkTransportOpts, type SemanticConceptJudgeAdapterOpts, SkillUsageAnalyst, type SkillUsageRecord, type SkillUsageReport, type SkillUsageScanConfig, type StepLabelAdapterOptions, type TraceAnalysisEngine, type TraceAnalysisEngineRequest, type TraceAnalysisEngineResult, type TraceAnalystDefinition, type TraceAnalystLimits, type TraceToolGroupName, type UpstreamPredictionAdapterOptions, type VerificationArtifactFile, type VerificationArtifactManifest, type VerificationArtifactRole, type VerificationAvailabilitySummary, type VerificationOutcome, type VerificationOutcomeSource, type VerificationOutcomeStatus, type VerificationResultFile, type VerifierAdapterOpts, adaptPublicBenchmarkFindings, agentRxBenchmarkCase, agentRxPredictionsToFindings, analystBenchmarkDependencyLockDigest, analystBenchmarkImplementationDigest, appendVerificationArtifactsToOtlp, assertExactRegistryRunOpts, assertProposalFindings, behavioralAnalyst, buildDefaultAnalystRegistry, buildSkillUsageReport, buildTraceToolsForGroup, codeTraceBenchCase, codeTracerPredictionsToFindings, coerceJson, coerceToFindingRows, compareAnalystRunners, computeFindingId, createChatClient, createDspyRlmTraceEngine, createJudgeAdapter, createPublicBenchmarkDirectRunner, createPublicBenchmarkRlmRunner, createRunCriticAdapter, createSemanticConceptJudgeAdapter, createTraceAnalyst, createVerifierAdapter, defaultIsMaterial, defineCustomAnalyst, defineTraceAnalyst, deriveEfficiencyFindings, diffFindings, emitControlIntegrityFindings, emitSkillUsageFindings, emptyPublicBenchmarkRunner, evidenceRefsFromRawFinding, findingSubjectGrammarPromptFor, isProposalFinding, liftSeverity, loadCodeTraceVerificationArtifacts, loadPublicBenchmarkRows, makeFinding, makeProposalFinding, normalizeAgentRxCategory, normalizeBenchmarkLabel, parseFindingSubject, parseRawFinding, parseVerificationOutcome, preparePublicAnalystBenchmark, publicBenchmarkDistributions, publicBenchmarkProtocolSha256, publicBenchmarkSelectionReport, readAnalystBenchmarkArtifact, registryBenchmarkRunner, renderAgentRxCalibrationMarkdown, renderAnalystBenchmarkMarkdown, renderCodeTraceCalibrationMarkdown, renderFindingSubject, renderPriorFindings, renderUpstreamFindings, resolveTraceAnalystLimits, roundAgentRxStep, runAnalystBenchmark, runAnalystBenchmarkCommand, runTraceAnalyst, scoreAnalystFindings, selectPublicBenchmarkRows, stripCodeFences, summarizeAgentRxCalibration, summarizeCodeTraceCalibration, traceStoreEvidenceResolver };
|
|
712
659
|
//# sourceMappingURL=index.d.ts.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"index.d.ts","names":[],"sources":["../../src/analyst/adapters.ts","../../src/analyst/benchmark-agentrx-calibration.ts","../../src/analyst/benchmark-dataset-types.ts","../../src/analyst/benchmark-dataset-agentrx.ts","../../src/analyst/benchmark-dataset-codetrace.ts","../../src/analyst/benchmark-dataset-utils.ts","../../src/analyst/benchmark-verification-outcome.ts","../../src/analyst/benchmark-verification-artifacts.ts","../../src/analyst/benchmark-public-types.ts","../../src/analyst/benchmark-public-adapters.ts","../../src/analyst/benchmark-public-data.ts","../../src/analyst/benchmark-public-model.ts","../../src/analyst/benchmark-
|
|
1
|
+
{"version":3,"file":"index.d.ts","names":[],"sources":["../../src/analyst/adapters.ts","../../src/analyst/benchmark-agentrx-calibration.ts","../../src/analyst/benchmark-dataset-types.ts","../../src/analyst/benchmark-dataset-agentrx.ts","../../src/analyst/benchmark-dataset-codetrace.ts","../../src/analyst/benchmark-dataset-utils.ts","../../src/analyst/benchmark-verification-outcome.ts","../../src/analyst/benchmark-verification-artifacts.ts","../../src/analyst/benchmark-public-types.ts","../../src/analyst/benchmark-public-adapters.ts","../../src/analyst/benchmark-public-data.ts","../../src/analyst/benchmark-public-model.ts","../../src/analyst/benchmark-public-rlm.ts","../../src/analyst/benchmark-comparison.ts","../../src/analyst/benchmark-public-calibration.ts","../../src/analyst/benchmark-command-artifact.ts","../../src/analyst/benchmark-command-result.ts","../../src/analyst/benchmark-command.ts","../../src/analyst/benchmark-implementation.ts","../../src/analyst/benchmark-report.ts","../../src/analyst/parse-tolerant.ts","../../src/analyst/proposal-findings.ts"],"mappings":";;;;;;;;;;;;iBA4CgB,aAAa,GAAG,WAAgB;UAe/B,oBAAoB;EACnC;EACA;EACA,UAAU,mBAAmB;;;;;EAK7B,UAAU,KAAK,cAAc;;iBAGf,sBAAsB,KAAK,MAAM,oBAAoB,OAAO,QAAQ;UAwEnE;EACf;EACA;EACA,SAAS;;EAET;;iBAGc,uBAAuB,OAAM,uBAA4B,QAAQ;UAiEhE;EACf;EACA;EACA,OAAO;;EAEP,MAAM;;EAEN,OAAO;;EAEP;;iBAGc,mBAAmB,MAAM,mBAAmB,QAAQ;UAiDnD;EACf;EACA;;EAEA,UAAU,KAAK;;EAEf;;iBAGc,kCACd,OAAM,kCACL,QAAQ;;;cC5RE;UAEI;EACf;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;;UAGe;EACf;EACA;EACA;EACA,SAAS;;iBAGK,4BACd,QAAQ,wBACR,2BACC;iBAkBa,iCAAiC,SAAS;;;KCtD9C;UAEK;EACf,YAAY;EACZ;EACA;EACA;EACA;EACA;;UAGe;EACf,eAAe;EACf,mBAAmB;EACnB;IAAe,YAAY;IAAY;;EACvC,wBAAwB;EACxB;EACA;EACA;;UAGe;EACf,UAAU;EACV;EACA;EACA;EACA;;UAGe;EACf,UAAU;EACV,mBAAmB;EACnB;EACA;EACA;EACA;EACA;;UAGe;EACf;EACA;EACA;EACA;;UAGe;EACf;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;;UAGe;EACf;EACA;EACA;EACA,iBAAiB;;KAGP,0CAEC,sCACA,iCACA;UAEI;EACf;EACA;EACA;EACA;;EAEA;EACA;EACA;EACA;EACA;EACA;EACA,oCAAoC;;UAGrB;EACf,eAAe;EACf,WAAW,sBAAsB;;KAGvB;UAEK;;;;;EAKf,WAAW;;UAGI,kCACP,yBACN;UAEa,oCAAoC;EACnD;;EAEA;;UAGe,yCAAyC;EACxD;EACA;EACA;EACA;;UAGe,2CACP,kCACN;;;iBCtGY,qBAAqB,QACnC,KAAK,YACL,OAAO,QACP,UAAS,8BACR,qBAAqB;;iBAoGR,6BACd,mBAAmB,YACnB,iBACA,UAAS,mCACR;iBA6Ea,yBAAyB;;iBAoNzB,iBAAiB;;;iBC7YjB,mBAAmB,QACjC,KAAK,mBACL,OAAO,QACP,UAAS,4BACR,qBAAqB;;iBAwFR,gCACd,2BACA,aAAa,uBACb,UAAS,qCACR;;;iBClHa,wBAAwB;;;KCA5B;UAEK;EACf;EACA;EACA,QAAQ;;UAGO;EACf,QAAQ;EACR;EAKA;IAAe;IAAe;;EAC9B,SAAS;EACT;EACA;EACA;EACA;;UAGe;EACf;EACA;;iBA8Ec,yBACd,gBAAgB,2BACf;;;cChGU;KAED;UAEK;EACf,MAAM;EACN;EACA;EACA;EACA;EACA;;UAGe;EACf;EACA;EACA,SAAS;EACT;EACA;EACA;EACA;EACA;EACA,OAAO;EACP,cAAc;EACd,UAAU,OAAO;;UAGF;EACf,UAAU;EACV,SAAS;EACT,OAAO,MAAM;IAA6B;;;iBAatB,mCAAmC;EACvD;EACA,KAAK;EACL;IACE,QAAQ;iBAwHI,kCACd,kBACA,iBACA,WAAW,6BACX;;;KChLU;UAEK;EACf;EACA;EACA;EACA;EACA;;EAEA,UAAU;;EAEV;EACA;IACE,SAAS;IACT;IACA;IACA;IACA;;EAEF,aAAa;EACb;IACE;IACA;;;EAGF,mBAAmB;;UAGJ;EACf,OAAO,qBAAqB;EAC5B;EACA;EACA;EACA,YAAY;IACV;IACA;IACA;;EAEF,uBAAuB;EACvB,WAAW;;UAGI;EACf;EACA;EACA,QAAQ;;UAGO;EACf,OAAO;EACP,OAAO;EACP,OAAO;EACP,YAAY;EACZ,QAAQ;;UAGO;EACf;EACA;EACA;EACA;EACA;EACA;EACA,QAAQ;EACR,UAAU;;;;iBC5DI,8BAA8B,uBAAuB;iBAiBrD,6BACd,SAAS,+BACT,sBACA,mBAAmB,kBACnB,oBACC;;;iBCgBmB,wBACpB,eACC,QAAQ,MAAM;iBA2BD,0BACd,SAAS,+BACT,eAAe,2BACf;EAAW;EAAe;IACzB,MAAM;iBAwBO,6BACd,SAAS,+BACT,eAAe,4BACd;iBAyCa,+BACd,SAAS,+BACT,iBAAiB,2BACjB,mBAAmB,2BACnB,eACC;iBAcmB,8BAA8B;EAClD,SAAS;EACT;EACA;EACA;EACA;EACA;EACA;IACE,QAAQ;;;;iBC7HI,kCACd,SAAS,+BACT,QAAQ,oCACP,uBAAuB;iBAofV,8BAA8B,SAAS;cAoE1C;;;;iBCljBG,+BACd,SAAS,+BACT,QAAQ,oCACP,uBAAuB;;;KC5Dd;UAoBK;EACf,QAAQ;EACR;;EAEA;;EAEA;;EAEA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;;UAGe;EACf;EACA;EACA,SAAS;;iBASK,sBACd,QAAQ,wBACR;EACE;EACA;EACA;EACA;EACA;IAED;;;UCnEc;EACf;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;;EAEA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;;UAGe;EACf;EACA;EACA,SAAS;;iBAGK,8BACd,QAAQ,yBACP;iBAca,mCAAmC,SAAS;;;UCrC3C;EACf;EACA;EACA;IACE,SAAS;IACT;IACA;IACA;IACA;IACA,YAAY;MAAQ;MAAiB;MAAsB;;IAC3D,uBAAuB;IACvB,0BAA0B;IAC1B;MACE;MACA;MACA;MACA,QAAQ;;IAEV;MACE;MACA;MACA;MACA;MACA;MACA;MACA;MACA;MACA;MACA;;;EAGJ,QAAQ;EACR,aAAa;EACb,uBAAuB;EACvB,qBAAqB;;UAGN;EACf;EACA;EACA;EACA;IACE;IACA;IACA;;;UAIa;EACf;IACE,SAAS;IACT;IACA;IACA;MACE;MACA;MACA;;IAEF;IACA;IACA;IACA;IACA;IACA;IACA;IACA;IACA;IACA;;EAEF;IACE;IACA;IACA;IACA,YAAY;MAAQ;MAAiB;MAAsB;;IAC3D;IACA;;;UAIa;EACf;EACA;EACA;EACA;EACA,UAAU;;UAGK;EACf;EACA;EACA;EACA;IACE;IACA;IACA;IACA;IACA;IACA;;EAEF;EACA;IACE;IACA;IACA;;EAEF;IACE;IACA;IACA;IACA;IACA;IACA;;;UAIa;EACf;EACA;EACA;EACA,aAAa;EACb;;cAGW;cACA;cACA;cACA;;;iBCxHS,6BACpB,eACC,QAAQ;;;UCyDM;EACf,uBACE,SAAS,+BACT,QAAQ,sCACL,uBAAuB;;UAGb;EACf,SAAS;EACT;EACA;EACA;EACA;EACA;EACA;EACA,OAAO;EACP;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;;iBAKoB,2BACpB,yBACA,MAAK,OAAO,YACZ,eAAc,sCACb;cA8QU;;;cC7XA;cAEA;cAEA;cAOA;;;;;;;;cAUA;cAEA;cAGA;cAGA;cAmFA;iBAGG;iBAIA;;;iBCpHA,+BACd,QAAQ,wBACR,uBAAsB;;;;;;;;;;;;;;;iBCQR,gBAAgB;;;;;iBAgBhB,WAAW;;;;;;;iBAeX,oBAAoB;;;;iBCXpB,kBAAkB,mBAAmB,WAAW;;;;;;iBAShD,uBACd,mBACA,mBACC,cAAc"}
|
package/dist/analyst/index.js
CHANGED
|
@@ -1,10 +1,12 @@
|
|
|
1
|
-
import {
|
|
2
|
-
import { i as
|
|
3
|
-
import {
|
|
1
|
+
import { i as CostLedger } from "../cost-ledger-CZ9diLxY.js";
|
|
2
|
+
import { C as createChatClient, _ as CONTROL_INTEGRITY_ANALYST, b as behavioralAnalyst, f as DEFAULT_TRACE_ANALYST_KINDS, g as FAILURE_MODE_KIND_SPEC, h as IMPROVEMENT_KIND_SPEC, i as assertExactRegistryRunOpts, m as KNOWLEDGE_GAP_KIND_SPEC, n as AnalystRegistry, p as KNOWLEDGE_POISONING_KIND_SPEC, r as ExactAnalystRunExecutionError, t as buildDefaultAnalystRegistry, v as ControlIntegrityAnalyst, x as deriveEfficiencyFindings, y as emitControlIntegrityFindings } from "../default-registry-BgJJItGr.js";
|
|
3
|
+
import { A as coerceJson, B as renderFindingSubject, D as RawAnalystFindingSchema, E as RawAnalystEvidenceSchema, F as FINDING_SUBJECT_SYNTAX, G as resolveTraceAnalystLimits, I as FindingSubjectStringSchema, L as KIND_EXPECTED_SUBJECTS, M as stripCodeFences, N as FINDING_SUBJECT_GRAMMAR_PROMPT, O as evidenceRefsFromRawFinding, P as FINDING_SUBJECT_KINDS, R as findingSubjectGrammarPromptFor, T as RAW_FINDING_SCHEMA_PROMPT, W as DEFAULT_TRACE_ANALYST_LIMITS, a as buildTraceToolsForGroup, i as runTraceAnalyst, j as coerceToFindingRows, k as parseRawFinding, n as renderPriorFindings, r as renderUpstreamFindings, t as createTraceAnalyst, w as ANALYST_SEVERITIES, z as parseFindingSubject } from "../kind-factory-CFxA0JQX.js";
|
|
4
|
+
import { a as computeFindingId, i as validateUsageSettlementTimeout, n as settleUsageReceiptFromCostLedger, o as makeFinding, s as makeProposalFinding } from "../usage-receipt-CgxMEBZq.js";
|
|
4
5
|
import { n as isProposalFinding, t as assertProposalFindings } from "../proposal-findings-2GIUo1et.js";
|
|
5
|
-
import { a as RunCritic, c as buildSkillUsageReport, d as defaultIsMaterial, f as diffFindings, i as runSemanticConceptJudge, l as emitSkillUsageFindings, n as SEMANTIC_CONCEPT_JUDGE_VERSION, o as SKILL_USAGE_ANALYST, s as SkillUsageAnalyst, u as FindingsStore } from "../semantic-concept-judge-
|
|
6
|
-
import {
|
|
7
|
-
import {
|
|
6
|
+
import { a as RunCritic, c as buildSkillUsageReport, d as defaultIsMaterial, f as diffFindings, h as defineTraceAnalyst, i as runSemanticConceptJudge, l as emitSkillUsageFindings, m as defineCustomAnalyst, n as SEMANTIC_CONCEPT_JUDGE_VERSION, o as SKILL_USAGE_ANALYST, s as SkillUsageAnalyst, u as FindingsStore } from "../semantic-concept-judge-BuIJ9IfB.js";
|
|
7
|
+
import { t as createDspyRlmTraceEngine } from "../dspy-rlm-engine-DTkVyDX-.js";
|
|
8
|
+
import { a as scoreAnalystFindings, n as runAnalystBenchmark, r as traceStoreEvidenceResolver, t as registryBenchmarkRunner } from "../benchmark-CYtcIF2V.js";
|
|
9
|
+
import { A as ANALYST_BENCHMARK_IMPLEMENTATION_FILES, B as summarizeAgentRxCalibration, C as ANALYST_BENCHMARK_DEPENDENCY_LOCK_DIGEST_ALGORITHM, D as ANALYST_BENCHMARK_EVIDENCE_IMPLEMENTATION_SHA256, E as ANALYST_BENCHMARK_EVIDENCE_DEPENDENCY_LOCK_SHA256, F as ANALYST_BENCHMARK_LOCAL_RECEIPT_FILE, G as normalizeAgentRxCategory, H as codeTracerPredictionsToFindings, I as ANALYST_BENCHMARK_MANIFEST_FILE, K as roundAgentRxStep, L as ANALYST_BENCHMARK_OBSERVATIONS_FILE, M as analystBenchmarkDependencyLockDigest, N as analystBenchmarkImplementationDigest, O as ANALYST_BENCHMARK_EVIDENCE_PACKAGE_VERSION, P as ANALYST_BENCHMARK_COST_LEDGER_FILE, R as AGENT_RX_UPSTREAM_REVISION, S as emptyPublicBenchmarkRunner, T as ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256, U as agentRxBenchmarkCase, V as codeTraceBenchCase, W as agentRxPredictionsToFindings, _ as DEFAULT_MAX_VERIFICATION_ARTIFACT_BYTES, a as renderCodeTraceCalibrationMarkdown, b as parseVerificationOutcome, c as createPublicBenchmarkRlmRunner, d as publicBenchmarkProtocolSha256, f as loadPublicBenchmarkRows, g as selectPublicBenchmarkRows, h as publicBenchmarkSelectionReport, i as readAnalystBenchmarkArtifact, j as ANALYST_BENCHMARK_IMPLEMENTATION_SHA256, k as ANALYST_BENCHMARK_IMPLEMENTATION_DIGEST_ALGORITHM, l as CODE_TRACE_BENCH_ANALYST_PROMPT, m as publicBenchmarkDistributions, n as runAnalystBenchmarkCommand, o as summarizeCodeTraceCalibration, p as preparePublicAnalystBenchmark, q as normalizeBenchmarkLabel, r as renderAnalystBenchmarkMarkdown, s as compareAnalystRunners, t as ANALYST_BENCHMARK_HELP, u as createPublicBenchmarkDirectRunner, v as appendVerificationArtifactsToOtlp, w as ANALYST_BENCHMARK_DEPENDENCY_LOCK_FILES, x as adaptPublicBenchmarkFindings, y as loadCodeTraceVerificationArtifacts, z as renderAgentRxCalibrationMarkdown } from "../benchmark-command-BKfjOBJ5.js";
|
|
8
10
|
//#region src/analyst/adapters.ts
|
|
9
11
|
/**
|
|
10
12
|
* Adapter factories — lift each existing agent-eval primitive into the
|
|
@@ -292,23 +294,6 @@ function createSemanticConceptJudgeAdapter(opts = {}) {
|
|
|
292
294
|
};
|
|
293
295
|
}
|
|
294
296
|
//#endregion
|
|
295
|
-
|
|
296
|
-
function defineTraceAnalyst(options) {
|
|
297
|
-
if (!options.id.trim()) throw new TypeError("defineTraceAnalyst: id must not be empty");
|
|
298
|
-
if (!options.description.trim()) throw new TypeError("defineTraceAnalyst: description must not be empty");
|
|
299
|
-
if (options.cost === void 0) throw new TypeError("defineTraceAnalyst: cost must be declared");
|
|
300
|
-
if ("executionConfig" in options && (!options.executionConfig || typeof options.executionConfig !== "object" || Array.isArray(options.executionConfig))) throw new TypeError("defineTraceAnalyst: executionConfig must be an object");
|
|
301
|
-
return {
|
|
302
|
-
id: options.id,
|
|
303
|
-
description: options.description,
|
|
304
|
-
version: options.version ?? "1.0.0",
|
|
305
|
-
inputKind: "trace-store",
|
|
306
|
-
cost: options.cost,
|
|
307
|
-
analyze: options.analyze,
|
|
308
|
-
..."executionConfig" in options ? { executionConfig: options.executionConfig } : {}
|
|
309
|
-
};
|
|
310
|
-
}
|
|
311
|
-
//#endregion
|
|
312
|
-
export { AGENT_RX_UPSTREAM_REVISION, ANALYST_BENCHMARK_COST_LEDGER_FILE, ANALYST_BENCHMARK_DEPENDENCY_LOCK_DIGEST_ALGORITHM, ANALYST_BENCHMARK_DEPENDENCY_LOCK_FILES, ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256, ANALYST_BENCHMARK_EVIDENCE_DEPENDENCY_LOCK_SHA256, ANALYST_BENCHMARK_EVIDENCE_PACKAGE_VERSION, ANALYST_BENCHMARK_HELP, ANALYST_BENCHMARK_IMPLEMENTATION_DIGEST_ALGORITHM, ANALYST_BENCHMARK_IMPLEMENTATION_FILES, ANALYST_BENCHMARK_IMPLEMENTATION_SHA256, ANALYST_BENCHMARK_LOCAL_RECEIPT_FILE, ANALYST_BENCHMARK_MANIFEST_FILE, ANALYST_BENCHMARK_OBSERVATIONS_FILE, ANALYST_SEVERITIES, AnalystRegistry, CODE_TRACE_BENCH_ANALYST_PROMPT, CONTROL_INTEGRITY_ANALYST, ControlIntegrityAnalyst, DEFAULT_MAX_VERIFICATION_ARTIFACT_BYTES, DEFAULT_TRACE_ANALYST_KINDS, ExactAnalystRunExecutionError, FAILURE_MODE_KIND_SPEC, FINDING_SUBJECT_GRAMMAR_PROMPT, FINDING_SUBJECT_KINDS, FINDING_SUBJECT_SYNTAX, FindingSubjectStringSchema, FindingsStore, IMPROVEMENT_KIND_SPEC, KIND_EXPECTED_SUBJECTS, KNOWLEDGE_GAP_KIND_SPEC, KNOWLEDGE_POISONING_KIND_SPEC, RAW_FINDING_SCHEMA_PROMPT, RawAnalystEvidenceSchema, RawAnalystFindingSchema, SKILL_USAGE_ANALYST, SkillUsageAnalyst, adaptPublicBenchmarkFindings, agentRxBenchmarkCase, agentRxPredictionsToFindings, analystBenchmarkDependencyLockDigest, analystBenchmarkImplementationDigest, appendVerificationArtifactsToOtlp, assertExactRegistryRunOpts, assertProposalFindings, behavioralAnalyst, buildDefaultAnalystRegistry, buildSkillUsageReport, buildTraceToolsForGroup, codeTraceBenchCase, codeTracerPredictionsToFindings, coerceJson, coerceToFindingRows, compareAnalystRunners, computeFindingId, createAnalystAi, createChatClient, createJudgeAdapter, createPublicBenchmarkModelRunner, createRunCriticAdapter, createSemanticConceptJudgeAdapter, createTraceAnalystKind, createVerifierAdapter, defaultIsMaterial, defineTraceAnalyst, deriveEfficiencyFindings, diffFindings, emitControlIntegrityFindings, emitSkillUsageFindings, emptyPublicBenchmarkRunner, evidenceRefsFromRawFinding, findingSubjectGrammarPromptFor, isProposalFinding, liftSeverity, loadCodeTraceVerificationArtifacts, loadPublicBenchmarkRows, makeFinding, makeProposalFinding, normalizeAgentRxCategory, normalizeBenchmarkLabel, parseFindingSubject, parseRawFinding, parseVerificationOutcome, preparePublicAnalystBenchmark, publicBenchmarkDistributions, publicBenchmarkProtocolSha256, publicBenchmarkSelectionReport, readAnalystBenchmarkArtifact, registryBenchmarkRunner, renderAgentRxCalibrationMarkdown, renderAnalystBenchmarkMarkdown, renderCodeTraceCalibrationMarkdown, renderFindingSubject, renderPriorFindings, renderUpstreamFindings, roundAgentRxStep, runAnalystBenchmark, runAnalystBenchmarkCommand, scoreAnalystFindings, selectPublicBenchmarkRows, stripCodeFences, structureFindings, summarizeAgentRxCalibration, summarizeCodeTraceCalibration, traceStoreEvidenceResolver };
|
|
297
|
+
export { AGENT_RX_UPSTREAM_REVISION, ANALYST_BENCHMARK_COST_LEDGER_FILE, ANALYST_BENCHMARK_DEPENDENCY_LOCK_DIGEST_ALGORITHM, ANALYST_BENCHMARK_DEPENDENCY_LOCK_FILES, ANALYST_BENCHMARK_DEPENDENCY_LOCK_SHA256, ANALYST_BENCHMARK_EVIDENCE_DEPENDENCY_LOCK_SHA256, ANALYST_BENCHMARK_EVIDENCE_IMPLEMENTATION_SHA256, ANALYST_BENCHMARK_EVIDENCE_PACKAGE_VERSION, ANALYST_BENCHMARK_HELP, ANALYST_BENCHMARK_IMPLEMENTATION_DIGEST_ALGORITHM, ANALYST_BENCHMARK_IMPLEMENTATION_FILES, ANALYST_BENCHMARK_IMPLEMENTATION_SHA256, ANALYST_BENCHMARK_LOCAL_RECEIPT_FILE, ANALYST_BENCHMARK_MANIFEST_FILE, ANALYST_BENCHMARK_OBSERVATIONS_FILE, ANALYST_SEVERITIES, AnalystRegistry, CODE_TRACE_BENCH_ANALYST_PROMPT, CONTROL_INTEGRITY_ANALYST, ControlIntegrityAnalyst, DEFAULT_MAX_VERIFICATION_ARTIFACT_BYTES, DEFAULT_TRACE_ANALYST_KINDS, DEFAULT_TRACE_ANALYST_LIMITS, ExactAnalystRunExecutionError, FAILURE_MODE_KIND_SPEC, FINDING_SUBJECT_GRAMMAR_PROMPT, FINDING_SUBJECT_KINDS, FINDING_SUBJECT_SYNTAX, FindingSubjectStringSchema, FindingsStore, IMPROVEMENT_KIND_SPEC, KIND_EXPECTED_SUBJECTS, KNOWLEDGE_GAP_KIND_SPEC, KNOWLEDGE_POISONING_KIND_SPEC, RAW_FINDING_SCHEMA_PROMPT, RawAnalystEvidenceSchema, RawAnalystFindingSchema, SKILL_USAGE_ANALYST, SkillUsageAnalyst, adaptPublicBenchmarkFindings, agentRxBenchmarkCase, agentRxPredictionsToFindings, analystBenchmarkDependencyLockDigest, analystBenchmarkImplementationDigest, appendVerificationArtifactsToOtlp, assertExactRegistryRunOpts, assertProposalFindings, behavioralAnalyst, buildDefaultAnalystRegistry, buildSkillUsageReport, buildTraceToolsForGroup, codeTraceBenchCase, codeTracerPredictionsToFindings, coerceJson, coerceToFindingRows, compareAnalystRunners, computeFindingId, createChatClient, createDspyRlmTraceEngine, createJudgeAdapter, createPublicBenchmarkDirectRunner, createPublicBenchmarkRlmRunner, createRunCriticAdapter, createSemanticConceptJudgeAdapter, createTraceAnalyst, createVerifierAdapter, defaultIsMaterial, defineCustomAnalyst, defineTraceAnalyst, deriveEfficiencyFindings, diffFindings, emitControlIntegrityFindings, emitSkillUsageFindings, emptyPublicBenchmarkRunner, evidenceRefsFromRawFinding, findingSubjectGrammarPromptFor, isProposalFinding, liftSeverity, loadCodeTraceVerificationArtifacts, loadPublicBenchmarkRows, makeFinding, makeProposalFinding, normalizeAgentRxCategory, normalizeBenchmarkLabel, parseFindingSubject, parseRawFinding, parseVerificationOutcome, preparePublicAnalystBenchmark, publicBenchmarkDistributions, publicBenchmarkProtocolSha256, publicBenchmarkSelectionReport, readAnalystBenchmarkArtifact, registryBenchmarkRunner, renderAgentRxCalibrationMarkdown, renderAnalystBenchmarkMarkdown, renderCodeTraceCalibrationMarkdown, renderFindingSubject, renderPriorFindings, renderUpstreamFindings, resolveTraceAnalystLimits, roundAgentRxStep, runAnalystBenchmark, runAnalystBenchmarkCommand, runTraceAnalyst, scoreAnalystFindings, selectPublicBenchmarkRows, stripCodeFences, summarizeAgentRxCalibration, summarizeCodeTraceCalibration, traceStoreEvidenceResolver };
|
|
313
298
|
|
|
314
299
|
//# sourceMappingURL=index.js.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"index.js","names":[],"sources":["../../src/analyst/adapters.ts","../../src/analyst/define.ts"],"sourcesContent":["/**\n * Adapter factories — lift each existing agent-eval primitive into the\n * Analyst contract without re-implementing it.\n *\n * Five primitives, five factories. Each one:\n * - Builds an Analyst with a stable id (caller chooses; defaults\n * given), a sensible default `inputKind`, a version derived from\n * the wrapped primitive's version + an adapter revision, and an\n * `analyze()` that calls the primitive and lifts its output to\n * AnalystFinding[] using `makeFinding()`.\n * - Maps severities: the existing `Severity` ('critical' | 'major' |\n * 'minor' | 'info') projects onto AnalystSeverity ('critical' |\n * 'high' | 'medium' | 'low' | 'info'); 'major' → 'high', 'minor' →\n * 'medium'. Domain analysts that want finer-grained mapping override.\n *\n * Adapters never own state. Calling the same factory twice with the\n * same primitive instance is safe.\n */\n\nimport { CostLedger } from '../cost-ledger'\nimport type {\n Finding as LayerFinding,\n Severity as LayerSeverity,\n MultiLayerVerifier,\n VerifyOptions,\n} from '../multi-layer-verifier'\nimport { RunCritic, type RunTrace } from '../run-critic'\nimport {\n runSemanticConceptJudge,\n SEMANTIC_CONCEPT_JUDGE_VERSION,\n type SemanticConceptJudgeInput,\n type SemanticConceptJudgeOptions,\n type SemanticConceptJudgeResult,\n} from '../semantic-concept-judge'\nimport type { JudgeFn, JudgeInput, JudgeScore } from '../types'\nimport type { ChatClient } from './chat-client'\nimport type { Analyst, AnalystFinding, AnalystSeverity } from './types'\nimport { makeFinding } from './types'\nimport { settleUsageReceiptFromCostLedger, validateUsageSettlementTimeout } from './usage-receipt'\n\nconst ADAPTER_REV = '1'\n\n// ── Severity bridges ───────────────────────────────────────────────\n\nexport function liftSeverity(s: LayerSeverity): AnalystSeverity {\n switch (s) {\n case 'critical':\n return 'critical'\n case 'major':\n return 'high'\n case 'minor':\n return 'medium'\n case 'info':\n return 'info'\n }\n}\n\n// ── 1. MultiLayerVerifier → Analyst ─────────────────────────────────\n\nexport interface VerifierAdapterOpts<Env> {\n id?: string\n area?: string\n verifier: MultiLayerVerifier<Env>\n /**\n * The verifier expects an `env` per run. Adapters take it from\n * `AnalystRunInputs.custom[<id>]` via the registry's 'custom' routing.\n */\n options?: Omit<VerifyOptions<Env>, 'env'>\n}\n\nexport function createVerifierAdapter<Env>(opts: VerifierAdapterOpts<Env>): Analyst<Env> {\n const id = opts.id ?? 'multi-layer-verifier'\n const area = opts.area ?? 'verification'\n return {\n id,\n description:\n \"Runs a MultiLayerVerifier and lifts each layer's findings into the analyst envelope.\",\n inputKind: 'custom',\n cost: { kind: 'deterministic' },\n version: `verifier-${ADAPTER_REV}`,\n async analyze(env, ctx) {\n const report = await opts.verifier.run({ env, ...opts.options })\n const out: AnalystFinding[] = []\n for (const layer of report.layers) {\n for (const finding of layer.findings) {\n out.push(liftLayerFinding(id, area, layer.layer, finding))\n }\n // Layer-level signal: a failed/error layer is itself a finding\n // even if it didn't emit per-finding rows.\n if (layer.status === 'fail' || layer.status === 'error' || layer.status === 'timeout') {\n out.push(\n makeFinding({\n analyst_id: id,\n area,\n subject: layer.layer,\n claim: `layer \"${layer.layer}\" ${layer.status}: ${layer.reason ?? 'no reason given'}`,\n severity:\n layer.status === 'error' ? 'high' : layer.status === 'timeout' ? 'medium' : 'high',\n confidence: 1,\n evidence_refs: [],\n metadata: {\n layer_status: layer.status,\n duration_ms: layer.durationMs,\n score: layer.score,\n diagnostics: layer.diagnostics,\n },\n }),\n )\n }\n }\n ctx.log?.('verifier complete', {\n layers: report.layers.length,\n blended: report.blendedScore,\n all_pass: report.allPass,\n })\n return out\n },\n }\n}\n\nfunction liftLayerFinding(\n analyst_id: string,\n area: string,\n layer: string,\n f: LayerFinding,\n): AnalystFinding {\n return makeFinding({\n analyst_id,\n area,\n subject: f.layer ?? layer,\n claim: f.message,\n severity: liftSeverity(f.severity),\n confidence: 0.85,\n evidence_refs: f.evidence\n ? [{ kind: 'artifact', uri: 'inline:evidence', excerpt: f.evidence }]\n : [],\n metadata: f.detail,\n })\n}\n\n// ── 2. RunCritic → Analyst ──────────────────────────────────────────\n\nexport interface RunCriticAdapterOpts {\n id?: string\n area?: string\n critic?: RunCritic\n /** Optional threshold below which a dimension is reported as a finding. Default 0.5. */\n threshold?: number\n}\n\nexport function createRunCriticAdapter(opts: RunCriticAdapterOpts = {}): Analyst<RunTrace> {\n const id = opts.id ?? 'run-critic'\n const area = opts.area ?? 'run-quality'\n const critic = opts.critic ?? new RunCritic()\n const threshold = opts.threshold ?? 0.5\n return {\n id,\n description:\n 'Scores a single run across success / grounding / drift / tool-quality and surfaces below-threshold dimensions.',\n inputKind: 'custom',\n cost: { kind: 'deterministic' },\n version: `run-critic-${ADAPTER_REV}`,\n async analyze(trace) {\n const score = critic.scoreTrace(trace)\n const out: AnalystFinding[] = []\n const dims: Array<[keyof typeof score, AnalystSeverity, string]> = [\n ['success', 'critical', 'run did not complete successfully'],\n ['goalProgress', 'high', 'goal progress is low'],\n ['repoGroundedness', 'high', 'output is poorly grounded in the repository'],\n ['toolUseQuality', 'medium', 'tool use quality is low'],\n ['patchQuality', 'medium', 'no real patch/edit evidence'],\n ['testReality', 'high', 'no real test/build evidence'],\n ['finalGate', 'critical', 'final gate is blocking'],\n ]\n for (const [dim, sev, msg] of dims) {\n const value = score[dim] as number\n if (typeof value === 'number' && value < threshold) {\n out.push(\n makeFinding({\n analyst_id: id,\n area,\n subject: dim,\n claim: msg,\n rationale: `${dim}=${value.toFixed(2)} below threshold ${threshold}`,\n severity: sev,\n confidence: 1,\n evidence_refs: [],\n metadata: { dimension: dim, value, threshold, run_id: trace.run.runId },\n }),\n )\n }\n }\n // Drift penalty is high → surface as a finding (inverse threshold).\n if (score.driftPenalty > 1 - threshold) {\n out.push(\n makeFinding({\n analyst_id: id,\n area,\n subject: 'drift',\n claim: 'agent output drifted from repository signal',\n rationale: `driftPenalty=${score.driftPenalty.toFixed(2)}`,\n severity: 'medium',\n confidence: 0.9,\n evidence_refs: [],\n metadata: { drift_penalty: score.driftPenalty, notes: score.notes },\n }),\n )\n }\n return out\n },\n }\n}\n\n// ── 3. JudgeFn → Analyst ────────────────────────────────────────────\n\nexport interface JudgeAdapterOpts {\n id?: string\n area?: string\n judge: JudgeFn\n /** Chat client passed to the JudgeFn. */\n chat: ChatClient\n /** Optional cost classification — most judges call an LLM. */\n cost?: Analyst['cost']\n /** Optional threshold below which a JudgeScore becomes a finding. Default 6 (on 0-10 scale). */\n threshold?: number\n}\n\nexport function createJudgeAdapter(opts: JudgeAdapterOpts): Analyst<JudgeInput> {\n const id = opts.id ?? 'judge'\n const area = opts.area ?? 'judge'\n const threshold = opts.threshold ?? 6\n return {\n id,\n description:\n 'Wraps an agent-eval JudgeFn into an analyst; below-threshold dimensions surface as findings.',\n inputKind: 'judge-input',\n cost: opts.cost ?? { kind: 'llm' },\n version: `judge-${ADAPTER_REV}`,\n async analyze(input) {\n const scores = await opts.judge(opts.chat, input)\n return scores\n .filter((s) => normalize10(s.score) < threshold)\n .map((s) => liftJudgeScore(id, area, s))\n },\n }\n}\n\nfunction normalize10(s: number): number {\n // JudgeScore convention is 0-10 but some judges emit 0-1. Coerce to 0-10.\n return s <= 1 ? s * 10 : s\n}\n\nfunction liftJudgeScore(analyst_id: string, area: string, s: JudgeScore): AnalystFinding {\n const score10 = normalize10(s.score)\n const severity: AnalystSeverity =\n score10 < 3 ? 'critical' : score10 < 5 ? 'high' : score10 < 7 ? 'medium' : 'low'\n return makeFinding({\n analyst_id,\n area,\n subject: s.dimension,\n claim: `${s.judgeName}/${s.dimension} scored ${score10.toFixed(1)}/10`,\n rationale: s.reasoning,\n severity,\n confidence: 0.8,\n evidence_refs: s.evidence\n ? [{ kind: 'artifact', uri: 'inline:evidence', excerpt: s.evidence }]\n : [],\n // Descriptive origin only. A caller may admit search feedback into candidate\n // generation, but final evaluation findings have no allowed proposal origin.\n derived_from_judge: true,\n metadata: { judge_name: s.judgeName, dimension: s.dimension, score_10: score10 },\n })\n}\n\n// ── 4. SemanticConceptJudge → Analyst ──────────────────────────────\n\nexport interface SemanticConceptJudgeAdapterOpts {\n id?: string\n area?: string\n /** Registry context owns cancellation and the per-analyst cost ledger. */\n options?: Omit<SemanticConceptJudgeOptions, 'costLedger' | 'signal'>\n /** Maximum post-cancellation wait for a provider receipt. Default 5 seconds. */\n settlementTimeoutMs?: number\n}\n\nexport function createSemanticConceptJudgeAdapter(\n opts: SemanticConceptJudgeAdapterOpts = {},\n): Analyst<SemanticConceptJudgeInput> {\n const id = opts.id ?? 'semantic-concept-judge'\n const area = opts.area ?? 'concept-coverage'\n const settlementTimeoutMs = validateUsageSettlementTimeout(opts.settlementTimeoutMs)\n return {\n id,\n description:\n 'Runs the semantic-concept judge and surfaces missing / weak concepts as findings.',\n inputKind: 'custom',\n cost: {\n kind: 'llm',\n models: opts.options?.model ? [opts.options.model] : undefined,\n settlement_timeout_ms: settlementTimeoutMs,\n },\n version: `${SEMANTIC_CONCEPT_JUDGE_VERSION}-adapter-${ADAPTER_REV}`,\n async analyze(input, ctx) {\n const costLedger = new CostLedger(ctx.budgetUsd)\n let result: SemanticConceptJudgeResult\n try {\n result = await runSemanticConceptJudge(input, {\n ...opts.options,\n costLedger,\n signal: ctx.signal,\n })\n } finally {\n const usage = await settleUsageReceiptFromCostLedger(costLedger, {\n channel: 'judge',\n timeoutMs: settlementTimeoutMs,\n })\n if (!usage.settled) {\n ctx.log?.('semantic-concept judge provider settlement timed out', {\n pending_calls: usage.pendingCalls,\n timeout_ms: settlementTimeoutMs,\n })\n }\n ctx.recordUsage?.(usage.receipt)\n }\n if (!result.available) {\n return [\n makeFinding({\n analyst_id: id,\n area,\n claim: 'semantic-concept judge unavailable',\n rationale: result.error,\n severity: 'info',\n confidence: 1,\n evidence_refs: [],\n metadata: { reason: result.error },\n }),\n ]\n }\n const out: AnalystFinding[] = []\n for (const f of result.findings) {\n // Only surface gaps: missing concepts or low scores. Concepts at\n // 7+/10 with present=true are not findings — they're successes.\n if (f.present && f.score >= 7) continue\n out.push(\n makeFinding({\n analyst_id: id,\n area,\n subject: f.concept,\n claim: f.present\n ? `concept \"${f.concept}\" is weak (${f.score}/10)`\n : `concept \"${f.concept}\" is missing`,\n rationale: f.evidence,\n severity: liftSeverity(f.severity),\n confidence: 0.85,\n evidence_refs: [{ kind: 'artifact', uri: 'inline:evidence', excerpt: f.evidence }],\n metadata: {\n concept: f.concept,\n present: f.present,\n score_10: f.score,\n },\n }),\n )\n }\n return out\n },\n }\n}\n","import type { TraceAnalysisStore } from '../trace-analyst/store'\nimport type { ExactCapableAnalyst } from './exact-types'\nimport type { Analyst, AnalystCost, AnalystFinding } from './types'\n\nexport interface DefineTraceAnalystOptions {\n id: string\n description: string\n version?: string\n cost: AnalystCost\n analyze: Analyst<TraceAnalysisStore>['analyze']\n}\n\nexport interface DefineExactTraceAnalystOptions extends DefineTraceAnalystOptions {\n /** Canonical JSON for behavior knobs not already bound by `version`. */\n executionConfig: Readonly<Record<string, unknown>>\n}\n\n/** Define a custom trace analyst without repeating fixed registry fields. */\nexport function defineTraceAnalyst(\n options: DefineExactTraceAnalystOptions,\n): ExactCapableAnalyst<TraceAnalysisStore>\nexport function defineTraceAnalyst(options: DefineTraceAnalystOptions): Analyst<TraceAnalysisStore>\nexport function defineTraceAnalyst(\n options: DefineTraceAnalystOptions | DefineExactTraceAnalystOptions,\n): Analyst<TraceAnalysisStore> | ExactCapableAnalyst<TraceAnalysisStore> {\n if (!options.id.trim()) throw new TypeError('defineTraceAnalyst: id must not be empty')\n if (!options.description.trim()) {\n throw new TypeError('defineTraceAnalyst: description must not be empty')\n }\n if (options.cost === undefined) {\n throw new TypeError('defineTraceAnalyst: cost must be declared')\n }\n if (\n 'executionConfig' in options &&\n (!options.executionConfig ||\n typeof options.executionConfig !== 'object' ||\n Array.isArray(options.executionConfig))\n ) {\n throw new TypeError('defineTraceAnalyst: executionConfig must be an object')\n }\n return {\n id: options.id,\n description: options.description,\n version: options.version ?? '1.0.0',\n inputKind: 'trace-store',\n cost: options.cost,\n analyze: options.analyze,\n ...('executionConfig' in options ? { executionConfig: options.executionConfig } : {}),\n }\n}\n\nexport type TraceAnalystAnalyze = (\n store: TraceAnalysisStore,\n context: Parameters<Analyst<TraceAnalysisStore>['analyze']>[1],\n) => Promise<AnalystFinding[]>\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;AAwCA,MAAM,cAAc;AAIpB,SAAgB,aAAa,GAAmC;CAC9D,QAAQ,GAAR;EACE,KAAK,YACH,OAAO;EACT,KAAK,SACH,OAAO;EACT,KAAK,SACH,OAAO;EACT,KAAK,QACH,OAAO;CACX;AACF;AAeA,SAAgB,sBAA2B,MAA8C;CACvF,MAAM,KAAK,KAAK,MAAM;CACtB,MAAM,OAAO,KAAK,QAAQ;CAC1B,OAAO;EACL;EACA,aACE;EACF,WAAW;EACX,MAAM,EAAE,MAAM,gBAAgB;EAC9B,SAAS,YAAY;EACrB,MAAM,QAAQ,KAAK,KAAK;GACtB,MAAM,SAAS,MAAM,KAAK,SAAS,IAAI;IAAE;IAAK,GAAG,KAAK;GAAQ,CAAC;GAC/D,MAAM,MAAwB,CAAC;GAC/B,KAAK,MAAM,SAAS,OAAO,QAAQ;IACjC,KAAK,MAAM,WAAW,MAAM,UAC1B,IAAI,KAAK,iBAAiB,IAAI,MAAM,MAAM,OAAO,OAAO,CAAC;IAI3D,IAAI,MAAM,WAAW,UAAU,MAAM,WAAW,WAAW,MAAM,WAAW,WAC1E,IAAI,KACF,YAAY;KACV,YAAY;KACZ;KACA,SAAS,MAAM;KACf,OAAO,UAAU,MAAM,MAAM,IAAI,MAAM,OAAO,IAAI,MAAM,UAAU;KAClE,UACE,MAAM,WAAW,UAAU,SAAS,MAAM,WAAW,YAAY,WAAW;KAC9E,YAAY;KACZ,eAAe,CAAC;KAChB,UAAU;MACR,cAAc,MAAM;MACpB,aAAa,MAAM;MACnB,OAAO,MAAM;MACb,aAAa,MAAM;KACrB;IACF,CAAC,CACH;GAEJ;GACA,IAAI,MAAM,qBAAqB;IAC7B,QAAQ,OAAO,OAAO;IACtB,SAAS,OAAO;IAChB,UAAU,OAAO;GACnB,CAAC;GACD,OAAO;EACT;CACF;AACF;AAEA,SAAS,iBACP,YACA,MACA,OACA,GACgB;CAChB,OAAO,YAAY;EACjB;EACA;EACA,SAAS,EAAE,SAAS;EACpB,OAAO,EAAE;EACT,UAAU,aAAa,EAAE,QAAQ;EACjC,YAAY;EACZ,eAAe,EAAE,WACb,CAAC;GAAE,MAAM;GAAY,KAAK;GAAmB,SAAS,EAAE;EAAS,CAAC,IAClE,CAAC;EACL,UAAU,EAAE;CACd,CAAC;AACH;AAYA,SAAgB,uBAAuB,OAA6B,CAAC,GAAsB;CACzF,MAAM,KAAK,KAAK,MAAM;CACtB,MAAM,OAAO,KAAK,QAAQ;CAC1B,MAAM,SAAS,KAAK,UAAU,IAAI,UAAU;CAC5C,MAAM,YAAY,KAAK,aAAa;CACpC,OAAO;EACL;EACA,aACE;EACF,WAAW;EACX,MAAM,EAAE,MAAM,gBAAgB;EAC9B,SAAS,cAAc;EACvB,MAAM,QAAQ,OAAO;GACnB,MAAM,QAAQ,OAAO,WAAW,KAAK;GACrC,MAAM,MAAwB,CAAC;GAU/B,KAAK,MAAM,CAAC,KAAK,KAAK,QAAQ;IAR5B;KAAC;KAAW;KAAY;IAAmC;IAC3D;KAAC;KAAgB;KAAQ;IAAsB;IAC/C;KAAC;KAAoB;KAAQ;IAA6C;IAC1E;KAAC;KAAkB;KAAU;IAAyB;IACtD;KAAC;KAAgB;KAAU;IAA6B;IACxD;KAAC;KAAe;KAAQ;IAA6B;IACrD;KAAC;KAAa;KAAY;IAAwB;GAEnB,GAAG;IAClC,MAAM,QAAQ,MAAM;IACpB,IAAI,OAAO,UAAU,YAAY,QAAQ,WACvC,IAAI,KACF,YAAY;KACV,YAAY;KACZ;KACA,SAAS;KACT,OAAO;KACP,WAAW,GAAG,IAAI,GAAG,MAAM,QAAQ,CAAC,EAAE,mBAAmB;KACzD,UAAU;KACV,YAAY;KACZ,eAAe,CAAC;KAChB,UAAU;MAAE,WAAW;MAAK;MAAO;MAAW,QAAQ,MAAM,IAAI;KAAM;IACxE,CAAC,CACH;GAEJ;GAEA,IAAI,MAAM,eAAe,IAAI,WAC3B,IAAI,KACF,YAAY;IACV,YAAY;IACZ;IACA,SAAS;IACT,OAAO;IACP,WAAW,gBAAgB,MAAM,aAAa,QAAQ,CAAC;IACvD,UAAU;IACV,YAAY;IACZ,eAAe,CAAC;IAChB,UAAU;KAAE,eAAe,MAAM;KAAc,OAAO,MAAM;IAAM;GACpE,CAAC,CACH;GAEF,OAAO;EACT;CACF;AACF;AAgBA,SAAgB,mBAAmB,MAA6C;CAC9E,MAAM,KAAK,KAAK,MAAM;CACtB,MAAM,OAAO,KAAK,QAAQ;CAC1B,MAAM,YAAY,KAAK,aAAa;CACpC,OAAO;EACL;EACA,aACE;EACF,WAAW;EACX,MAAM,KAAK,QAAQ,EAAE,MAAM,MAAM;EACjC,SAAS,SAAS;EAClB,MAAM,QAAQ,OAAO;GAEnB,QAAO,MADc,KAAK,MAAM,KAAK,MAAM,KAAK,EAAA,CAE7C,QAAQ,MAAM,YAAY,EAAE,KAAK,IAAI,SAAS,CAAC,CAC/C,KAAK,MAAM,eAAe,IAAI,MAAM,CAAC,CAAC;EAC3C;CACF;AACF;AAEA,SAAS,YAAY,GAAmB;CAEtC,OAAO,KAAK,IAAI,IAAI,KAAK;AAC3B;AAEA,SAAS,eAAe,YAAoB,MAAc,GAA+B;CACvF,MAAM,UAAU,YAAY,EAAE,KAAK;CACnC,MAAM,WACJ,UAAU,IAAI,aAAa,UAAU,IAAI,SAAS,UAAU,IAAI,WAAW;CAC7E,OAAO,YAAY;EACjB;EACA;EACA,SAAS,EAAE;EACX,OAAO,GAAG,EAAE,UAAU,GAAG,EAAE,UAAU,UAAU,QAAQ,QAAQ,CAAC,EAAE;EAClE,WAAW,EAAE;EACb;EACA,YAAY;EACZ,eAAe,EAAE,WACb,CAAC;GAAE,MAAM;GAAY,KAAK;GAAmB,SAAS,EAAE;EAAS,CAAC,IAClE,CAAC;EAGL,oBAAoB;EACpB,UAAU;GAAE,YAAY,EAAE;GAAW,WAAW,EAAE;GAAW,UAAU;EAAQ;CACjF,CAAC;AACH;AAaA,SAAgB,kCACd,OAAwC,CAAC,GACL;CACpC,MAAM,KAAK,KAAK,MAAM;CACtB,MAAM,OAAO,KAAK,QAAQ;CAC1B,MAAM,sBAAsB,+BAA+B,KAAK,mBAAmB;CACnF,OAAO;EACL;EACA,aACE;EACF,WAAW;EACX,MAAM;GACJ,MAAM;GACN,QAAQ,KAAK,SAAS,QAAQ,CAAC,KAAK,QAAQ,KAAK,IAAI,KAAA;GACrD,uBAAuB;EACzB;EACA,SAAS,GAAG,+BAA+B,WAAW;EACtD,MAAM,QAAQ,OAAO,KAAK;GACxB,MAAM,aAAa,IAAI,WAAW,IAAI,SAAS;GAC/C,IAAI;GACJ,IAAI;IACF,SAAS,MAAM,wBAAwB,OAAO;KAC5C,GAAG,KAAK;KACR;KACA,QAAQ,IAAI;IACd,CAAC;GACH,UAAU;IACR,MAAM,QAAQ,MAAM,iCAAiC,YAAY;KAC/D,SAAS;KACT,WAAW;IACb,CAAC;IACD,IAAI,CAAC,MAAM,SACT,IAAI,MAAM,wDAAwD;KAChE,eAAe,MAAM;KACrB,YAAY;IACd,CAAC;IAEH,IAAI,cAAc,MAAM,OAAO;GACjC;GACA,IAAI,CAAC,OAAO,WACV,OAAO,CACL,YAAY;IACV,YAAY;IACZ;IACA,OAAO;IACP,WAAW,OAAO;IAClB,UAAU;IACV,YAAY;IACZ,eAAe,CAAC;IAChB,UAAU,EAAE,QAAQ,OAAO,MAAM;GACnC,CAAC,CACH;GAEF,MAAM,MAAwB,CAAC;GAC/B,KAAK,MAAM,KAAK,OAAO,UAAU;IAG/B,IAAI,EAAE,WAAW,EAAE,SAAS,GAAG;IAC/B,IAAI,KACF,YAAY;KACV,YAAY;KACZ;KACA,SAAS,EAAE;KACX,OAAO,EAAE,UACL,YAAY,EAAE,QAAQ,aAAa,EAAE,MAAM,QAC3C,YAAY,EAAE,QAAQ;KAC1B,WAAW,EAAE;KACb,UAAU,aAAa,EAAE,QAAQ;KACjC,YAAY;KACZ,eAAe,CAAC;MAAE,MAAM;MAAY,KAAK;MAAmB,SAAS,EAAE;KAAS,CAAC;KACjF,UAAU;MACR,SAAS,EAAE;MACX,SAAS,EAAE;MACX,UAAU,EAAE;KACd;IACF,CAAC,CACH;GACF;GACA,OAAO;EACT;CACF;AACF;;;ACxVA,SAAgB,mBACd,SACuE;CACvE,IAAI,CAAC,QAAQ,GAAG,KAAK,GAAG,MAAM,IAAI,UAAU,0CAA0C;CACtF,IAAI,CAAC,QAAQ,YAAY,KAAK,GAC5B,MAAM,IAAI,UAAU,mDAAmD;CAEzE,IAAI,QAAQ,SAAS,KAAA,GACnB,MAAM,IAAI,UAAU,2CAA2C;CAEjE,IACE,qBAAqB,YACpB,CAAC,QAAQ,mBACR,OAAO,QAAQ,oBAAoB,YACnC,MAAM,QAAQ,QAAQ,eAAe,IAEvC,MAAM,IAAI,UAAU,uDAAuD;CAE7E,OAAO;EACL,IAAI,QAAQ;EACZ,aAAa,QAAQ;EACrB,SAAS,QAAQ,WAAW;EAC5B,WAAW;EACX,MAAM,QAAQ;EACd,SAAS,QAAQ;EACjB,GAAI,qBAAqB,UAAU,EAAE,iBAAiB,QAAQ,gBAAgB,IAAI,CAAC;CACrF;AACF"}
|
|
1
|
+
{"version":3,"file":"index.js","names":[],"sources":["../../src/analyst/adapters.ts"],"sourcesContent":["/**\n * Adapter factories — lift each existing agent-eval primitive into the\n * Analyst contract without re-implementing it.\n *\n * Five primitives, five factories. Each one:\n * - Builds an Analyst with a stable id (caller chooses; defaults\n * given), a sensible default `inputKind`, a version derived from\n * the wrapped primitive's version + an adapter revision, and an\n * `analyze()` that calls the primitive and lifts its output to\n * AnalystFinding[] using `makeFinding()`.\n * - Maps severities: the existing `Severity` ('critical' | 'major' |\n * 'minor' | 'info') projects onto AnalystSeverity ('critical' |\n * 'high' | 'medium' | 'low' | 'info'); 'major' → 'high', 'minor' →\n * 'medium'. Domain analysts that want finer-grained mapping override.\n *\n * Adapters never own state. Calling the same factory twice with the\n * same primitive instance is safe.\n */\n\nimport { CostLedger } from '../cost-ledger'\nimport type {\n Finding as LayerFinding,\n Severity as LayerSeverity,\n MultiLayerVerifier,\n VerifyOptions,\n} from '../multi-layer-verifier'\nimport { RunCritic, type RunTrace } from '../run-critic'\nimport {\n runSemanticConceptJudge,\n SEMANTIC_CONCEPT_JUDGE_VERSION,\n type SemanticConceptJudgeInput,\n type SemanticConceptJudgeOptions,\n type SemanticConceptJudgeResult,\n} from '../semantic-concept-judge'\nimport type { JudgeFn, JudgeInput, JudgeScore } from '../types'\nimport type { ChatClient } from './chat-client'\nimport type { Analyst, AnalystFinding, AnalystSeverity } from './types'\nimport { makeFinding } from './types'\nimport { settleUsageReceiptFromCostLedger, validateUsageSettlementTimeout } from './usage-receipt'\n\nconst ADAPTER_REV = '1'\n\n// ── Severity bridges ───────────────────────────────────────────────\n\nexport function liftSeverity(s: LayerSeverity): AnalystSeverity {\n switch (s) {\n case 'critical':\n return 'critical'\n case 'major':\n return 'high'\n case 'minor':\n return 'medium'\n case 'info':\n return 'info'\n }\n}\n\n// ── 1. MultiLayerVerifier → Analyst ─────────────────────────────────\n\nexport interface VerifierAdapterOpts<Env> {\n id?: string\n area?: string\n verifier: MultiLayerVerifier<Env>\n /**\n * The verifier expects an `env` per run. Adapters take it from\n * `AnalystRunInputs.custom[<id>]` via the registry's 'custom' routing.\n */\n options?: Omit<VerifyOptions<Env>, 'env'>\n}\n\nexport function createVerifierAdapter<Env>(opts: VerifierAdapterOpts<Env>): Analyst<Env> {\n const id = opts.id ?? 'multi-layer-verifier'\n const area = opts.area ?? 'verification'\n return {\n id,\n description:\n \"Runs a MultiLayerVerifier and lifts each layer's findings into the analyst envelope.\",\n inputKind: 'custom',\n cost: { kind: 'deterministic' },\n version: `verifier-${ADAPTER_REV}`,\n async analyze(env, ctx) {\n const report = await opts.verifier.run({ env, ...opts.options })\n const out: AnalystFinding[] = []\n for (const layer of report.layers) {\n for (const finding of layer.findings) {\n out.push(liftLayerFinding(id, area, layer.layer, finding))\n }\n // Layer-level signal: a failed/error layer is itself a finding\n // even if it didn't emit per-finding rows.\n if (layer.status === 'fail' || layer.status === 'error' || layer.status === 'timeout') {\n out.push(\n makeFinding({\n analyst_id: id,\n area,\n subject: layer.layer,\n claim: `layer \"${layer.layer}\" ${layer.status}: ${layer.reason ?? 'no reason given'}`,\n severity:\n layer.status === 'error' ? 'high' : layer.status === 'timeout' ? 'medium' : 'high',\n confidence: 1,\n evidence_refs: [],\n metadata: {\n layer_status: layer.status,\n duration_ms: layer.durationMs,\n score: layer.score,\n diagnostics: layer.diagnostics,\n },\n }),\n )\n }\n }\n ctx.log?.('verifier complete', {\n layers: report.layers.length,\n blended: report.blendedScore,\n all_pass: report.allPass,\n })\n return out\n },\n }\n}\n\nfunction liftLayerFinding(\n analyst_id: string,\n area: string,\n layer: string,\n f: LayerFinding,\n): AnalystFinding {\n return makeFinding({\n analyst_id,\n area,\n subject: f.layer ?? layer,\n claim: f.message,\n severity: liftSeverity(f.severity),\n confidence: 0.85,\n evidence_refs: f.evidence\n ? [{ kind: 'artifact', uri: 'inline:evidence', excerpt: f.evidence }]\n : [],\n metadata: f.detail,\n })\n}\n\n// ── 2. RunCritic → Analyst ──────────────────────────────────────────\n\nexport interface RunCriticAdapterOpts {\n id?: string\n area?: string\n critic?: RunCritic\n /** Optional threshold below which a dimension is reported as a finding. Default 0.5. */\n threshold?: number\n}\n\nexport function createRunCriticAdapter(opts: RunCriticAdapterOpts = {}): Analyst<RunTrace> {\n const id = opts.id ?? 'run-critic'\n const area = opts.area ?? 'run-quality'\n const critic = opts.critic ?? new RunCritic()\n const threshold = opts.threshold ?? 0.5\n return {\n id,\n description:\n 'Scores a single run across success / grounding / drift / tool-quality and surfaces below-threshold dimensions.',\n inputKind: 'custom',\n cost: { kind: 'deterministic' },\n version: `run-critic-${ADAPTER_REV}`,\n async analyze(trace) {\n const score = critic.scoreTrace(trace)\n const out: AnalystFinding[] = []\n const dims: Array<[keyof typeof score, AnalystSeverity, string]> = [\n ['success', 'critical', 'run did not complete successfully'],\n ['goalProgress', 'high', 'goal progress is low'],\n ['repoGroundedness', 'high', 'output is poorly grounded in the repository'],\n ['toolUseQuality', 'medium', 'tool use quality is low'],\n ['patchQuality', 'medium', 'no real patch/edit evidence'],\n ['testReality', 'high', 'no real test/build evidence'],\n ['finalGate', 'critical', 'final gate is blocking'],\n ]\n for (const [dim, sev, msg] of dims) {\n const value = score[dim] as number\n if (typeof value === 'number' && value < threshold) {\n out.push(\n makeFinding({\n analyst_id: id,\n area,\n subject: dim,\n claim: msg,\n rationale: `${dim}=${value.toFixed(2)} below threshold ${threshold}`,\n severity: sev,\n confidence: 1,\n evidence_refs: [],\n metadata: { dimension: dim, value, threshold, run_id: trace.run.runId },\n }),\n )\n }\n }\n // Drift penalty is high → surface as a finding (inverse threshold).\n if (score.driftPenalty > 1 - threshold) {\n out.push(\n makeFinding({\n analyst_id: id,\n area,\n subject: 'drift',\n claim: 'agent output drifted from repository signal',\n rationale: `driftPenalty=${score.driftPenalty.toFixed(2)}`,\n severity: 'medium',\n confidence: 0.9,\n evidence_refs: [],\n metadata: { drift_penalty: score.driftPenalty, notes: score.notes },\n }),\n )\n }\n return out\n },\n }\n}\n\n// ── 3. JudgeFn → Analyst ────────────────────────────────────────────\n\nexport interface JudgeAdapterOpts {\n id?: string\n area?: string\n judge: JudgeFn\n /** Chat client passed to the JudgeFn. */\n chat: ChatClient\n /** Optional cost classification — most judges call an LLM. */\n cost?: Analyst['cost']\n /** Optional threshold below which a JudgeScore becomes a finding. Default 6 (on 0-10 scale). */\n threshold?: number\n}\n\nexport function createJudgeAdapter(opts: JudgeAdapterOpts): Analyst<JudgeInput> {\n const id = opts.id ?? 'judge'\n const area = opts.area ?? 'judge'\n const threshold = opts.threshold ?? 6\n return {\n id,\n description:\n 'Wraps an agent-eval JudgeFn into an analyst; below-threshold dimensions surface as findings.',\n inputKind: 'judge-input',\n cost: opts.cost ?? { kind: 'llm' },\n version: `judge-${ADAPTER_REV}`,\n async analyze(input) {\n const scores = await opts.judge(opts.chat, input)\n return scores\n .filter((s) => normalize10(s.score) < threshold)\n .map((s) => liftJudgeScore(id, area, s))\n },\n }\n}\n\nfunction normalize10(s: number): number {\n // JudgeScore convention is 0-10 but some judges emit 0-1. Coerce to 0-10.\n return s <= 1 ? s * 10 : s\n}\n\nfunction liftJudgeScore(analyst_id: string, area: string, s: JudgeScore): AnalystFinding {\n const score10 = normalize10(s.score)\n const severity: AnalystSeverity =\n score10 < 3 ? 'critical' : score10 < 5 ? 'high' : score10 < 7 ? 'medium' : 'low'\n return makeFinding({\n analyst_id,\n area,\n subject: s.dimension,\n claim: `${s.judgeName}/${s.dimension} scored ${score10.toFixed(1)}/10`,\n rationale: s.reasoning,\n severity,\n confidence: 0.8,\n evidence_refs: s.evidence\n ? [{ kind: 'artifact', uri: 'inline:evidence', excerpt: s.evidence }]\n : [],\n // Descriptive origin only. A caller may admit search feedback into candidate\n // generation, but final evaluation findings have no allowed proposal origin.\n derived_from_judge: true,\n metadata: { judge_name: s.judgeName, dimension: s.dimension, score_10: score10 },\n })\n}\n\n// ── 4. SemanticConceptJudge → Analyst ──────────────────────────────\n\nexport interface SemanticConceptJudgeAdapterOpts {\n id?: string\n area?: string\n /** Registry context owns cancellation and the per-analyst cost ledger. */\n options?: Omit<SemanticConceptJudgeOptions, 'costLedger' | 'signal'>\n /** Maximum post-cancellation wait for a provider receipt. Default 5 seconds. */\n settlementTimeoutMs?: number\n}\n\nexport function createSemanticConceptJudgeAdapter(\n opts: SemanticConceptJudgeAdapterOpts = {},\n): Analyst<SemanticConceptJudgeInput> {\n const id = opts.id ?? 'semantic-concept-judge'\n const area = opts.area ?? 'concept-coverage'\n const settlementTimeoutMs = validateUsageSettlementTimeout(opts.settlementTimeoutMs)\n return {\n id,\n description:\n 'Runs the semantic-concept judge and surfaces missing / weak concepts as findings.',\n inputKind: 'custom',\n cost: {\n kind: 'llm',\n models: opts.options?.model ? [opts.options.model] : undefined,\n settlement_timeout_ms: settlementTimeoutMs,\n },\n version: `${SEMANTIC_CONCEPT_JUDGE_VERSION}-adapter-${ADAPTER_REV}`,\n async analyze(input, ctx) {\n const costLedger = new CostLedger(ctx.budgetUsd)\n let result: SemanticConceptJudgeResult\n try {\n result = await runSemanticConceptJudge(input, {\n ...opts.options,\n costLedger,\n signal: ctx.signal,\n })\n } finally {\n const usage = await settleUsageReceiptFromCostLedger(costLedger, {\n channel: 'judge',\n timeoutMs: settlementTimeoutMs,\n })\n if (!usage.settled) {\n ctx.log?.('semantic-concept judge provider settlement timed out', {\n pending_calls: usage.pendingCalls,\n timeout_ms: settlementTimeoutMs,\n })\n }\n ctx.recordUsage?.(usage.receipt)\n }\n if (!result.available) {\n return [\n makeFinding({\n analyst_id: id,\n area,\n claim: 'semantic-concept judge unavailable',\n rationale: result.error,\n severity: 'info',\n confidence: 1,\n evidence_refs: [],\n metadata: { reason: result.error },\n }),\n ]\n }\n const out: AnalystFinding[] = []\n for (const f of result.findings) {\n // Only surface gaps: missing concepts or low scores. Concepts at\n // 7+/10 with present=true are not findings — they're successes.\n if (f.present && f.score >= 7) continue\n out.push(\n makeFinding({\n analyst_id: id,\n area,\n subject: f.concept,\n claim: f.present\n ? `concept \"${f.concept}\" is weak (${f.score}/10)`\n : `concept \"${f.concept}\" is missing`,\n rationale: f.evidence,\n severity: liftSeverity(f.severity),\n confidence: 0.85,\n evidence_refs: [{ kind: 'artifact', uri: 'inline:evidence', excerpt: f.evidence }],\n metadata: {\n concept: f.concept,\n present: f.present,\n score_10: f.score,\n },\n }),\n )\n }\n return out\n },\n }\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;;AAwCA,MAAM,cAAc;AAIpB,SAAgB,aAAa,GAAmC;CAC9D,QAAQ,GAAR;EACE,KAAK,YACH,OAAO;EACT,KAAK,SACH,OAAO;EACT,KAAK,SACH,OAAO;EACT,KAAK,QACH,OAAO;CACX;AACF;AAeA,SAAgB,sBAA2B,MAA8C;CACvF,MAAM,KAAK,KAAK,MAAM;CACtB,MAAM,OAAO,KAAK,QAAQ;CAC1B,OAAO;EACL;EACA,aACE;EACF,WAAW;EACX,MAAM,EAAE,MAAM,gBAAgB;EAC9B,SAAS,YAAY;EACrB,MAAM,QAAQ,KAAK,KAAK;GACtB,MAAM,SAAS,MAAM,KAAK,SAAS,IAAI;IAAE;IAAK,GAAG,KAAK;GAAQ,CAAC;GAC/D,MAAM,MAAwB,CAAC;GAC/B,KAAK,MAAM,SAAS,OAAO,QAAQ;IACjC,KAAK,MAAM,WAAW,MAAM,UAC1B,IAAI,KAAK,iBAAiB,IAAI,MAAM,MAAM,OAAO,OAAO,CAAC;IAI3D,IAAI,MAAM,WAAW,UAAU,MAAM,WAAW,WAAW,MAAM,WAAW,WAC1E,IAAI,KACF,YAAY;KACV,YAAY;KACZ;KACA,SAAS,MAAM;KACf,OAAO,UAAU,MAAM,MAAM,IAAI,MAAM,OAAO,IAAI,MAAM,UAAU;KAClE,UACE,MAAM,WAAW,UAAU,SAAS,MAAM,WAAW,YAAY,WAAW;KAC9E,YAAY;KACZ,eAAe,CAAC;KAChB,UAAU;MACR,cAAc,MAAM;MACpB,aAAa,MAAM;MACnB,OAAO,MAAM;MACb,aAAa,MAAM;KACrB;IACF,CAAC,CACH;GAEJ;GACA,IAAI,MAAM,qBAAqB;IAC7B,QAAQ,OAAO,OAAO;IACtB,SAAS,OAAO;IAChB,UAAU,OAAO;GACnB,CAAC;GACD,OAAO;EACT;CACF;AACF;AAEA,SAAS,iBACP,YACA,MACA,OACA,GACgB;CAChB,OAAO,YAAY;EACjB;EACA;EACA,SAAS,EAAE,SAAS;EACpB,OAAO,EAAE;EACT,UAAU,aAAa,EAAE,QAAQ;EACjC,YAAY;EACZ,eAAe,EAAE,WACb,CAAC;GAAE,MAAM;GAAY,KAAK;GAAmB,SAAS,EAAE;EAAS,CAAC,IAClE,CAAC;EACL,UAAU,EAAE;CACd,CAAC;AACH;AAYA,SAAgB,uBAAuB,OAA6B,CAAC,GAAsB;CACzF,MAAM,KAAK,KAAK,MAAM;CACtB,MAAM,OAAO,KAAK,QAAQ;CAC1B,MAAM,SAAS,KAAK,UAAU,IAAI,UAAU;CAC5C,MAAM,YAAY,KAAK,aAAa;CACpC,OAAO;EACL;EACA,aACE;EACF,WAAW;EACX,MAAM,EAAE,MAAM,gBAAgB;EAC9B,SAAS,cAAc;EACvB,MAAM,QAAQ,OAAO;GACnB,MAAM,QAAQ,OAAO,WAAW,KAAK;GACrC,MAAM,MAAwB,CAAC;GAU/B,KAAK,MAAM,CAAC,KAAK,KAAK,QAAQ;IAR5B;KAAC;KAAW;KAAY;IAAmC;IAC3D;KAAC;KAAgB;KAAQ;IAAsB;IAC/C;KAAC;KAAoB;KAAQ;IAA6C;IAC1E;KAAC;KAAkB;KAAU;IAAyB;IACtD;KAAC;KAAgB;KAAU;IAA6B;IACxD;KAAC;KAAe;KAAQ;IAA6B;IACrD;KAAC;KAAa;KAAY;IAAwB;GAEnB,GAAG;IAClC,MAAM,QAAQ,MAAM;IACpB,IAAI,OAAO,UAAU,YAAY,QAAQ,WACvC,IAAI,KACF,YAAY;KACV,YAAY;KACZ;KACA,SAAS;KACT,OAAO;KACP,WAAW,GAAG,IAAI,GAAG,MAAM,QAAQ,CAAC,EAAE,mBAAmB;KACzD,UAAU;KACV,YAAY;KACZ,eAAe,CAAC;KAChB,UAAU;MAAE,WAAW;MAAK;MAAO;MAAW,QAAQ,MAAM,IAAI;KAAM;IACxE,CAAC,CACH;GAEJ;GAEA,IAAI,MAAM,eAAe,IAAI,WAC3B,IAAI,KACF,YAAY;IACV,YAAY;IACZ;IACA,SAAS;IACT,OAAO;IACP,WAAW,gBAAgB,MAAM,aAAa,QAAQ,CAAC;IACvD,UAAU;IACV,YAAY;IACZ,eAAe,CAAC;IAChB,UAAU;KAAE,eAAe,MAAM;KAAc,OAAO,MAAM;IAAM;GACpE,CAAC,CACH;GAEF,OAAO;EACT;CACF;AACF;AAgBA,SAAgB,mBAAmB,MAA6C;CAC9E,MAAM,KAAK,KAAK,MAAM;CACtB,MAAM,OAAO,KAAK,QAAQ;CAC1B,MAAM,YAAY,KAAK,aAAa;CACpC,OAAO;EACL;EACA,aACE;EACF,WAAW;EACX,MAAM,KAAK,QAAQ,EAAE,MAAM,MAAM;EACjC,SAAS,SAAS;EAClB,MAAM,QAAQ,OAAO;GAEnB,QAAO,MADc,KAAK,MAAM,KAAK,MAAM,KAAK,EAAA,CAE7C,QAAQ,MAAM,YAAY,EAAE,KAAK,IAAI,SAAS,CAAC,CAC/C,KAAK,MAAM,eAAe,IAAI,MAAM,CAAC,CAAC;EAC3C;CACF;AACF;AAEA,SAAS,YAAY,GAAmB;CAEtC,OAAO,KAAK,IAAI,IAAI,KAAK;AAC3B;AAEA,SAAS,eAAe,YAAoB,MAAc,GAA+B;CACvF,MAAM,UAAU,YAAY,EAAE,KAAK;CACnC,MAAM,WACJ,UAAU,IAAI,aAAa,UAAU,IAAI,SAAS,UAAU,IAAI,WAAW;CAC7E,OAAO,YAAY;EACjB;EACA;EACA,SAAS,EAAE;EACX,OAAO,GAAG,EAAE,UAAU,GAAG,EAAE,UAAU,UAAU,QAAQ,QAAQ,CAAC,EAAE;EAClE,WAAW,EAAE;EACb;EACA,YAAY;EACZ,eAAe,EAAE,WACb,CAAC;GAAE,MAAM;GAAY,KAAK;GAAmB,SAAS,EAAE;EAAS,CAAC,IAClE,CAAC;EAGL,oBAAoB;EACpB,UAAU;GAAE,YAAY,EAAE;GAAW,WAAW,EAAE;GAAW,UAAU;EAAQ;CACjF,CAAC;AACH;AAaA,SAAgB,kCACd,OAAwC,CAAC,GACL;CACpC,MAAM,KAAK,KAAK,MAAM;CACtB,MAAM,OAAO,KAAK,QAAQ;CAC1B,MAAM,sBAAsB,+BAA+B,KAAK,mBAAmB;CACnF,OAAO;EACL;EACA,aACE;EACF,WAAW;EACX,MAAM;GACJ,MAAM;GACN,QAAQ,KAAK,SAAS,QAAQ,CAAC,KAAK,QAAQ,KAAK,IAAI,KAAA;GACrD,uBAAuB;EACzB;EACA,SAAS,GAAG,+BAA+B,WAAW;EACtD,MAAM,QAAQ,OAAO,KAAK;GACxB,MAAM,aAAa,IAAI,WAAW,IAAI,SAAS;GAC/C,IAAI;GACJ,IAAI;IACF,SAAS,MAAM,wBAAwB,OAAO;KAC5C,GAAG,KAAK;KACR;KACA,QAAQ,IAAI;IACd,CAAC;GACH,UAAU;IACR,MAAM,QAAQ,MAAM,iCAAiC,YAAY;KAC/D,SAAS;KACT,WAAW;IACb,CAAC;IACD,IAAI,CAAC,MAAM,SACT,IAAI,MAAM,wDAAwD;KAChE,eAAe,MAAM;KACrB,YAAY;IACd,CAAC;IAEH,IAAI,cAAc,MAAM,OAAO;GACjC;GACA,IAAI,CAAC,OAAO,WACV,OAAO,CACL,YAAY;IACV,YAAY;IACZ;IACA,OAAO;IACP,WAAW,OAAO;IAClB,UAAU;IACV,YAAY;IACZ,eAAe,CAAC;IAChB,UAAU,EAAE,QAAQ,OAAO,MAAM;GACnC,CAAC,CACH;GAEF,MAAM,MAAwB,CAAC;GAC/B,KAAK,MAAM,KAAK,OAAO,UAAU;IAG/B,IAAI,EAAE,WAAW,EAAE,SAAS,GAAG;IAC/B,IAAI,KACF,YAAY;KACV,YAAY;KACZ;KACA,SAAS,EAAE;KACX,OAAO,EAAE,UACL,YAAY,EAAE,QAAQ,aAAa,EAAE,MAAM,QAC3C,YAAY,EAAE,QAAQ;KAC1B,WAAW,EAAE;KACb,UAAU,aAAa,EAAE,QAAQ;KACjC,YAAY;KACZ,eAAe,CAAC;MAAE,MAAM;MAAY,KAAK;MAAmB,SAAS,EAAE;KAAS,CAAC;KACjF,UAAU;MACR,SAAS,EAAE;MACX,SAAS,EAAE;MACX,UAAU,EAAE;KACd;IACF,CAAC,CACH;GACF;GACA,OAAO;EACT;CACF;AACF"}
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import {
|
|
1
|
+
import { t as assertValidAnalystUsageReceipt } from "./usage-receipt-CgxMEBZq.js";
|
|
2
2
|
import { performance } from "node:perf_hooks";
|
|
3
3
|
import { linearSumAssignment } from "linear-sum-assignment";
|
|
4
4
|
//#region src/analyst/benchmark-scoring.ts
|
|
@@ -551,4 +551,4 @@ function mergeRegistryUsage(result) {
|
|
|
551
551
|
//#endregion
|
|
552
552
|
export { scoreAnalystFindings as a, summarizeAnalystBenchmarkRunner as i, runAnalystBenchmark as n, traceStoreEvidenceResolver as r, registryBenchmarkRunner as t };
|
|
553
553
|
|
|
554
|
-
//# sourceMappingURL=benchmark-
|
|
554
|
+
//# sourceMappingURL=benchmark-CYtcIF2V.js.map
|