@tangle-network/agent-eval 0.133.3 → 0.134.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +62 -0
- package/dist/analyst/index.d.ts +11 -35
- package/dist/analyst/index.d.ts.map +1 -1
- package/dist/analyst/index.js +4 -53
- package/dist/analyst/index.js.map +1 -1
- package/dist/{analyze-runs-BClW9OSe.d.ts → analyze-runs-DMo3Lb_y.d.ts} +4 -4
- package/dist/{analyze-runs-BClW9OSe.d.ts.map → analyze-runs-DMo3Lb_y.d.ts.map} +1 -1
- package/dist/benchmarks/index.d.ts +1 -1
- package/dist/benchmarks/index.js +1 -1
- package/dist/{benchmarks-BP9sgMia.js → benchmarks-v5piCeDl.js} +3 -3
- package/dist/{benchmarks-BP9sgMia.js.map → benchmarks-v5piCeDl.js.map} +1 -1
- package/dist/campaign/index.d.ts +5 -4
- package/dist/campaign/index.js +4 -3
- package/dist/{campaign--V4ffEKR.js → campaign-DEC_7DLn.js} +2 -2
- package/dist/{campaign--V4ffEKR.js.map → campaign-DEC_7DLn.js.map} +1 -1
- package/dist/{client-Du7B81wW.d.ts → client-BIyh1RCr.d.ts} +3 -3
- package/dist/{client-Du7B81wW.d.ts.map → client-BIyh1RCr.d.ts.map} +1 -1
- package/dist/contract/index.d.ts +12 -11
- package/dist/contract/index.d.ts.map +1 -1
- package/dist/contract/index.js +4 -11
- package/dist/contract/index.js.map +1 -1
- package/dist/{default-registry-Cl3pHo4n.d.ts → default-registry-Brxr728w.d.ts} +4 -268
- package/dist/default-registry-Brxr728w.d.ts.map +1 -0
- package/dist/{default-registry-D3T9XbuY.js → default-registry-IjYs7T8l.js} +4 -61
- package/dist/default-registry-IjYs7T8l.js.map +1 -0
- package/dist/hosted/index.d.ts +2 -2
- package/dist/{index-DOqvIJ8I.d.ts → index-BoJNQR6n.d.ts} +4 -3
- package/dist/index-BoJNQR6n.d.ts.map +1 -0
- package/dist/{index-B5MNN1f1.d.ts → index-C21xKtxu.d.ts} +4 -4
- package/dist/{index-B5MNN1f1.d.ts.map → index-C21xKtxu.d.ts.map} +1 -1
- package/dist/index.d.ts +12 -11
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +109 -11
- package/dist/index.js.map +1 -1
- package/dist/multishot/index.d.ts +1 -1
- package/dist/openapi.json +1 -1
- package/dist/proposal-findings-DCawte-y.js +164 -0
- package/dist/proposal-findings-DCawte-y.js.map +1 -0
- package/dist/{release-report-DKBtegGt.d.ts → release-report-CuULWKyk.d.ts} +2 -2
- package/dist/{release-report-DKBtegGt.d.ts.map → release-report-CuULWKyk.d.ts.map} +1 -1
- package/dist/reporting.d.ts +2 -2
- package/dist/{researcher-BtD5U1Up.d.ts → researcher-DVtruQ9U.d.ts} +2 -2
- package/dist/{researcher-BtD5U1Up.d.ts.map → researcher-DVtruQ9U.d.ts.map} +1 -1
- package/dist/rl.d.ts +2 -2
- package/dist/rl.d.ts.map +1 -1
- package/dist/rl.js +14 -1
- package/dist/rl.js.map +1 -1
- package/dist/{semantic-concept-judge-BypLt6Fw.js → semantic-concept-judge-C0P1VTXD.js} +2 -3
- package/dist/{semantic-concept-judge-BypLt6Fw.js.map → semantic-concept-judge-C0P1VTXD.js.map} +1 -1
- package/dist/{skill-usage-BaaxFSJR.d.ts → skill-usage-BDQVPIG1.d.ts} +3 -2
- package/dist/skill-usage-BDQVPIG1.d.ts.map +1 -0
- package/dist/{skillopt-optimization-method-vvJ4bMNI.js → skillopt-optimization-method-BY6vKLJB.js} +51 -30
- package/dist/skillopt-optimization-method-BY6vKLJB.js.map +1 -0
- package/dist/{skillopt-optimization-method-Dxr8pdZd.d.ts → skillopt-optimization-method-DJ3l4w8W.d.ts} +10 -13
- package/dist/skillopt-optimization-method-DJ3l4w8W.d.ts.map +1 -0
- package/dist/{summary-report-DyOhItws.d.ts → summary-report-DGp0-_XO.d.ts} +59 -3
- package/dist/summary-report-DGp0-_XO.d.ts.map +1 -0
- package/dist/types-DVjczBM9.d.ts +276 -0
- package/dist/types-DVjczBM9.d.ts.map +1 -0
- package/dist/{types-BokuXvOG.d.ts → types-DiWLru6Z.d.ts} +20 -37
- package/dist/types-DiWLru6Z.d.ts.map +1 -0
- package/docs/campaign-proposers.md +5 -0
- package/package.json +1 -1
- package/dist/default-registry-Cl3pHo4n.d.ts.map +0 -1
- package/dist/default-registry-D3T9XbuY.js.map +0 -1
- package/dist/index-DOqvIJ8I.d.ts.map +0 -1
- package/dist/run-score-iEEAWiBY.js +0 -41
- package/dist/run-score-iEEAWiBY.js.map +0 -1
- package/dist/skill-usage-BaaxFSJR.d.ts.map +0 -1
- package/dist/skillopt-optimization-method-Dxr8pdZd.d.ts.map +0 -1
- package/dist/skillopt-optimization-method-vvJ4bMNI.js.map +0 -1
- package/dist/summary-report-DyOhItws.d.ts.map +0 -1
- package/dist/types-BokuXvOG.d.ts.map +0 -1
package/CHANGELOG.md
CHANGED
|
@@ -4,6 +4,16 @@ All notable changes to `@tangle-network/agent-eval` and its sibling `agent-eval-
|
|
|
4
4
|
|
|
5
5
|
---
|
|
6
6
|
|
|
7
|
+
## [0.134.0] - 2026-07-28 - isolated proposal inputs
|
|
8
|
+
|
|
9
|
+
### Changed
|
|
10
|
+
|
|
11
|
+
- `runOptimization()` now accepts only findings labeled from search runs or observed production behavior.
|
|
12
|
+
- `ProposalFinding` carries the required `proposal_origin`.
|
|
13
|
+
- Removed opaque `report` and capture-store access from `ProposeContext`; final evaluation data has no internal path into candidate generation.
|
|
14
|
+
- Removed `assertNoJudgeVerdict`, `isJudgeVerdict`, and `isTraceObservable`; use validated `ProposalFinding` inputs instead.
|
|
15
|
+
- `runOptimization()` snapshots its baseline and candidate outputs before measurement.
|
|
16
|
+
|
|
7
17
|
## [0.133.3] - 2026-07-27 - trustworthy statistical decisions
|
|
8
18
|
|
|
9
19
|
### Consumer notice — reported p-values were too small in every release from 0.1.0 to 0.133.0
|
|
@@ -53,8 +63,60 @@ Three further cautions, independent of the CDF:
|
|
|
53
63
|
Full evidence, per-statistic verdicts, and the dependency argument:
|
|
54
64
|
[`docs/design/statistics-decisions.md`](./docs/design/statistics-decisions.md).
|
|
55
65
|
|
|
66
|
+
### Consumer notice — `HeldOutGate` promoted candidates that scored nothing on most held-out items
|
|
67
|
+
|
|
68
|
+
Every release through 0.133.2 decided a promotion over the items where BOTH arms
|
|
69
|
+
produced a finite score, and dropped the rest without a word. An item the candidate
|
|
70
|
+
crashed on, timed out on, or wrote no row for simply left the comparison.
|
|
71
|
+
|
|
72
|
+
**Measured, deterministic fixture, no model calls:** 26 held-out items, a candidate
|
|
73
|
+
that produced no score at all on 20 of them and 0.95 on the 6 it answered against a
|
|
74
|
+
0.60 baseline. Published 0.133.2 and `origin/main` PROMOTE it at every threshold
|
|
75
|
+
from −0.05 to +0.30 — `productiveRuns: 6`, `unpairedBaselineRuns: 20` sitting in the
|
|
76
|
+
evidence, read by nothing. The control, the same 20 failures scored as the 0 they
|
|
77
|
+
earned, is correctly refused at a mean paired delta of −0.3808. An agent that failed
|
|
78
|
+
77 % of its tasks was promoted, and the paired-delta, overfit-gap and cost gates all
|
|
79
|
+
sat behind that filter.
|
|
80
|
+
|
|
81
|
+
A second shape, same cause: a crashed first attempt plus a scored retry at the same
|
|
82
|
+
`(experimentId, scenarioId, seed)` PROMOTED on 0.133.2, even though the gate's own
|
|
83
|
+
docstring says duplicate identities throw — the crashed row was filtered out before
|
|
84
|
+
the duplicate could be seen. It now throws, exactly as two scored rows already did.
|
|
85
|
+
|
|
86
|
+
**How to re-check a decision you already made.** Read `unpairedBaselineRuns` and
|
|
87
|
+
`unpairedCandidateRuns` on any recorded `GateDecision`: a nonzero value means the
|
|
88
|
+
verdict was computed over a subset. `productiveRuns` below the number of held-out
|
|
89
|
+
items you dealt means the same thing. Those promotions are not valid at the stated
|
|
90
|
+
threshold and should be re-run on this release.
|
|
91
|
+
|
|
56
92
|
### Fixed
|
|
57
93
|
|
|
94
|
+
- `HeldOutGate`'s cost median is taken over the rows that DECIDED the verdict — the
|
|
95
|
+
matched pairs on both splits — instead of over every row the caller passed. The old
|
|
96
|
+
population was a denominator nobody measured and was trivially movable: 48 rows tagged
|
|
97
|
+
`dev` at \$0.0001, which the gate never scores, drag a real \$5.00/task candidate to a
|
|
98
|
+
reported \$0.0001 and clear a \$1.00 `costPerTaskCeiling`. Measured on `origin/main`
|
|
99
|
+
(2789970): 12 fully-covered items at \$5.00/task, ceiling \$1.00 — 0 pad rows rejects
|
|
100
|
+
with `cost_ceiling`, 24 pad rows reports \$2.50005, 48 pad rows reports \$0.0001 and
|
|
101
|
+
PROMOTES. The population is derived from the pairing rather than from a list of split
|
|
102
|
+
tags, so there is no tag that sits outside the rule. On a comparison with no rows
|
|
103
|
+
outside the two decided splits the reported number is unchanged.
|
|
104
|
+
- `HeldOutGate` requires COVERAGE before it decides anything: on both the search and
|
|
105
|
+
the holdout split, `answered / dealt` must be at least the new `minCoverage`
|
|
106
|
+
(default **1** — every item the comparison was dealt carries a real score on both
|
|
107
|
+
arms), else it refuses with the new `incomplete_coverage` rejection code. The
|
|
108
|
+
denominator is measured, not declared: it is what `pairRunRecords` reports when
|
|
109
|
+
given every row of a split rather than only the scored ones, so an item counts as
|
|
110
|
+
dealt because a row for it exists on at least one arm. The gate does not impute a
|
|
111
|
+
value for a missing score — it does not know the failure value of the caller's
|
|
112
|
+
metric, and a caller who does knows to write it onto the record before calling.
|
|
113
|
+
`GateEvidence` gains `holdoutCoverage` and `searchCoverage` (`SplitCoverage`:
|
|
114
|
+
`dealt`, `answered`, `unscoredPairs`, `candidateOnly`, `baselineOnly`, `coverage`),
|
|
115
|
+
so a shrunken denominator can never be read without seeing it.
|
|
116
|
+
Verified monotone against `origin/main` over 6000 randomised comparisons: on
|
|
117
|
+
complete inputs the verdict, rejection code, CI and n are identical in 3000/3000
|
|
118
|
+
cases; across both sweeps there are 0 cases where the new gate promotes something
|
|
119
|
+
the old one refused.
|
|
58
120
|
- `regularizedIncompleteBeta` takes the mandatory symmetry branch `I_x(a,b) = 1 − I_{1−x}(b,a)`.
|
|
59
121
|
`studentTCdf(0.005, 100)` returned `0.89152130` against a true `0.50198972`; a
|
|
60
122
|
perfectly null paired result reported `p < 0.05`. This survived 0.133.1, which
|
package/dist/analyst/index.d.ts
CHANGED
|
@@ -1,10 +1,11 @@
|
|
|
1
1
|
import { a as MultiLayerVerifier, l as VerifyOptions, o as Severity } from "../multi-layer-verifier-BHY1gWAc.js";
|
|
2
|
-
import { C as FindingSubjectKind, D as parseFindingSubject, E as findingSubjectGrammarPromptFor, F as createAnalystAi, H as SemanticConceptJudgeInput, O as renderFindingSubject, P as CreateAnalystAiConfig, S as FindingSubject, T as KIND_EXPECTED_SUBJECTS, U as SemanticConceptJudgeOptions, Y as RunTrace, _ as defaultIsMaterial, a as SkillUsageScanConfig, b as FINDING_SUBJECT_KINDS, c as DEFAULT_TRACE_ANALYST_KINDS, d as IMPROVEMENT_KIND_SPEC, f as FAILURE_MODE_KIND_SPEC, g as PersistedFinding, h as FindingsStore, i as SkillUsageReport, k as BehavioralMetrics, l as KNOWLEDGE_POISONING_KIND_SPEC, m as FindingsDiff, n as SkillUsageAnalyst, o as buildSkillUsageReport, p as DiffPolicy, q as RunCritic, r as SkillUsageRecord, s as emitSkillUsageFindings, t as SKILL_USAGE_ANALYST, u as KNOWLEDGE_GAP_KIND_SPEC, v as diffFindings, w as FindingSubjectStringSchema, x as FINDING_SUBJECT_SYNTAX, y as FINDING_SUBJECT_GRAMMAR_PROMPT } from "../skill-usage-
|
|
2
|
+
import { C as FindingSubjectKind, D as parseFindingSubject, E as findingSubjectGrammarPromptFor, F as createAnalystAi, H as SemanticConceptJudgeInput, O as renderFindingSubject, P as CreateAnalystAiConfig, S as FindingSubject, T as KIND_EXPECTED_SUBJECTS, U as SemanticConceptJudgeOptions, Y as RunTrace, _ as defaultIsMaterial, a as SkillUsageScanConfig, b as FINDING_SUBJECT_KINDS, c as DEFAULT_TRACE_ANALYST_KINDS, d as IMPROVEMENT_KIND_SPEC, f as FAILURE_MODE_KIND_SPEC, g as PersistedFinding, h as FindingsStore, i as SkillUsageReport, k as BehavioralMetrics, l as KNOWLEDGE_POISONING_KIND_SPEC, m as FindingsDiff, n as SkillUsageAnalyst, o as buildSkillUsageReport, p as DiffPolicy, q as RunCritic, r as SkillUsageRecord, s as emitSkillUsageFindings, t as SKILL_USAGE_ANALYST, u as KNOWLEDGE_GAP_KIND_SPEC, v as diffFindings, w as FindingSubjectStringSchema, x as FINDING_SUBJECT_SYNTAX, y as FINDING_SUBJECT_GRAMMAR_PROMPT } from "../skill-usage-BDQVPIG1.js";
|
|
3
3
|
import { c as CostLedgerHandle } from "../cost-ledger-fGS_u_O1.js";
|
|
4
4
|
import { o as LlmClientOptions } from "../llm-client-BiK4HW0u.js";
|
|
5
5
|
import { A as ChatClient, B as SandboxSdkTransportOpts, F as CreateChatClientOpts, I as CustomTransportOpts, L as DirectProviderTransportOpts, M as ChatResponse, N as ChatTransport, P as CliBridgeTransportOpts, R as MockTransportOpts, V as createChatClient, j as ChatRequest, k as ChatCallOpts, m as JudgeInput, p as JudgeFn, z as RouterTransportOpts } from "../types-Cc3qbqzj.js";
|
|
6
6
|
import { t as TraceAnalysisStore } from "../store-CxJry_cs.js";
|
|
7
|
-
import {
|
|
7
|
+
import { _ as makeFinding, a as AnalystInputKind, c as AnalystRunInputs, d as AnalystSeverity, f as AnalystUsageReceipt, g as computeFindingId, h as ProposalFindingOrigin, i as AnalystFinding, l as AnalystRunResult, m as ProposalFinding, n as AnalystContext, o as AnalystRequirements, p as EvidenceRef, r as AnalystCost, s as AnalystRunEvent, t as Analyst, u as AnalystRunSummary, v as makeProposalFinding } from "../types-DVjczBM9.js";
|
|
8
|
+
import { _ as RawAnalystEvidenceSchema, a as AnalystRegistryOptions, b as evidenceRefsFromRawFinding, c as CreateTraceAnalystKindOpts, d as createTraceAnalystKind, f as renderPriorFindings, g as RawAnalystEvidence, h as RAW_FINDING_SCHEMA_PROMPT, i as AnalystRegistry, l as TraceAnalystGolden, m as ANALYST_SEVERITIES, n as buildDefaultAnalystRegistry, o as BudgetPolicy, p as renderUpstreamFindings, r as AnalystHooks, s as RegistryRunOpts, t as DefaultAnalystRegistryOptions, u as TraceAnalystKindSpec, v as RawAnalystFinding, x as parseRawFinding, y as RawAnalystFindingSchema } from "../default-registry-Brxr728w.js";
|
|
8
9
|
import { AxFunction } from "@ax-llm/ax";
|
|
9
10
|
//#region src/analyst/adapters.d.ts
|
|
10
11
|
declare function liftSeverity(s: Severity): AnalystSeverity;
|
|
@@ -88,40 +89,15 @@ declare function coerceJson(text: string): unknown;
|
|
|
88
89
|
*/
|
|
89
90
|
declare function coerceToFindingRows(raw: unknown): unknown[];
|
|
90
91
|
//#endregion
|
|
91
|
-
//#region src/analyst/
|
|
92
|
-
/**
|
|
93
|
-
|
|
94
|
-
* rendering — it is NOT the steer gate. Evidence presence is the WRONG
|
|
95
|
-
* discriminator for steering: a legitimate trace-analyst observation may cite
|
|
96
|
-
* nothing (it would be wrongly rejected), and a judge verdict may cite an
|
|
97
|
-
* artifact (it would be wrongly admitted). Use `assertNoJudgeVerdict` to gate
|
|
98
|
-
* steering; use this only where "is this grounded in observable evidence" is the
|
|
99
|
-
* literal question. */
|
|
100
|
-
declare function isTraceObservable(finding: AnalystFinding): boolean;
|
|
101
|
-
/** True iff the finding is a JUDGE VERDICT (an acceptance score lifted into a
|
|
102
|
-
* finding), identified by provenance set at the lift site — independent of
|
|
103
|
-
* whatever evidence it cites. */
|
|
104
|
-
declare function isJudgeVerdict(finding: AnalystFinding): boolean;
|
|
92
|
+
//#region src/analyst/proposal-findings.d.ts
|
|
93
|
+
/** True when a finding names a source candidate generation may learn from. */
|
|
94
|
+
declare function isProposalFinding(finding: unknown): finding is ProposalFinding;
|
|
105
95
|
/**
|
|
106
|
-
*
|
|
107
|
-
*
|
|
108
|
-
*
|
|
109
|
-
* loop. Returns the findings unchanged for chaining.
|
|
110
|
-
*
|
|
111
|
-
* Call this at the chokepoint where a detector that ALSO scores/gates has its
|
|
112
|
-
* findings turned into a steer (the judge-and-steer dual-role case). It keys on
|
|
113
|
-
* provenance, so it correctly admits evidence-less trace-analyst observations and
|
|
114
|
-
* correctly rejects an artifact-citing judge verdict — the cases an evidence
|
|
115
|
-
* check gets backwards.
|
|
116
|
-
*
|
|
117
|
-
* It is necessary, not sufficient: it stops PROVENANCE-tagged verdicts. A judge
|
|
118
|
-
* whose output is laundered through a hand-built finding with no provenance flag
|
|
119
|
-
* is out of its reach — provenance must be honestly set at every judge→finding
|
|
120
|
-
* lift (today: createJudgeAdapter). That is why the integrity rule lives at the
|
|
121
|
-
* lift site, and why ProposeContext.judgeScores?: never is the complementary
|
|
122
|
-
* compile-time tripwire on the obvious direct channel.
|
|
96
|
+
* Reject findings whose source has not been explicitly admitted for candidate
|
|
97
|
+
* generation. Search feedback and observed production behavior are allowed;
|
|
98
|
+
* final evaluation data has no allowed origin.
|
|
123
99
|
*/
|
|
124
|
-
declare function
|
|
100
|
+
declare function assertProposalFindings(findings: unknown, context?: string): ReadonlyArray<ProposalFinding>;
|
|
125
101
|
//#endregion
|
|
126
102
|
//#region src/analyst/structure-findings.d.ts
|
|
127
103
|
interface StructureFindingsOptions {
|
|
@@ -176,5 +152,5 @@ type TraceToolGroupName =
|
|
|
176
152
|
*/
|
|
177
153
|
declare function buildTraceToolsForGroup(group: TraceToolGroupName, store: TraceAnalysisStore): AxFunction[];
|
|
178
154
|
//#endregion
|
|
179
|
-
export { ANALYST_SEVERITIES, type Analyst, type AnalystContext, type AnalystCost, type AnalystFinding, type AnalystHooks, type AnalystInputKind, AnalystRegistry, type AnalystRegistryOptions, type AnalystRequirements, type AnalystRunEvent, type AnalystRunInputs, type AnalystRunResult, type AnalystRunSummary, type AnalystSeverity, type AnalystUsageReceipt, type BudgetPolicy, type ChatCallOpts, type ChatClient, type ChatRequest, type ChatResponse, type ChatTransport, type CliBridgeTransportOpts, type CreateAnalystAiConfig, type CreateChatClientOpts, type CreateTraceAnalystKindOpts, type CustomTransportOpts, DEFAULT_TRACE_ANALYST_KINDS, type DefaultAnalystRegistryOptions, type DiffPolicy, type DirectProviderTransportOpts, type EvidenceRef, FAILURE_MODE_KIND_SPEC, FINDING_SUBJECT_GRAMMAR_PROMPT, FINDING_SUBJECT_KINDS, FINDING_SUBJECT_SYNTAX, type FindingSubject, type FindingSubjectKind, FindingSubjectStringSchema, type FindingsDiff, FindingsStore, IMPROVEMENT_KIND_SPEC, type JudgeAdapterOpts, KIND_EXPECTED_SUBJECTS, KNOWLEDGE_GAP_KIND_SPEC, KNOWLEDGE_POISONING_KIND_SPEC, type MockTransportOpts, type PersistedFinding, RAW_FINDING_SCHEMA_PROMPT, type RawAnalystEvidence, RawAnalystEvidenceSchema, type RawAnalystFinding, RawAnalystFindingSchema, type RegistryRunOpts, type RouterTransportOpts, type RunCriticAdapterOpts, SKILL_USAGE_ANALYST, type SandboxSdkTransportOpts, type SemanticConceptJudgeAdapterOpts, SkillUsageAnalyst, type SkillUsageRecord, type SkillUsageReport, type SkillUsageScanConfig, type StructureFindingsOptions, type StructureFindingsResult, type TraceAnalystGolden, type TraceAnalystKindSpec, type TraceToolGroupName, type VerifierAdapterOpts,
|
|
155
|
+
export { ANALYST_SEVERITIES, type Analyst, type AnalystContext, type AnalystCost, type AnalystFinding, type AnalystHooks, type AnalystInputKind, AnalystRegistry, type AnalystRegistryOptions, type AnalystRequirements, type AnalystRunEvent, type AnalystRunInputs, type AnalystRunResult, type AnalystRunSummary, type AnalystSeverity, type AnalystUsageReceipt, type BudgetPolicy, type ChatCallOpts, type ChatClient, type ChatRequest, type ChatResponse, type ChatTransport, type CliBridgeTransportOpts, type CreateAnalystAiConfig, type CreateChatClientOpts, type CreateTraceAnalystKindOpts, type CustomTransportOpts, DEFAULT_TRACE_ANALYST_KINDS, type DefaultAnalystRegistryOptions, type DiffPolicy, type DirectProviderTransportOpts, type EvidenceRef, FAILURE_MODE_KIND_SPEC, FINDING_SUBJECT_GRAMMAR_PROMPT, FINDING_SUBJECT_KINDS, FINDING_SUBJECT_SYNTAX, type FindingSubject, type FindingSubjectKind, FindingSubjectStringSchema, type FindingsDiff, FindingsStore, IMPROVEMENT_KIND_SPEC, type JudgeAdapterOpts, KIND_EXPECTED_SUBJECTS, KNOWLEDGE_GAP_KIND_SPEC, KNOWLEDGE_POISONING_KIND_SPEC, type MockTransportOpts, type PersistedFinding, type ProposalFinding, type ProposalFindingOrigin, RAW_FINDING_SCHEMA_PROMPT, type RawAnalystEvidence, RawAnalystEvidenceSchema, type RawAnalystFinding, RawAnalystFindingSchema, type RegistryRunOpts, type RouterTransportOpts, type RunCriticAdapterOpts, SKILL_USAGE_ANALYST, type SandboxSdkTransportOpts, type SemanticConceptJudgeAdapterOpts, SkillUsageAnalyst, type SkillUsageRecord, type SkillUsageReport, type SkillUsageScanConfig, type StructureFindingsOptions, type StructureFindingsResult, type TraceAnalystGolden, type TraceAnalystKindSpec, type TraceToolGroupName, type VerifierAdapterOpts, assertProposalFindings, behavioralAnalyst, buildDefaultAnalystRegistry, buildSkillUsageReport, buildTraceToolsForGroup, coerceJson, coerceToFindingRows, computeFindingId, createAnalystAi, createChatClient, createJudgeAdapter, createRunCriticAdapter, createSemanticConceptJudgeAdapter, createTraceAnalystKind, createVerifierAdapter, defaultIsMaterial, deriveEfficiencyFindings, diffFindings, emitSkillUsageFindings, evidenceRefsFromRawFinding, findingSubjectGrammarPromptFor, isProposalFinding, liftSeverity, makeFinding, makeProposalFinding, parseFindingSubject, parseRawFinding, renderFindingSubject, renderPriorFindings, renderUpstreamFindings, stripCodeFences, structureFindings };
|
|
180
156
|
//# sourceMappingURL=index.d.ts.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"index.d.ts","names":[],"sources":["../../src/analyst/adapters.ts","../../src/analyst/behavioral-analyst.ts","../../src/analyst/parse-tolerant.ts","../../src/analyst/
|
|
1
|
+
{"version":3,"file":"index.d.ts","names":[],"sources":["../../src/analyst/adapters.ts","../../src/analyst/behavioral-analyst.ts","../../src/analyst/parse-tolerant.ts","../../src/analyst/proposal-findings.ts","../../src/analyst/structure-findings.ts","../../src/analyst/tool-groups.ts"],"mappings":";;;;;;;;;;iBA4CgB,aAAa,GAAG,WAAgB;UAe/B,oBAAoB;EACnC;EACA;EACA,UAAU,mBAAmB;;;;;EAK7B,UAAU,KAAK,cAAc;;iBAGf,sBAAsB,KAAK,MAAM,oBAAoB,OAAO,QAAQ;UAwEnE;EACf;EACA;EACA,SAAS;;EAET;;iBAGc,uBAAuB,OAAM,uBAA4B,QAAQ;UAiEhE;EACf;EACA;EACA,OAAO;;EAEP,MAAM;;EAEN,OAAO;;EAEP;;iBAGc,mBAAmB,MAAM,mBAAmB,QAAQ;UAiDnD;EACf;EACA;;EAEA,UAAU,KAAK;;EAEf;;iBAGc,kCACd,OAAM,kCACL,QAAQ;;;;;;;iBC5OK,yBACd,SAAS,mBACT;EAAQ;EAAoB;IAC3B;;iBAkCa,qBAAqB,QAAQ;;;;;;;;;;;;;;;iBC3E7B,gBAAgB;;;;;iBAgBhB,WAAW;;;;;;;iBAeX,oBAAoB;;;;iBCXpB,kBAAkB,mBAAmB,WAAW;;;;;;iBAShD,uBACd,mBACA,mBACC,cAAc;;;UCXA;;EAEf;EACA;;EAEA;EACA;EACA;EACA;;EAEA,aAAa;EACb;EACA,WAAW;EACX;EACA,SAAS;;EAET;;EAEA,cAAc,KAAK,sBAAsB;;EAEzC,kBAAkB;;EAElB,YAAY;;UAGG;EACf,UAAU;EACV;;iBAuCoB,kBACpB,MAAM,2BACL,QAAQ;;;;KCrFC;;;;;;;;;;;;;;;;;;iBAuCI,wBACd,OAAO,oBACP,OAAO,qBACN"}
|
package/dist/analyst/index.js
CHANGED
|
@@ -1,6 +1,7 @@
|
|
|
1
|
-
import { A as parseFindingSubject, C as stripCodeFences, D as FindingSubjectStringSchema, E as FINDING_SUBJECT_SYNTAX, F as
|
|
1
|
+
import { A as parseFindingSubject, C as stripCodeFences, D as FindingSubjectStringSchema, E as FINDING_SUBJECT_SYNTAX, F as createChatClient, I as createAnalystAi, M as behavioralAnalyst, N as deriveEfficiencyFindings, O as KIND_EXPECTED_SUBJECTS, S as coerceToFindingRows, T as FINDING_SUBJECT_KINDS, _ as RawAnalystEvidenceSchema, a as KNOWLEDGE_GAP_KIND_SPEC, b as parseRawFinding, c as buildTraceToolsForGroup, d as renderUpstreamFindings, f as settleUsageReceiptFromCostLedger, g as RAW_FINDING_SCHEMA_PROMPT, h as ANALYST_SEVERITIES, i as KNOWLEDGE_POISONING_KIND_SPEC, j as renderFindingSubject, k as findingSubjectGrammarPromptFor, l as createTraceAnalystKind, m as structureFindings, n as AnalystRegistry, o as IMPROVEMENT_KIND_SPEC, p as validateUsageSettlementTimeout, r as DEFAULT_TRACE_ANALYST_KINDS, s as FAILURE_MODE_KIND_SPEC, t as buildDefaultAnalystRegistry, u as renderPriorFindings, v as RawAnalystFindingSchema, w as FINDING_SUBJECT_GRAMMAR_PROMPT, x as coerceJson, y as evidenceRefsFromRawFinding } from "../default-registry-IjYs7T8l.js";
|
|
2
2
|
import { i as CostLedger } from "../cost-ledger-BrJxbrMy.js";
|
|
3
|
-
import {
|
|
3
|
+
import { c as makeFinding, l as makeProposalFinding, n as isProposalFinding, s as computeFindingId, t as assertProposalFindings } from "../proposal-findings-DCawte-y.js";
|
|
4
|
+
import { a as RunCritic, c as buildSkillUsageReport, d as defaultIsMaterial, f as diffFindings, i as runSemanticConceptJudge, l as emitSkillUsageFindings, n as SEMANTIC_CONCEPT_JUDGE_VERSION, o as SKILL_USAGE_ANALYST, s as SkillUsageAnalyst, u as FindingsStore } from "../semantic-concept-judge-C0P1VTXD.js";
|
|
4
5
|
//#region src/analyst/adapters.ts
|
|
5
6
|
/**
|
|
6
7
|
* Adapter factories — lift each existing agent-eval primitive into the
|
|
@@ -288,56 +289,6 @@ function createSemanticConceptJudgeAdapter(opts = {}) {
|
|
|
288
289
|
};
|
|
289
290
|
}
|
|
290
291
|
//#endregion
|
|
291
|
-
|
|
292
|
-
/** Evidence grounded in the agent's OWN execution: OTLP trace elements
|
|
293
|
-
* (`span`/`event`) or the artifact it produced (`artifact`). */
|
|
294
|
-
const OBSERVABLE_KINDS = /* @__PURE__ */ new Set([
|
|
295
|
-
"span",
|
|
296
|
-
"event",
|
|
297
|
-
"artifact"
|
|
298
|
-
]);
|
|
299
|
-
/** DESCRIPTIVE predicate: does the finding cite at least one observable
|
|
300
|
-
* (span/event/artifact) evidence ref. Useful for ranking evidence quality or
|
|
301
|
-
* rendering — it is NOT the steer gate. Evidence presence is the WRONG
|
|
302
|
-
* discriminator for steering: a legitimate trace-analyst observation may cite
|
|
303
|
-
* nothing (it would be wrongly rejected), and a judge verdict may cite an
|
|
304
|
-
* artifact (it would be wrongly admitted). Use `assertNoJudgeVerdict` to gate
|
|
305
|
-
* steering; use this only where "is this grounded in observable evidence" is the
|
|
306
|
-
* literal question. */
|
|
307
|
-
function isTraceObservable(finding) {
|
|
308
|
-
return finding.evidence_refs.some((ref) => OBSERVABLE_KINDS.has(ref.kind));
|
|
309
|
-
}
|
|
310
|
-
/** True iff the finding is a JUDGE VERDICT (an acceptance score lifted into a
|
|
311
|
-
* finding), identified by provenance set at the lift site — independent of
|
|
312
|
-
* whatever evidence it cites. */
|
|
313
|
-
function isJudgeVerdict(finding) {
|
|
314
|
-
return finding.derived_from_judge === true;
|
|
315
|
-
}
|
|
316
|
-
/**
|
|
317
|
-
* THE steer firewall. Fail-loud guard for any path that admits analyst findings
|
|
318
|
-
* as STEERING input (the `f(trace)` role): rejects — naming the offenders — any
|
|
319
|
-
* finding whose provenance is a judge verdict, rather than let `J` leak into the
|
|
320
|
-
* loop. Returns the findings unchanged for chaining.
|
|
321
|
-
*
|
|
322
|
-
* Call this at the chokepoint where a detector that ALSO scores/gates has its
|
|
323
|
-
* findings turned into a steer (the judge-and-steer dual-role case). It keys on
|
|
324
|
-
* provenance, so it correctly admits evidence-less trace-analyst observations and
|
|
325
|
-
* correctly rejects an artifact-citing judge verdict — the cases an evidence
|
|
326
|
-
* check gets backwards.
|
|
327
|
-
*
|
|
328
|
-
* It is necessary, not sufficient: it stops PROVENANCE-tagged verdicts. A judge
|
|
329
|
-
* whose output is laundered through a hand-built finding with no provenance flag
|
|
330
|
-
* is out of its reach — provenance must be honestly set at every judge→finding
|
|
331
|
-
* lift (today: createJudgeAdapter). That is why the integrity rule lives at the
|
|
332
|
-
* lift site, and why ProposeContext.judgeScores?: never is the complementary
|
|
333
|
-
* compile-time tripwire on the obvious direct channel.
|
|
334
|
-
*/
|
|
335
|
-
function assertNoJudgeVerdict(findings, context = "steer") {
|
|
336
|
-
const leaks = findings.filter(isJudgeVerdict);
|
|
337
|
-
if (leaks.length > 0) throw new Error(`${context}: a judge verdict cannot be admitted as steering input — that is the held-out judge leaking into the loop. Offending judge-derived findings: [${leaks.map((f) => f.finding_id).join(", ")}]. Steering consumes observations of behavior, never acceptance verdicts.`);
|
|
338
|
-
return findings;
|
|
339
|
-
}
|
|
340
|
-
//#endregion
|
|
341
|
-
export { ANALYST_SEVERITIES, AnalystRegistry, DEFAULT_TRACE_ANALYST_KINDS, FAILURE_MODE_KIND_SPEC, FINDING_SUBJECT_GRAMMAR_PROMPT, FINDING_SUBJECT_KINDS, FINDING_SUBJECT_SYNTAX, FindingSubjectStringSchema, FindingsStore, IMPROVEMENT_KIND_SPEC, KIND_EXPECTED_SUBJECTS, KNOWLEDGE_GAP_KIND_SPEC, KNOWLEDGE_POISONING_KIND_SPEC, RAW_FINDING_SCHEMA_PROMPT, RawAnalystEvidenceSchema, RawAnalystFindingSchema, SKILL_USAGE_ANALYST, SkillUsageAnalyst, assertNoJudgeVerdict, behavioralAnalyst, buildDefaultAnalystRegistry, buildSkillUsageReport, buildTraceToolsForGroup, coerceJson, coerceToFindingRows, computeFindingId, createAnalystAi, createChatClient, createJudgeAdapter, createRunCriticAdapter, createSemanticConceptJudgeAdapter, createTraceAnalystKind, createVerifierAdapter, defaultIsMaterial, deriveEfficiencyFindings, diffFindings, emitSkillUsageFindings, evidenceRefsFromRawFinding, findingSubjectGrammarPromptFor, isJudgeVerdict, isTraceObservable, liftSeverity, makeFinding, parseFindingSubject, parseRawFinding, renderFindingSubject, renderPriorFindings, renderUpstreamFindings, stripCodeFences, structureFindings };
|
|
292
|
+
export { ANALYST_SEVERITIES, AnalystRegistry, DEFAULT_TRACE_ANALYST_KINDS, FAILURE_MODE_KIND_SPEC, FINDING_SUBJECT_GRAMMAR_PROMPT, FINDING_SUBJECT_KINDS, FINDING_SUBJECT_SYNTAX, FindingSubjectStringSchema, FindingsStore, IMPROVEMENT_KIND_SPEC, KIND_EXPECTED_SUBJECTS, KNOWLEDGE_GAP_KIND_SPEC, KNOWLEDGE_POISONING_KIND_SPEC, RAW_FINDING_SCHEMA_PROMPT, RawAnalystEvidenceSchema, RawAnalystFindingSchema, SKILL_USAGE_ANALYST, SkillUsageAnalyst, assertProposalFindings, behavioralAnalyst, buildDefaultAnalystRegistry, buildSkillUsageReport, buildTraceToolsForGroup, coerceJson, coerceToFindingRows, computeFindingId, createAnalystAi, createChatClient, createJudgeAdapter, createRunCriticAdapter, createSemanticConceptJudgeAdapter, createTraceAnalystKind, createVerifierAdapter, defaultIsMaterial, deriveEfficiencyFindings, diffFindings, emitSkillUsageFindings, evidenceRefsFromRawFinding, findingSubjectGrammarPromptFor, isProposalFinding, liftSeverity, makeFinding, makeProposalFinding, parseFindingSubject, parseRawFinding, renderFindingSubject, renderPriorFindings, renderUpstreamFindings, stripCodeFences, structureFindings };
|
|
342
293
|
|
|
343
294
|
//# sourceMappingURL=index.js.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"index.js","names":[],"sources":["../../src/analyst/adapters.ts","../../src/analyst/steer-firewall.ts"],"sourcesContent":["/**\n * Adapter factories — lift each existing agent-eval primitive into the\n * Analyst contract without re-implementing it.\n *\n * Five primitives, five factories. Each one:\n * - Builds an Analyst with a stable id (caller chooses; defaults\n * given), a sensible default `inputKind`, a version derived from\n * the wrapped primitive's version + an adapter revision, and an\n * `analyze()` that calls the primitive and lifts its output to\n * AnalystFinding[] using `makeFinding()`.\n * - Maps severities: the existing `Severity` ('critical' | 'major' |\n * 'minor' | 'info') projects onto AnalystSeverity ('critical' |\n * 'high' | 'medium' | 'low' | 'info'); 'major' → 'high', 'minor' →\n * 'medium'. Domain analysts that want finer-grained mapping override.\n *\n * Adapters never own state. Calling the same factory twice with the\n * same primitive instance is safe.\n */\n\nimport { CostLedger } from '../cost-ledger'\nimport type {\n Finding as LayerFinding,\n Severity as LayerSeverity,\n MultiLayerVerifier,\n VerifyOptions,\n} from '../multi-layer-verifier'\nimport { RunCritic, type RunTrace } from '../run-critic'\nimport {\n runSemanticConceptJudge,\n SEMANTIC_CONCEPT_JUDGE_VERSION,\n type SemanticConceptJudgeInput,\n type SemanticConceptJudgeOptions,\n type SemanticConceptJudgeResult,\n} from '../semantic-concept-judge'\nimport type { JudgeFn, JudgeInput, JudgeScore } from '../types'\nimport type { ChatClient } from './chat-client'\nimport type { Analyst, AnalystFinding, AnalystSeverity } from './types'\nimport { makeFinding } from './types'\nimport { settleUsageReceiptFromCostLedger, validateUsageSettlementTimeout } from './usage-receipt'\n\nconst ADAPTER_REV = '1'\n\n// ── Severity bridges ───────────────────────────────────────────────\n\nexport function liftSeverity(s: LayerSeverity): AnalystSeverity {\n switch (s) {\n case 'critical':\n return 'critical'\n case 'major':\n return 'high'\n case 'minor':\n return 'medium'\n case 'info':\n return 'info'\n }\n}\n\n// ── 1. MultiLayerVerifier → Analyst ─────────────────────────────────\n\nexport interface VerifierAdapterOpts<Env> {\n id?: string\n area?: string\n verifier: MultiLayerVerifier<Env>\n /**\n * The verifier expects an `env` per run. Adapters take it from\n * `AnalystRunInputs.custom[<id>]` via the registry's 'custom' routing.\n */\n options?: Omit<VerifyOptions<Env>, 'env'>\n}\n\nexport function createVerifierAdapter<Env>(opts: VerifierAdapterOpts<Env>): Analyst<Env> {\n const id = opts.id ?? 'multi-layer-verifier'\n const area = opts.area ?? 'verification'\n return {\n id,\n description:\n \"Runs a MultiLayerVerifier and lifts each layer's findings into the analyst envelope.\",\n inputKind: 'custom',\n cost: { kind: 'deterministic' },\n version: `verifier-${ADAPTER_REV}`,\n async analyze(env, ctx) {\n const report = await opts.verifier.run({ env, ...opts.options })\n const out: AnalystFinding[] = []\n for (const layer of report.layers) {\n for (const finding of layer.findings) {\n out.push(liftLayerFinding(id, area, layer.layer, finding))\n }\n // Layer-level signal: a failed/error layer is itself a finding\n // even if it didn't emit per-finding rows.\n if (layer.status === 'fail' || layer.status === 'error' || layer.status === 'timeout') {\n out.push(\n makeFinding({\n analyst_id: id,\n area,\n subject: layer.layer,\n claim: `layer \"${layer.layer}\" ${layer.status}: ${layer.reason ?? 'no reason given'}`,\n severity:\n layer.status === 'error' ? 'high' : layer.status === 'timeout' ? 'medium' : 'high',\n confidence: 1,\n evidence_refs: [],\n metadata: {\n layer_status: layer.status,\n duration_ms: layer.durationMs,\n score: layer.score,\n diagnostics: layer.diagnostics,\n },\n }),\n )\n }\n }\n ctx.log?.('verifier complete', {\n layers: report.layers.length,\n blended: report.blendedScore,\n all_pass: report.allPass,\n })\n return out\n },\n }\n}\n\nfunction liftLayerFinding(\n analyst_id: string,\n area: string,\n layer: string,\n f: LayerFinding,\n): AnalystFinding {\n return makeFinding({\n analyst_id,\n area,\n subject: f.layer ?? layer,\n claim: f.message,\n severity: liftSeverity(f.severity),\n confidence: 0.85,\n evidence_refs: f.evidence\n ? [{ kind: 'artifact', uri: 'inline:evidence', excerpt: f.evidence }]\n : [],\n metadata: f.detail,\n })\n}\n\n// ── 2. RunCritic → Analyst ──────────────────────────────────────────\n\nexport interface RunCriticAdapterOpts {\n id?: string\n area?: string\n critic?: RunCritic\n /** Optional threshold below which a dimension is reported as a finding. Default 0.5. */\n threshold?: number\n}\n\nexport function createRunCriticAdapter(opts: RunCriticAdapterOpts = {}): Analyst<RunTrace> {\n const id = opts.id ?? 'run-critic'\n const area = opts.area ?? 'run-quality'\n const critic = opts.critic ?? new RunCritic()\n const threshold = opts.threshold ?? 0.5\n return {\n id,\n description:\n 'Scores a single run across success / grounding / drift / tool-quality and surfaces below-threshold dimensions.',\n inputKind: 'custom',\n cost: { kind: 'deterministic' },\n version: `run-critic-${ADAPTER_REV}`,\n async analyze(trace) {\n const score = critic.scoreTrace(trace)\n const out: AnalystFinding[] = []\n const dims: Array<[keyof typeof score, AnalystSeverity, string]> = [\n ['success', 'critical', 'run did not complete successfully'],\n ['goalProgress', 'high', 'goal progress is low'],\n ['repoGroundedness', 'high', 'output is poorly grounded in the repository'],\n ['toolUseQuality', 'medium', 'tool use quality is low'],\n ['patchQuality', 'medium', 'no real patch/edit evidence'],\n ['testReality', 'high', 'no real test/build evidence'],\n ['finalGate', 'critical', 'final gate is blocking'],\n ]\n for (const [dim, sev, msg] of dims) {\n const value = score[dim] as number\n if (typeof value === 'number' && value < threshold) {\n out.push(\n makeFinding({\n analyst_id: id,\n area,\n subject: dim,\n claim: msg,\n rationale: `${dim}=${value.toFixed(2)} below threshold ${threshold}`,\n severity: sev,\n confidence: 1,\n evidence_refs: [],\n metadata: { dimension: dim, value, threshold, run_id: trace.run.runId },\n }),\n )\n }\n }\n // Drift penalty is high → surface as a finding (inverse threshold).\n if (score.driftPenalty > 1 - threshold) {\n out.push(\n makeFinding({\n analyst_id: id,\n area,\n subject: 'drift',\n claim: 'agent output drifted from repository signal',\n rationale: `driftPenalty=${score.driftPenalty.toFixed(2)}`,\n severity: 'medium',\n confidence: 0.9,\n evidence_refs: [],\n metadata: { drift_penalty: score.driftPenalty, notes: score.notes },\n }),\n )\n }\n return out\n },\n }\n}\n\n// ── 3. JudgeFn → Analyst ────────────────────────────────────────────\n\nexport interface JudgeAdapterOpts {\n id?: string\n area?: string\n judge: JudgeFn\n /** Chat client passed to the JudgeFn. */\n chat: ChatClient\n /** Optional cost classification — most judges call an LLM. */\n cost?: Analyst['cost']\n /** Optional threshold below which a JudgeScore becomes a finding. Default 6 (on 0-10 scale). */\n threshold?: number\n}\n\nexport function createJudgeAdapter(opts: JudgeAdapterOpts): Analyst<JudgeInput> {\n const id = opts.id ?? 'judge'\n const area = opts.area ?? 'judge'\n const threshold = opts.threshold ?? 6\n return {\n id,\n description:\n 'Wraps an agent-eval JudgeFn into an analyst; below-threshold dimensions surface as findings.',\n inputKind: 'judge-input',\n cost: opts.cost ?? { kind: 'llm' },\n version: `judge-${ADAPTER_REV}`,\n async analyze(input) {\n const scores = await opts.judge(opts.chat, input)\n return scores\n .filter((s) => normalize10(s.score) < threshold)\n .map((s) => liftJudgeScore(id, area, s))\n },\n }\n}\n\nfunction normalize10(s: number): number {\n // JudgeScore convention is 0-10 but some judges emit 0-1. Coerce to 0-10.\n return s <= 1 ? s * 10 : s\n}\n\nfunction liftJudgeScore(analyst_id: string, area: string, s: JudgeScore): AnalystFinding {\n const score10 = normalize10(s.score)\n const severity: AnalystSeverity =\n score10 < 3 ? 'critical' : score10 < 5 ? 'high' : score10 < 7 ? 'medium' : 'low'\n return makeFinding({\n analyst_id,\n area,\n subject: s.dimension,\n claim: `${s.judgeName}/${s.dimension} scored ${score10.toFixed(1)}/10`,\n rationale: s.reasoning,\n severity,\n confidence: 0.8,\n evidence_refs: s.evidence\n ? [{ kind: 'artifact', uri: 'inline:evidence', excerpt: s.evidence }]\n : [],\n // Provenance: this finding IS a judge verdict (an acceptance score), not an\n // observation of behavior. The steer firewall (assertNoJudgeVerdict) rejects\n // it from steering — even when it cites an artifact above — because letting a\n // verdict steer the next attempt is the held-out judge leaking into the loop.\n derived_from_judge: true,\n metadata: { judge_name: s.judgeName, dimension: s.dimension, score_10: score10 },\n })\n}\n\n// ── 4. SemanticConceptJudge → Analyst ──────────────────────────────\n\nexport interface SemanticConceptJudgeAdapterOpts {\n id?: string\n area?: string\n /** Registry context owns cancellation and the per-analyst cost ledger. */\n options?: Omit<SemanticConceptJudgeOptions, 'costLedger' | 'signal'>\n /** Maximum post-cancellation wait for a provider receipt. Default 5 seconds. */\n settlementTimeoutMs?: number\n}\n\nexport function createSemanticConceptJudgeAdapter(\n opts: SemanticConceptJudgeAdapterOpts = {},\n): Analyst<SemanticConceptJudgeInput> {\n const id = opts.id ?? 'semantic-concept-judge'\n const area = opts.area ?? 'concept-coverage'\n const settlementTimeoutMs = validateUsageSettlementTimeout(opts.settlementTimeoutMs)\n return {\n id,\n description:\n 'Runs the semantic-concept judge and surfaces missing / weak concepts as findings.',\n inputKind: 'custom',\n cost: {\n kind: 'llm',\n models: opts.options?.model ? [opts.options.model] : undefined,\n settlement_timeout_ms: settlementTimeoutMs,\n },\n version: `${SEMANTIC_CONCEPT_JUDGE_VERSION}-adapter-${ADAPTER_REV}`,\n async analyze(input, ctx) {\n const costLedger = new CostLedger(ctx.budgetUsd)\n let result: SemanticConceptJudgeResult\n try {\n result = await runSemanticConceptJudge(input, {\n ...opts.options,\n costLedger,\n signal: ctx.signal,\n })\n } finally {\n const usage = await settleUsageReceiptFromCostLedger(costLedger, {\n channel: 'judge',\n timeoutMs: settlementTimeoutMs,\n })\n if (!usage.settled) {\n ctx.log?.('semantic-concept judge provider settlement timed out', {\n pending_calls: usage.pendingCalls,\n timeout_ms: settlementTimeoutMs,\n })\n }\n ctx.recordUsage?.(usage.receipt)\n }\n if (!result.available) {\n return [\n makeFinding({\n analyst_id: id,\n area,\n claim: 'semantic-concept judge unavailable',\n rationale: result.error,\n severity: 'info',\n confidence: 1,\n evidence_refs: [],\n metadata: { reason: result.error },\n }),\n ]\n }\n const out: AnalystFinding[] = []\n for (const f of result.findings) {\n // Only surface gaps: missing concepts or low scores. Concepts at\n // 7+/10 with present=true are not findings — they're successes.\n if (f.present && f.score >= 7) continue\n out.push(\n makeFinding({\n analyst_id: id,\n area,\n subject: f.concept,\n claim: f.present\n ? `concept \"${f.concept}\" is weak (${f.score}/10)`\n : `concept \"${f.concept}\" is missing`,\n rationale: f.evidence,\n severity: liftSeverity(f.severity),\n confidence: 0.85,\n evidence_refs: [{ kind: 'artifact', uri: 'inline:evidence', excerpt: f.evidence }],\n metadata: {\n concept: f.concept,\n present: f.present,\n score_10: f.score,\n },\n }),\n )\n }\n return out\n },\n }\n}\n","// The realness-oracle firewall (docs/learning-flywheel.md, \"The steer is f(trace)\").\n//\n// A realness/authenticity signal has TWO legitimate roles that must stay\n// separated by a firewall:\n// (a) anchor judge J — write-only: scores the chosen output, gates promotion,\n// NEVER seen by the worker/optimizer mid-run (else the loop games it).\n// (b) steer f(trace) — an analyst observes the agent's OWN behavior in the\n// trace (\"imported a stub\", \"used a non-crypto PRNG where encryption was\n// required\") and steers the next attempt. Legitimate, because it is derived\n// from OBSERVABLE BEHAVIOR, not from J's held-out verdict.\n//\n// The correct discriminator is PROVENANCE, not evidence presence. A judge verdict\n// lifted into a finding (createJudgeAdapter → liftJudgeScore) is a verdict even\n// when it cites an artifact; an evidence-less trace-analyst bullet is an\n// observation even though it cites nothing. So the firewall keys on\n// `AnalystFinding.derived_from_judge` (set at the judge lift site), NOT on whether\n// evidence_refs is populated. The instant a verdict steers the next attempt it is\n// a back-channel for J and the loop Goodharts realness exactly as it would\n// Goodhart pass-rate.\n\nimport type { AnalystFinding, EvidenceRef } from './types'\n\n/** Evidence grounded in the agent's OWN execution: OTLP trace elements\n * (`span`/`event`) or the artifact it produced (`artifact`). */\nconst OBSERVABLE_KINDS: ReadonlySet<EvidenceRef['kind']> = new Set<EvidenceRef['kind']>([\n 'span',\n 'event',\n 'artifact',\n])\n\n/** DESCRIPTIVE predicate: does the finding cite at least one observable\n * (span/event/artifact) evidence ref. Useful for ranking evidence quality or\n * rendering — it is NOT the steer gate. Evidence presence is the WRONG\n * discriminator for steering: a legitimate trace-analyst observation may cite\n * nothing (it would be wrongly rejected), and a judge verdict may cite an\n * artifact (it would be wrongly admitted). Use `assertNoJudgeVerdict` to gate\n * steering; use this only where \"is this grounded in observable evidence\" is the\n * literal question. */\nexport function isTraceObservable(finding: AnalystFinding): boolean {\n return finding.evidence_refs.some((ref) => OBSERVABLE_KINDS.has(ref.kind))\n}\n\n/** True iff the finding is a JUDGE VERDICT (an acceptance score lifted into a\n * finding), identified by provenance set at the lift site — independent of\n * whatever evidence it cites. */\nexport function isJudgeVerdict(finding: AnalystFinding): boolean {\n return finding.derived_from_judge === true\n}\n\n/**\n * THE steer firewall. Fail-loud guard for any path that admits analyst findings\n * as STEERING input (the `f(trace)` role): rejects — naming the offenders — any\n * finding whose provenance is a judge verdict, rather than let `J` leak into the\n * loop. Returns the findings unchanged for chaining.\n *\n * Call this at the chokepoint where a detector that ALSO scores/gates has its\n * findings turned into a steer (the judge-and-steer dual-role case). It keys on\n * provenance, so it correctly admits evidence-less trace-analyst observations and\n * correctly rejects an artifact-citing judge verdict — the cases an evidence\n * check gets backwards.\n *\n * It is necessary, not sufficient: it stops PROVENANCE-tagged verdicts. A judge\n * whose output is laundered through a hand-built finding with no provenance flag\n * is out of its reach — provenance must be honestly set at every judge→finding\n * lift (today: createJudgeAdapter). That is why the integrity rule lives at the\n * lift site, and why ProposeContext.judgeScores?: never is the complementary\n * compile-time tripwire on the obvious direct channel.\n */\nexport function assertNoJudgeVerdict(\n findings: ReadonlyArray<AnalystFinding>,\n context = 'steer',\n): ReadonlyArray<AnalystFinding> {\n const leaks = findings.filter(isJudgeVerdict)\n if (leaks.length > 0) {\n throw new Error(\n `${context}: a judge verdict cannot be admitted as steering input — that is the ` +\n `held-out judge leaking into the loop. Offending judge-derived findings: [${leaks\n .map((f) => f.finding_id)\n .join(', ')}]. Steering consumes observations of behavior, never acceptance verdicts.`,\n )\n }\n return findings\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;AAwCA,MAAM,cAAc;AAIpB,SAAgB,aAAa,GAAmC;CAC9D,QAAQ,GAAR;EACE,KAAK,YACH,OAAO;EACT,KAAK,SACH,OAAO;EACT,KAAK,SACH,OAAO;EACT,KAAK,QACH,OAAO;CACX;AACF;AAeA,SAAgB,sBAA2B,MAA8C;CACvF,MAAM,KAAK,KAAK,MAAM;CACtB,MAAM,OAAO,KAAK,QAAQ;CAC1B,OAAO;EACL;EACA,aACE;EACF,WAAW;EACX,MAAM,EAAE,MAAM,gBAAgB;EAC9B,SAAS,YAAY;EACrB,MAAM,QAAQ,KAAK,KAAK;GACtB,MAAM,SAAS,MAAM,KAAK,SAAS,IAAI;IAAE;IAAK,GAAG,KAAK;GAAQ,CAAC;GAC/D,MAAM,MAAwB,CAAC;GAC/B,KAAK,MAAM,SAAS,OAAO,QAAQ;IACjC,KAAK,MAAM,WAAW,MAAM,UAC1B,IAAI,KAAK,iBAAiB,IAAI,MAAM,MAAM,OAAO,OAAO,CAAC;IAI3D,IAAI,MAAM,WAAW,UAAU,MAAM,WAAW,WAAW,MAAM,WAAW,WAC1E,IAAI,KACF,YAAY;KACV,YAAY;KACZ;KACA,SAAS,MAAM;KACf,OAAO,UAAU,MAAM,MAAM,IAAI,MAAM,OAAO,IAAI,MAAM,UAAU;KAClE,UACE,MAAM,WAAW,UAAU,SAAS,MAAM,WAAW,YAAY,WAAW;KAC9E,YAAY;KACZ,eAAe,CAAC;KAChB,UAAU;MACR,cAAc,MAAM;MACpB,aAAa,MAAM;MACnB,OAAO,MAAM;MACb,aAAa,MAAM;KACrB;IACF,CAAC,CACH;GAEJ;GACA,IAAI,MAAM,qBAAqB;IAC7B,QAAQ,OAAO,OAAO;IACtB,SAAS,OAAO;IAChB,UAAU,OAAO;GACnB,CAAC;GACD,OAAO;EACT;CACF;AACF;AAEA,SAAS,iBACP,YACA,MACA,OACA,GACgB;CAChB,OAAO,YAAY;EACjB;EACA;EACA,SAAS,EAAE,SAAS;EACpB,OAAO,EAAE;EACT,UAAU,aAAa,EAAE,QAAQ;EACjC,YAAY;EACZ,eAAe,EAAE,WACb,CAAC;GAAE,MAAM;GAAY,KAAK;GAAmB,SAAS,EAAE;EAAS,CAAC,IAClE,CAAC;EACL,UAAU,EAAE;CACd,CAAC;AACH;AAYA,SAAgB,uBAAuB,OAA6B,CAAC,GAAsB;CACzF,MAAM,KAAK,KAAK,MAAM;CACtB,MAAM,OAAO,KAAK,QAAQ;CAC1B,MAAM,SAAS,KAAK,UAAU,IAAI,UAAU;CAC5C,MAAM,YAAY,KAAK,aAAa;CACpC,OAAO;EACL;EACA,aACE;EACF,WAAW;EACX,MAAM,EAAE,MAAM,gBAAgB;EAC9B,SAAS,cAAc;EACvB,MAAM,QAAQ,OAAO;GACnB,MAAM,QAAQ,OAAO,WAAW,KAAK;GACrC,MAAM,MAAwB,CAAC;GAU/B,KAAK,MAAM,CAAC,KAAK,KAAK,QAAQ;IAR5B;KAAC;KAAW;KAAY;IAAmC;IAC3D;KAAC;KAAgB;KAAQ;IAAsB;IAC/C;KAAC;KAAoB;KAAQ;IAA6C;IAC1E;KAAC;KAAkB;KAAU;IAAyB;IACtD;KAAC;KAAgB;KAAU;IAA6B;IACxD;KAAC;KAAe;KAAQ;IAA6B;IACrD;KAAC;KAAa;KAAY;IAAwB;GAEnB,GAAG;IAClC,MAAM,QAAQ,MAAM;IACpB,IAAI,OAAO,UAAU,YAAY,QAAQ,WACvC,IAAI,KACF,YAAY;KACV,YAAY;KACZ;KACA,SAAS;KACT,OAAO;KACP,WAAW,GAAG,IAAI,GAAG,MAAM,QAAQ,CAAC,EAAE,mBAAmB;KACzD,UAAU;KACV,YAAY;KACZ,eAAe,CAAC;KAChB,UAAU;MAAE,WAAW;MAAK;MAAO;MAAW,QAAQ,MAAM,IAAI;KAAM;IACxE,CAAC,CACH;GAEJ;GAEA,IAAI,MAAM,eAAe,IAAI,WAC3B,IAAI,KACF,YAAY;IACV,YAAY;IACZ;IACA,SAAS;IACT,OAAO;IACP,WAAW,gBAAgB,MAAM,aAAa,QAAQ,CAAC;IACvD,UAAU;IACV,YAAY;IACZ,eAAe,CAAC;IAChB,UAAU;KAAE,eAAe,MAAM;KAAc,OAAO,MAAM;IAAM;GACpE,CAAC,CACH;GAEF,OAAO;EACT;CACF;AACF;AAgBA,SAAgB,mBAAmB,MAA6C;CAC9E,MAAM,KAAK,KAAK,MAAM;CACtB,MAAM,OAAO,KAAK,QAAQ;CAC1B,MAAM,YAAY,KAAK,aAAa;CACpC,OAAO;EACL;EACA,aACE;EACF,WAAW;EACX,MAAM,KAAK,QAAQ,EAAE,MAAM,MAAM;EACjC,SAAS,SAAS;EAClB,MAAM,QAAQ,OAAO;GAEnB,QAAO,MADc,KAAK,MAAM,KAAK,MAAM,KAAK,EAAA,CAE7C,QAAQ,MAAM,YAAY,EAAE,KAAK,IAAI,SAAS,CAAC,CAC/C,KAAK,MAAM,eAAe,IAAI,MAAM,CAAC,CAAC;EAC3C;CACF;AACF;AAEA,SAAS,YAAY,GAAmB;CAEtC,OAAO,KAAK,IAAI,IAAI,KAAK;AAC3B;AAEA,SAAS,eAAe,YAAoB,MAAc,GAA+B;CACvF,MAAM,UAAU,YAAY,EAAE,KAAK;CACnC,MAAM,WACJ,UAAU,IAAI,aAAa,UAAU,IAAI,SAAS,UAAU,IAAI,WAAW;CAC7E,OAAO,YAAY;EACjB;EACA;EACA,SAAS,EAAE;EACX,OAAO,GAAG,EAAE,UAAU,GAAG,EAAE,UAAU,UAAU,QAAQ,QAAQ,CAAC,EAAE;EAClE,WAAW,EAAE;EACb;EACA,YAAY;EACZ,eAAe,EAAE,WACb,CAAC;GAAE,MAAM;GAAY,KAAK;GAAmB,SAAS,EAAE;EAAS,CAAC,IAClE,CAAC;EAKL,oBAAoB;EACpB,UAAU;GAAE,YAAY,EAAE;GAAW,WAAW,EAAE;GAAW,UAAU;EAAQ;CACjF,CAAC;AACH;AAaA,SAAgB,kCACd,OAAwC,CAAC,GACL;CACpC,MAAM,KAAK,KAAK,MAAM;CACtB,MAAM,OAAO,KAAK,QAAQ;CAC1B,MAAM,sBAAsB,+BAA+B,KAAK,mBAAmB;CACnF,OAAO;EACL;EACA,aACE;EACF,WAAW;EACX,MAAM;GACJ,MAAM;GACN,QAAQ,KAAK,SAAS,QAAQ,CAAC,KAAK,QAAQ,KAAK,IAAI,KAAA;GACrD,uBAAuB;EACzB;EACA,SAAS,GAAG,+BAA+B,WAAW;EACtD,MAAM,QAAQ,OAAO,KAAK;GACxB,MAAM,aAAa,IAAI,WAAW,IAAI,SAAS;GAC/C,IAAI;GACJ,IAAI;IACF,SAAS,MAAM,wBAAwB,OAAO;KAC5C,GAAG,KAAK;KACR;KACA,QAAQ,IAAI;IACd,CAAC;GACH,UAAU;IACR,MAAM,QAAQ,MAAM,iCAAiC,YAAY;KAC/D,SAAS;KACT,WAAW;IACb,CAAC;IACD,IAAI,CAAC,MAAM,SACT,IAAI,MAAM,wDAAwD;KAChE,eAAe,MAAM;KACrB,YAAY;IACd,CAAC;IAEH,IAAI,cAAc,MAAM,OAAO;GACjC;GACA,IAAI,CAAC,OAAO,WACV,OAAO,CACL,YAAY;IACV,YAAY;IACZ;IACA,OAAO;IACP,WAAW,OAAO;IAClB,UAAU;IACV,YAAY;IACZ,eAAe,CAAC;IAChB,UAAU,EAAE,QAAQ,OAAO,MAAM;GACnC,CAAC,CACH;GAEF,MAAM,MAAwB,CAAC;GAC/B,KAAK,MAAM,KAAK,OAAO,UAAU;IAG/B,IAAI,EAAE,WAAW,EAAE,SAAS,GAAG;IAC/B,IAAI,KACF,YAAY;KACV,YAAY;KACZ;KACA,SAAS,EAAE;KACX,OAAO,EAAE,UACL,YAAY,EAAE,QAAQ,aAAa,EAAE,MAAM,QAC3C,YAAY,EAAE,QAAQ;KAC1B,WAAW,EAAE;KACb,UAAU,aAAa,EAAE,QAAQ;KACjC,YAAY;KACZ,eAAe,CAAC;MAAE,MAAM;MAAY,KAAK;MAAmB,SAAS,EAAE;KAAS,CAAC;KACjF,UAAU;MACR,SAAS,EAAE;MACX,SAAS,EAAE;MACX,UAAU,EAAE;KACd;IACF,CAAC,CACH;GACF;GACA,OAAO;EACT;CACF;AACF;;;;;ACxVA,MAAM,mCAAqD,IAAI,IAAyB;CACtF;CACA;CACA;AACF,CAAC;;;;;;;;;AAUD,SAAgB,kBAAkB,SAAkC;CAClE,OAAO,QAAQ,cAAc,MAAM,QAAQ,iBAAiB,IAAI,IAAI,IAAI,CAAC;AAC3E;;;;AAKA,SAAgB,eAAe,SAAkC;CAC/D,OAAO,QAAQ,uBAAuB;AACxC;;;;;;;;;;;;;;;;;;;;AAqBA,SAAgB,qBACd,UACA,UAAU,SACqB;CAC/B,MAAM,QAAQ,SAAS,OAAO,cAAc;CAC5C,IAAI,MAAM,SAAS,GACjB,MAAM,IAAI,MACR,GAAG,QAAQ,gJACmE,MACzE,KAAK,MAAM,EAAE,UAAU,CAAC,CACxB,KAAK,IAAI,EAAE,0EAClB;CAEF,OAAO;AACT"}
|
|
1
|
+
{"version":3,"file":"index.js","names":[],"sources":["../../src/analyst/adapters.ts"],"sourcesContent":["/**\n * Adapter factories — lift each existing agent-eval primitive into the\n * Analyst contract without re-implementing it.\n *\n * Five primitives, five factories. Each one:\n * - Builds an Analyst with a stable id (caller chooses; defaults\n * given), a sensible default `inputKind`, a version derived from\n * the wrapped primitive's version + an adapter revision, and an\n * `analyze()` that calls the primitive and lifts its output to\n * AnalystFinding[] using `makeFinding()`.\n * - Maps severities: the existing `Severity` ('critical' | 'major' |\n * 'minor' | 'info') projects onto AnalystSeverity ('critical' |\n * 'high' | 'medium' | 'low' | 'info'); 'major' → 'high', 'minor' →\n * 'medium'. Domain analysts that want finer-grained mapping override.\n *\n * Adapters never own state. Calling the same factory twice with the\n * same primitive instance is safe.\n */\n\nimport { CostLedger } from '../cost-ledger'\nimport type {\n Finding as LayerFinding,\n Severity as LayerSeverity,\n MultiLayerVerifier,\n VerifyOptions,\n} from '../multi-layer-verifier'\nimport { RunCritic, type RunTrace } from '../run-critic'\nimport {\n runSemanticConceptJudge,\n SEMANTIC_CONCEPT_JUDGE_VERSION,\n type SemanticConceptJudgeInput,\n type SemanticConceptJudgeOptions,\n type SemanticConceptJudgeResult,\n} from '../semantic-concept-judge'\nimport type { JudgeFn, JudgeInput, JudgeScore } from '../types'\nimport type { ChatClient } from './chat-client'\nimport type { Analyst, AnalystFinding, AnalystSeverity } from './types'\nimport { makeFinding } from './types'\nimport { settleUsageReceiptFromCostLedger, validateUsageSettlementTimeout } from './usage-receipt'\n\nconst ADAPTER_REV = '1'\n\n// ── Severity bridges ───────────────────────────────────────────────\n\nexport function liftSeverity(s: LayerSeverity): AnalystSeverity {\n switch (s) {\n case 'critical':\n return 'critical'\n case 'major':\n return 'high'\n case 'minor':\n return 'medium'\n case 'info':\n return 'info'\n }\n}\n\n// ── 1. MultiLayerVerifier → Analyst ─────────────────────────────────\n\nexport interface VerifierAdapterOpts<Env> {\n id?: string\n area?: string\n verifier: MultiLayerVerifier<Env>\n /**\n * The verifier expects an `env` per run. Adapters take it from\n * `AnalystRunInputs.custom[<id>]` via the registry's 'custom' routing.\n */\n options?: Omit<VerifyOptions<Env>, 'env'>\n}\n\nexport function createVerifierAdapter<Env>(opts: VerifierAdapterOpts<Env>): Analyst<Env> {\n const id = opts.id ?? 'multi-layer-verifier'\n const area = opts.area ?? 'verification'\n return {\n id,\n description:\n \"Runs a MultiLayerVerifier and lifts each layer's findings into the analyst envelope.\",\n inputKind: 'custom',\n cost: { kind: 'deterministic' },\n version: `verifier-${ADAPTER_REV}`,\n async analyze(env, ctx) {\n const report = await opts.verifier.run({ env, ...opts.options })\n const out: AnalystFinding[] = []\n for (const layer of report.layers) {\n for (const finding of layer.findings) {\n out.push(liftLayerFinding(id, area, layer.layer, finding))\n }\n // Layer-level signal: a failed/error layer is itself a finding\n // even if it didn't emit per-finding rows.\n if (layer.status === 'fail' || layer.status === 'error' || layer.status === 'timeout') {\n out.push(\n makeFinding({\n analyst_id: id,\n area,\n subject: layer.layer,\n claim: `layer \"${layer.layer}\" ${layer.status}: ${layer.reason ?? 'no reason given'}`,\n severity:\n layer.status === 'error' ? 'high' : layer.status === 'timeout' ? 'medium' : 'high',\n confidence: 1,\n evidence_refs: [],\n metadata: {\n layer_status: layer.status,\n duration_ms: layer.durationMs,\n score: layer.score,\n diagnostics: layer.diagnostics,\n },\n }),\n )\n }\n }\n ctx.log?.('verifier complete', {\n layers: report.layers.length,\n blended: report.blendedScore,\n all_pass: report.allPass,\n })\n return out\n },\n }\n}\n\nfunction liftLayerFinding(\n analyst_id: string,\n area: string,\n layer: string,\n f: LayerFinding,\n): AnalystFinding {\n return makeFinding({\n analyst_id,\n area,\n subject: f.layer ?? layer,\n claim: f.message,\n severity: liftSeverity(f.severity),\n confidence: 0.85,\n evidence_refs: f.evidence\n ? [{ kind: 'artifact', uri: 'inline:evidence', excerpt: f.evidence }]\n : [],\n metadata: f.detail,\n })\n}\n\n// ── 2. RunCritic → Analyst ──────────────────────────────────────────\n\nexport interface RunCriticAdapterOpts {\n id?: string\n area?: string\n critic?: RunCritic\n /** Optional threshold below which a dimension is reported as a finding. Default 0.5. */\n threshold?: number\n}\n\nexport function createRunCriticAdapter(opts: RunCriticAdapterOpts = {}): Analyst<RunTrace> {\n const id = opts.id ?? 'run-critic'\n const area = opts.area ?? 'run-quality'\n const critic = opts.critic ?? new RunCritic()\n const threshold = opts.threshold ?? 0.5\n return {\n id,\n description:\n 'Scores a single run across success / grounding / drift / tool-quality and surfaces below-threshold dimensions.',\n inputKind: 'custom',\n cost: { kind: 'deterministic' },\n version: `run-critic-${ADAPTER_REV}`,\n async analyze(trace) {\n const score = critic.scoreTrace(trace)\n const out: AnalystFinding[] = []\n const dims: Array<[keyof typeof score, AnalystSeverity, string]> = [\n ['success', 'critical', 'run did not complete successfully'],\n ['goalProgress', 'high', 'goal progress is low'],\n ['repoGroundedness', 'high', 'output is poorly grounded in the repository'],\n ['toolUseQuality', 'medium', 'tool use quality is low'],\n ['patchQuality', 'medium', 'no real patch/edit evidence'],\n ['testReality', 'high', 'no real test/build evidence'],\n ['finalGate', 'critical', 'final gate is blocking'],\n ]\n for (const [dim, sev, msg] of dims) {\n const value = score[dim] as number\n if (typeof value === 'number' && value < threshold) {\n out.push(\n makeFinding({\n analyst_id: id,\n area,\n subject: dim,\n claim: msg,\n rationale: `${dim}=${value.toFixed(2)} below threshold ${threshold}`,\n severity: sev,\n confidence: 1,\n evidence_refs: [],\n metadata: { dimension: dim, value, threshold, run_id: trace.run.runId },\n }),\n )\n }\n }\n // Drift penalty is high → surface as a finding (inverse threshold).\n if (score.driftPenalty > 1 - threshold) {\n out.push(\n makeFinding({\n analyst_id: id,\n area,\n subject: 'drift',\n claim: 'agent output drifted from repository signal',\n rationale: `driftPenalty=${score.driftPenalty.toFixed(2)}`,\n severity: 'medium',\n confidence: 0.9,\n evidence_refs: [],\n metadata: { drift_penalty: score.driftPenalty, notes: score.notes },\n }),\n )\n }\n return out\n },\n }\n}\n\n// ── 3. JudgeFn → Analyst ────────────────────────────────────────────\n\nexport interface JudgeAdapterOpts {\n id?: string\n area?: string\n judge: JudgeFn\n /** Chat client passed to the JudgeFn. */\n chat: ChatClient\n /** Optional cost classification — most judges call an LLM. */\n cost?: Analyst['cost']\n /** Optional threshold below which a JudgeScore becomes a finding. Default 6 (on 0-10 scale). */\n threshold?: number\n}\n\nexport function createJudgeAdapter(opts: JudgeAdapterOpts): Analyst<JudgeInput> {\n const id = opts.id ?? 'judge'\n const area = opts.area ?? 'judge'\n const threshold = opts.threshold ?? 6\n return {\n id,\n description:\n 'Wraps an agent-eval JudgeFn into an analyst; below-threshold dimensions surface as findings.',\n inputKind: 'judge-input',\n cost: opts.cost ?? { kind: 'llm' },\n version: `judge-${ADAPTER_REV}`,\n async analyze(input) {\n const scores = await opts.judge(opts.chat, input)\n return scores\n .filter((s) => normalize10(s.score) < threshold)\n .map((s) => liftJudgeScore(id, area, s))\n },\n }\n}\n\nfunction normalize10(s: number): number {\n // JudgeScore convention is 0-10 but some judges emit 0-1. Coerce to 0-10.\n return s <= 1 ? s * 10 : s\n}\n\nfunction liftJudgeScore(analyst_id: string, area: string, s: JudgeScore): AnalystFinding {\n const score10 = normalize10(s.score)\n const severity: AnalystSeverity =\n score10 < 3 ? 'critical' : score10 < 5 ? 'high' : score10 < 7 ? 'medium' : 'low'\n return makeFinding({\n analyst_id,\n area,\n subject: s.dimension,\n claim: `${s.judgeName}/${s.dimension} scored ${score10.toFixed(1)}/10`,\n rationale: s.reasoning,\n severity,\n confidence: 0.8,\n evidence_refs: s.evidence\n ? [{ kind: 'artifact', uri: 'inline:evidence', excerpt: s.evidence }]\n : [],\n // Descriptive origin only. A caller may admit search feedback into candidate\n // generation, but final evaluation findings have no allowed proposal origin.\n derived_from_judge: true,\n metadata: { judge_name: s.judgeName, dimension: s.dimension, score_10: score10 },\n })\n}\n\n// ── 4. SemanticConceptJudge → Analyst ──────────────────────────────\n\nexport interface SemanticConceptJudgeAdapterOpts {\n id?: string\n area?: string\n /** Registry context owns cancellation and the per-analyst cost ledger. */\n options?: Omit<SemanticConceptJudgeOptions, 'costLedger' | 'signal'>\n /** Maximum post-cancellation wait for a provider receipt. Default 5 seconds. */\n settlementTimeoutMs?: number\n}\n\nexport function createSemanticConceptJudgeAdapter(\n opts: SemanticConceptJudgeAdapterOpts = {},\n): Analyst<SemanticConceptJudgeInput> {\n const id = opts.id ?? 'semantic-concept-judge'\n const area = opts.area ?? 'concept-coverage'\n const settlementTimeoutMs = validateUsageSettlementTimeout(opts.settlementTimeoutMs)\n return {\n id,\n description:\n 'Runs the semantic-concept judge and surfaces missing / weak concepts as findings.',\n inputKind: 'custom',\n cost: {\n kind: 'llm',\n models: opts.options?.model ? [opts.options.model] : undefined,\n settlement_timeout_ms: settlementTimeoutMs,\n },\n version: `${SEMANTIC_CONCEPT_JUDGE_VERSION}-adapter-${ADAPTER_REV}`,\n async analyze(input, ctx) {\n const costLedger = new CostLedger(ctx.budgetUsd)\n let result: SemanticConceptJudgeResult\n try {\n result = await runSemanticConceptJudge(input, {\n ...opts.options,\n costLedger,\n signal: ctx.signal,\n })\n } finally {\n const usage = await settleUsageReceiptFromCostLedger(costLedger, {\n channel: 'judge',\n timeoutMs: settlementTimeoutMs,\n })\n if (!usage.settled) {\n ctx.log?.('semantic-concept judge provider settlement timed out', {\n pending_calls: usage.pendingCalls,\n timeout_ms: settlementTimeoutMs,\n })\n }\n ctx.recordUsage?.(usage.receipt)\n }\n if (!result.available) {\n return [\n makeFinding({\n analyst_id: id,\n area,\n claim: 'semantic-concept judge unavailable',\n rationale: result.error,\n severity: 'info',\n confidence: 1,\n evidence_refs: [],\n metadata: { reason: result.error },\n }),\n ]\n }\n const out: AnalystFinding[] = []\n for (const f of result.findings) {\n // Only surface gaps: missing concepts or low scores. Concepts at\n // 7+/10 with present=true are not findings — they're successes.\n if (f.present && f.score >= 7) continue\n out.push(\n makeFinding({\n analyst_id: id,\n area,\n subject: f.concept,\n claim: f.present\n ? `concept \"${f.concept}\" is weak (${f.score}/10)`\n : `concept \"${f.concept}\" is missing`,\n rationale: f.evidence,\n severity: liftSeverity(f.severity),\n confidence: 0.85,\n evidence_refs: [{ kind: 'artifact', uri: 'inline:evidence', excerpt: f.evidence }],\n metadata: {\n concept: f.concept,\n present: f.present,\n score_10: f.score,\n },\n }),\n )\n }\n return out\n },\n }\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;AAwCA,MAAM,cAAc;AAIpB,SAAgB,aAAa,GAAmC;CAC9D,QAAQ,GAAR;EACE,KAAK,YACH,OAAO;EACT,KAAK,SACH,OAAO;EACT,KAAK,SACH,OAAO;EACT,KAAK,QACH,OAAO;CACX;AACF;AAeA,SAAgB,sBAA2B,MAA8C;CACvF,MAAM,KAAK,KAAK,MAAM;CACtB,MAAM,OAAO,KAAK,QAAQ;CAC1B,OAAO;EACL;EACA,aACE;EACF,WAAW;EACX,MAAM,EAAE,MAAM,gBAAgB;EAC9B,SAAS,YAAY;EACrB,MAAM,QAAQ,KAAK,KAAK;GACtB,MAAM,SAAS,MAAM,KAAK,SAAS,IAAI;IAAE;IAAK,GAAG,KAAK;GAAQ,CAAC;GAC/D,MAAM,MAAwB,CAAC;GAC/B,KAAK,MAAM,SAAS,OAAO,QAAQ;IACjC,KAAK,MAAM,WAAW,MAAM,UAC1B,IAAI,KAAK,iBAAiB,IAAI,MAAM,MAAM,OAAO,OAAO,CAAC;IAI3D,IAAI,MAAM,WAAW,UAAU,MAAM,WAAW,WAAW,MAAM,WAAW,WAC1E,IAAI,KACF,YAAY;KACV,YAAY;KACZ;KACA,SAAS,MAAM;KACf,OAAO,UAAU,MAAM,MAAM,IAAI,MAAM,OAAO,IAAI,MAAM,UAAU;KAClE,UACE,MAAM,WAAW,UAAU,SAAS,MAAM,WAAW,YAAY,WAAW;KAC9E,YAAY;KACZ,eAAe,CAAC;KAChB,UAAU;MACR,cAAc,MAAM;MACpB,aAAa,MAAM;MACnB,OAAO,MAAM;MACb,aAAa,MAAM;KACrB;IACF,CAAC,CACH;GAEJ;GACA,IAAI,MAAM,qBAAqB;IAC7B,QAAQ,OAAO,OAAO;IACtB,SAAS,OAAO;IAChB,UAAU,OAAO;GACnB,CAAC;GACD,OAAO;EACT;CACF;AACF;AAEA,SAAS,iBACP,YACA,MACA,OACA,GACgB;CAChB,OAAO,YAAY;EACjB;EACA;EACA,SAAS,EAAE,SAAS;EACpB,OAAO,EAAE;EACT,UAAU,aAAa,EAAE,QAAQ;EACjC,YAAY;EACZ,eAAe,EAAE,WACb,CAAC;GAAE,MAAM;GAAY,KAAK;GAAmB,SAAS,EAAE;EAAS,CAAC,IAClE,CAAC;EACL,UAAU,EAAE;CACd,CAAC;AACH;AAYA,SAAgB,uBAAuB,OAA6B,CAAC,GAAsB;CACzF,MAAM,KAAK,KAAK,MAAM;CACtB,MAAM,OAAO,KAAK,QAAQ;CAC1B,MAAM,SAAS,KAAK,UAAU,IAAI,UAAU;CAC5C,MAAM,YAAY,KAAK,aAAa;CACpC,OAAO;EACL;EACA,aACE;EACF,WAAW;EACX,MAAM,EAAE,MAAM,gBAAgB;EAC9B,SAAS,cAAc;EACvB,MAAM,QAAQ,OAAO;GACnB,MAAM,QAAQ,OAAO,WAAW,KAAK;GACrC,MAAM,MAAwB,CAAC;GAU/B,KAAK,MAAM,CAAC,KAAK,KAAK,QAAQ;IAR5B;KAAC;KAAW;KAAY;IAAmC;IAC3D;KAAC;KAAgB;KAAQ;IAAsB;IAC/C;KAAC;KAAoB;KAAQ;IAA6C;IAC1E;KAAC;KAAkB;KAAU;IAAyB;IACtD;KAAC;KAAgB;KAAU;IAA6B;IACxD;KAAC;KAAe;KAAQ;IAA6B;IACrD;KAAC;KAAa;KAAY;IAAwB;GAEnB,GAAG;IAClC,MAAM,QAAQ,MAAM;IACpB,IAAI,OAAO,UAAU,YAAY,QAAQ,WACvC,IAAI,KACF,YAAY;KACV,YAAY;KACZ;KACA,SAAS;KACT,OAAO;KACP,WAAW,GAAG,IAAI,GAAG,MAAM,QAAQ,CAAC,EAAE,mBAAmB;KACzD,UAAU;KACV,YAAY;KACZ,eAAe,CAAC;KAChB,UAAU;MAAE,WAAW;MAAK;MAAO;MAAW,QAAQ,MAAM,IAAI;KAAM;IACxE,CAAC,CACH;GAEJ;GAEA,IAAI,MAAM,eAAe,IAAI,WAC3B,IAAI,KACF,YAAY;IACV,YAAY;IACZ;IACA,SAAS;IACT,OAAO;IACP,WAAW,gBAAgB,MAAM,aAAa,QAAQ,CAAC;IACvD,UAAU;IACV,YAAY;IACZ,eAAe,CAAC;IAChB,UAAU;KAAE,eAAe,MAAM;KAAc,OAAO,MAAM;IAAM;GACpE,CAAC,CACH;GAEF,OAAO;EACT;CACF;AACF;AAgBA,SAAgB,mBAAmB,MAA6C;CAC9E,MAAM,KAAK,KAAK,MAAM;CACtB,MAAM,OAAO,KAAK,QAAQ;CAC1B,MAAM,YAAY,KAAK,aAAa;CACpC,OAAO;EACL;EACA,aACE;EACF,WAAW;EACX,MAAM,KAAK,QAAQ,EAAE,MAAM,MAAM;EACjC,SAAS,SAAS;EAClB,MAAM,QAAQ,OAAO;GAEnB,QAAO,MADc,KAAK,MAAM,KAAK,MAAM,KAAK,EAAA,CAE7C,QAAQ,MAAM,YAAY,EAAE,KAAK,IAAI,SAAS,CAAC,CAC/C,KAAK,MAAM,eAAe,IAAI,MAAM,CAAC,CAAC;EAC3C;CACF;AACF;AAEA,SAAS,YAAY,GAAmB;CAEtC,OAAO,KAAK,IAAI,IAAI,KAAK;AAC3B;AAEA,SAAS,eAAe,YAAoB,MAAc,GAA+B;CACvF,MAAM,UAAU,YAAY,EAAE,KAAK;CACnC,MAAM,WACJ,UAAU,IAAI,aAAa,UAAU,IAAI,SAAS,UAAU,IAAI,WAAW;CAC7E,OAAO,YAAY;EACjB;EACA;EACA,SAAS,EAAE;EACX,OAAO,GAAG,EAAE,UAAU,GAAG,EAAE,UAAU,UAAU,QAAQ,QAAQ,CAAC,EAAE;EAClE,WAAW,EAAE;EACb;EACA,YAAY;EACZ,eAAe,EAAE,WACb,CAAC;GAAE,MAAM;GAAY,KAAK;GAAmB,SAAS,EAAE;EAAS,CAAC,IAClE,CAAC;EAGL,oBAAoB;EACpB,UAAU;GAAE,YAAY,EAAE;GAAW,WAAW,EAAE;GAAW,UAAU;EAAQ;CACjF,CAAC;AACH;AAaA,SAAgB,kCACd,OAAwC,CAAC,GACL;CACpC,MAAM,KAAK,KAAK,MAAM;CACtB,MAAM,OAAO,KAAK,QAAQ;CAC1B,MAAM,sBAAsB,+BAA+B,KAAK,mBAAmB;CACnF,OAAO;EACL;EACA,aACE;EACF,WAAW;EACX,MAAM;GACJ,MAAM;GACN,QAAQ,KAAK,SAAS,QAAQ,CAAC,KAAK,QAAQ,KAAK,IAAI,KAAA;GACrD,uBAAuB;EACzB;EACA,SAAS,GAAG,+BAA+B,WAAW;EACtD,MAAM,QAAQ,OAAO,KAAK;GACxB,MAAM,aAAa,IAAI,WAAW,IAAI,SAAS;GAC/C,IAAI;GACJ,IAAI;IACF,SAAS,MAAM,wBAAwB,OAAO;KAC5C,GAAG,KAAK;KACR;KACA,QAAQ,IAAI;IACd,CAAC;GACH,UAAU;IACR,MAAM,QAAQ,MAAM,iCAAiC,YAAY;KAC/D,SAAS;KACT,WAAW;IACb,CAAC;IACD,IAAI,CAAC,MAAM,SACT,IAAI,MAAM,wDAAwD;KAChE,eAAe,MAAM;KACrB,YAAY;IACd,CAAC;IAEH,IAAI,cAAc,MAAM,OAAO;GACjC;GACA,IAAI,CAAC,OAAO,WACV,OAAO,CACL,YAAY;IACV,YAAY;IACZ;IACA,OAAO;IACP,WAAW,OAAO;IAClB,UAAU;IACV,YAAY;IACZ,eAAe,CAAC;IAChB,UAAU,EAAE,QAAQ,OAAO,MAAM;GACnC,CAAC,CACH;GAEF,MAAM,MAAwB,CAAC;GAC/B,KAAK,MAAM,KAAK,OAAO,UAAU;IAG/B,IAAI,EAAE,WAAW,EAAE,SAAS,GAAG;IAC/B,IAAI,KACF,YAAY;KACV,YAAY;KACZ;KACA,SAAS,EAAE;KACX,OAAO,EAAE,UACL,YAAY,EAAE,QAAQ,aAAa,EAAE,MAAM,QAC3C,YAAY,EAAE,QAAQ;KAC1B,WAAW,EAAE;KACb,UAAU,aAAa,EAAE,QAAQ;KACjC,YAAY;KACZ,eAAe,CAAC;MAAE,MAAM;MAAY,KAAK;MAAmB,SAAS,EAAE;KAAS,CAAC;KACjF,UAAU;MACR,SAAS,EAAE;MACX,SAAS,EAAE;MACX,UAAU,EAAE;KACd;IACF,CAAC,CACH;GACF;GACA,OAAO;EACT;CACF;AACF"}
|
|
@@ -1,8 +1,8 @@
|
|
|
1
1
|
import { a as RunRecord } from "./run-record-DcObtIGh.js";
|
|
2
|
-
import { i as AnalystRegistry } from "./default-registry-
|
|
2
|
+
import { i as AnalystRegistry } from "./default-registry-Brxr728w.js";
|
|
3
3
|
import { a as DatasetScenario } from "./dataset-BvtnC8Dc.js";
|
|
4
|
-
import "./summary-report-
|
|
5
|
-
import { C as InsightReport, b as ExecutionInsight, v as CostProvenanceSummary } from "./client-
|
|
4
|
+
import "./summary-report-DGp0-_XO.js";
|
|
5
|
+
import { C as InsightReport, b as ExecutionInsight, v as CostProvenanceSummary } from "./client-BIyh1RCr.js";
|
|
6
6
|
//#region src/contract/analyze-runs.d.ts
|
|
7
7
|
interface AnalyzeRunsOptions {
|
|
8
8
|
/** The runs to analyze. */
|
|
@@ -69,4 +69,4 @@ declare function summarizeExecution(opts: SummarizeExecutionOptions): ExecutionR
|
|
|
69
69
|
declare function analyzeRuns(opts: AnalyzeRunsOptions): Promise<InsightReport>;
|
|
70
70
|
//#endregion
|
|
71
71
|
export { summarizeExecution as a, analyzeRuns as i, ExecutionReport as n, SummarizeExecutionOptions as r, AnalyzeRunsOptions as t };
|
|
72
|
-
//# sourceMappingURL=analyze-runs-
|
|
72
|
+
//# sourceMappingURL=analyze-runs-DMo3Lb_y.d.ts.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"analyze-runs-
|
|
1
|
+
{"version":3,"file":"analyze-runs-DMo3Lb_y.d.ts","names":[],"sources":["../src/contract/analyze-runs.ts"],"mappings":";;;;;;UAoEiB;;EAEf,MAAM;;;EAGN;;;;;EAKA;EACA;;;EAGA,kBAAkB;;;EAGlB,UAAU;;;;EAIV;IACE;IACA,cAAc;;;;;EAKhB,cAAc;IAAQ;IAAe;IAAe;;;EAEpD;;;;EAIA;;;;;;;;;EASA,eAAe;;;EAGf;;UAGe;EACf,MAAM;EACN;;UAGe;EACf,WAAW;EACX,gBAAgB;;;iBAIF,mBAAmB,MAAM,4BAA4B;iBAS/C,YAAY,MAAM,qBAAqB,QAAQ"}
|
|
@@ -1,2 +1,2 @@
|
|
|
1
|
-
import { A as BenchmarkMetricCalibrationOptions, B as BenchmarkSource, C as BenchmarkRunOptions, D as runBenchmarkAdapter, E as renderBenchmarkReportMarkdown, F as BenchmarkDatasetItem, H as deterministicSplit, I as BenchmarkEvaluation, L as BenchmarkFamily, M as calibrateBenchmarkMetric, N as BENCHMARK_SPLIT_SEED, O as summarizeBenchmarkCampaign, P as BenchmarkAdapter, R as BenchmarkResponder, S as BenchmarkReport, T as BenchmarkSliceSummary, V as BenchmarkTaskKind, _ as parseJsonlRows, a as StandardRetrievalDocument, b as retrievalMetricsAtCutoff, c as StandardRetrievalQrel, d as buildStandardRetrievalItems, f as createRetrievalIdBenchmarkAdapter, g as parseBeirQueriesJsonl, h as parseBeirCorpusJsonl, i as StandardRetrievalArtifact, j as BenchmarkMetricCalibrationResult, k as index_d_exports, l as StandardRetrievalQuery, m as normalizeRetrievedDocumentIds, n as BuildStandardRetrievalItemsOptions, o as StandardRetrievalEvaluationOptions, p as evaluateStandardRetrieval, r as RetrievalIdAdapterOptions, s as StandardRetrievalPayload, u as StandardRetrievalResult, v as parseQrels, w as BenchmarkRunResult, x as BenchmarkDistribution, y as parseTsvRows, z as BenchmarkScenario } from "../index-
|
|
1
|
+
import { A as BenchmarkMetricCalibrationOptions, B as BenchmarkSource, C as BenchmarkRunOptions, D as runBenchmarkAdapter, E as renderBenchmarkReportMarkdown, F as BenchmarkDatasetItem, H as deterministicSplit, I as BenchmarkEvaluation, L as BenchmarkFamily, M as calibrateBenchmarkMetric, N as BENCHMARK_SPLIT_SEED, O as summarizeBenchmarkCampaign, P as BenchmarkAdapter, R as BenchmarkResponder, S as BenchmarkReport, T as BenchmarkSliceSummary, V as BenchmarkTaskKind, _ as parseJsonlRows, a as StandardRetrievalDocument, b as retrievalMetricsAtCutoff, c as StandardRetrievalQrel, d as buildStandardRetrievalItems, f as createRetrievalIdBenchmarkAdapter, g as parseBeirQueriesJsonl, h as parseBeirCorpusJsonl, i as StandardRetrievalArtifact, j as BenchmarkMetricCalibrationResult, k as index_d_exports, l as StandardRetrievalQuery, m as normalizeRetrievedDocumentIds, n as BuildStandardRetrievalItemsOptions, o as StandardRetrievalEvaluationOptions, p as evaluateStandardRetrieval, r as RetrievalIdAdapterOptions, s as StandardRetrievalPayload, u as StandardRetrievalResult, v as parseQrels, w as BenchmarkRunResult, x as BenchmarkDistribution, y as parseTsvRows, z as BenchmarkScenario } from "../index-C21xKtxu.js";
|
|
2
2
|
export { BENCHMARK_SPLIT_SEED, type BenchmarkAdapter, type BenchmarkDatasetItem, type BenchmarkDistribution, type BenchmarkEvaluation, type BenchmarkFamily, type BenchmarkMetricCalibrationOptions, type BenchmarkMetricCalibrationResult, type BenchmarkReport, type BenchmarkResponder, type BenchmarkRunOptions, type BenchmarkRunResult, type BenchmarkScenario, type BenchmarkSliceSummary, type BenchmarkSource, type BenchmarkTaskKind, type BuildStandardRetrievalItemsOptions, type RetrievalIdAdapterOptions, type StandardRetrievalArtifact, type StandardRetrievalDocument, type StandardRetrievalEvaluationOptions, type StandardRetrievalPayload, type StandardRetrievalQrel, type StandardRetrievalQuery, type StandardRetrievalResult, buildStandardRetrievalItems, calibrateBenchmarkMetric, createRetrievalIdBenchmarkAdapter, deterministicSplit, evaluateStandardRetrieval, normalizeRetrievedDocumentIds, parseBeirCorpusJsonl, parseBeirQueriesJsonl, parseJsonlRows, parseQrels, parseTsvRows, renderBenchmarkReportMarkdown, retrievalMetricsAtCutoff, index_d_exports as routing, runBenchmarkAdapter, summarizeBenchmarkCampaign };
|
package/dist/benchmarks/index.js
CHANGED
|
@@ -1,2 +1,2 @@
|
|
|
1
|
-
import { _ as deterministicSplit, a as normalizeRetrievedDocumentIds, c as parseJsonlRows, d as retrievalMetricsAtCutoff, f as renderBenchmarkReportMarkdown, g as BENCHMARK_SPLIT_SEED, h as routing_exports, i as evaluateStandardRetrieval, l as parseQrels, m as summarizeBenchmarkCampaign, n as buildStandardRetrievalItems, o as parseBeirCorpusJsonl, p as runBenchmarkAdapter, r as createRetrievalIdBenchmarkAdapter, s as parseBeirQueriesJsonl, u as parseTsvRows, v as calibrateBenchmarkMetric } from "../benchmarks-
|
|
1
|
+
import { _ as deterministicSplit, a as normalizeRetrievedDocumentIds, c as parseJsonlRows, d as retrievalMetricsAtCutoff, f as renderBenchmarkReportMarkdown, g as BENCHMARK_SPLIT_SEED, h as routing_exports, i as evaluateStandardRetrieval, l as parseQrels, m as summarizeBenchmarkCampaign, n as buildStandardRetrievalItems, o as parseBeirCorpusJsonl, p as runBenchmarkAdapter, r as createRetrievalIdBenchmarkAdapter, s as parseBeirQueriesJsonl, u as parseTsvRows, v as calibrateBenchmarkMetric } from "../benchmarks-v5piCeDl.js";
|
|
2
2
|
export { BENCHMARK_SPLIT_SEED, buildStandardRetrievalItems, calibrateBenchmarkMetric, createRetrievalIdBenchmarkAdapter, deterministicSplit, evaluateStandardRetrieval, normalizeRetrievedDocumentIds, parseBeirCorpusJsonl, parseBeirQueriesJsonl, parseJsonlRows, parseQrels, parseTsvRows, renderBenchmarkReportMarkdown, retrievalMetricsAtCutoff, routing_exports as routing, runBenchmarkAdapter, summarizeBenchmarkCampaign };
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import { t as __exportAll } from "./rolldown-runtime-8H4AJuhK.js";
|
|
2
|
-
import { Y as runCampaign, Z as fsCampaignStorage } from "./skillopt-optimization-method-
|
|
3
|
-
import "./campaign
|
|
2
|
+
import { Y as runCampaign, Z as fsCampaignStorage } from "./skillopt-optimization-method-BY6vKLJB.js";
|
|
3
|
+
import "./campaign-DEC_7DLn.js";
|
|
4
4
|
import { join } from "node:path";
|
|
5
5
|
//#region src/benchmarks/calibration.ts
|
|
6
6
|
async function calibrateBenchmarkMetric(options) {
|
|
@@ -751,4 +751,4 @@ var benchmarks_exports = /* @__PURE__ */ __exportAll({
|
|
|
751
751
|
//#endregion
|
|
752
752
|
export { deterministicSplit as _, normalizeRetrievedDocumentIds as a, parseJsonlRows as c, retrievalMetricsAtCutoff as d, renderBenchmarkReportMarkdown as f, BENCHMARK_SPLIT_SEED as g, routing_exports as h, evaluateStandardRetrieval as i, parseQrels as l, summarizeBenchmarkCampaign as m, buildStandardRetrievalItems as n, parseBeirCorpusJsonl as o, runBenchmarkAdapter as p, createRetrievalIdBenchmarkAdapter as r, parseBeirQueriesJsonl as s, benchmarks_exports as t, parseTsvRows as u, calibrateBenchmarkMetric as v };
|
|
753
753
|
|
|
754
|
-
//# sourceMappingURL=benchmarks-
|
|
754
|
+
//# sourceMappingURL=benchmarks-v5piCeDl.js.map
|