@tangle-network/agent-eval 0.116.0 → 0.117.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +31 -0
- package/dist/analyst/index.d.ts +18 -11
- package/dist/analyst/index.js +10 -7
- package/dist/analyst/index.js.map +1 -1
- package/dist/{analyst-CFBc14Wc.d.ts → analyst-C8HHvfJp.d.ts} +1 -1
- package/dist/{analyze-runs-0rz_m29H.d.ts → analyze-runs--2x39HZ7.d.ts} +3 -3
- package/dist/{baseline-DsNteOgR.d.ts → baseline-DKq3gJpP.d.ts} +6 -3
- package/dist/belief-state/index.d.ts +6 -6
- package/dist/belief-state/index.js +1 -1
- package/dist/benchmarks/index.d.ts +11 -8
- package/dist/benchmarks/index.js +11 -10
- package/dist/builder-eval/index.d.ts +4 -4
- package/dist/builder-eval/index.js +1 -1
- package/dist/{calibration-Dz8TQV4y.d.ts → calibration-C8MTS7cw.d.ts} +2 -2
- package/dist/campaign/index.d.ts +53 -29
- package/dist/campaign/index.js +18 -13
- package/dist/chunk-3YYRZDON.js +45 -0
- package/dist/chunk-3YYRZDON.js.map +1 -0
- package/dist/{chunk-RPDDVKI7.js → chunk-4JLWXDYA.js} +2 -2
- package/dist/{chunk-NBSS5NDZ.js → chunk-CCZIVI3F.js} +54 -115
- package/dist/chunk-CCZIVI3F.js.map +1 -0
- package/dist/{chunk-J6P6PK2R.js → chunk-FQNLDL4D.js} +3 -3
- package/dist/{chunk-ONM6PEAE.js → chunk-GQCZRZ7L.js} +2 -2
- package/dist/chunk-HHWE3POT.js +94 -0
- package/dist/chunk-HHWE3POT.js.map +1 -0
- package/dist/{chunk-3274WNK7.js → chunk-HQPHZGL6.js} +687 -44
- package/dist/chunk-HQPHZGL6.js.map +1 -0
- package/dist/{chunk-FAOEFFRT.js → chunk-IDZTTFRR.js} +390 -78
- package/dist/chunk-IDZTTFRR.js.map +1 -0
- package/dist/{chunk-3LXTCTWL.js → chunk-JSDVRFAP.js} +2 -2
- package/dist/{chunk-MHNQWM4I.js → chunk-LQUTGLOZ.js} +5 -1
- package/dist/chunk-LQUTGLOZ.js.map +1 -0
- package/dist/{chunk-4D5RVB3W.js → chunk-LTVG32KX.js} +30 -5
- package/dist/chunk-LTVG32KX.js.map +1 -0
- package/dist/{chunk-CIUOICJT.js → chunk-MGEHEHSN.js} +62 -15
- package/dist/chunk-MGEHEHSN.js.map +1 -0
- package/dist/{chunk-GY4SYVPJ.js → chunk-NJC7U437.js} +97 -25
- package/dist/chunk-NJC7U437.js.map +1 -0
- package/dist/{chunk-NYFUT3B3.js → chunk-ODVOOEWQ.js} +31 -10
- package/dist/chunk-ODVOOEWQ.js.map +1 -0
- package/dist/{chunk-LNQEP766.js → chunk-S2F4J57L.js} +44 -4
- package/dist/chunk-S2F4J57L.js.map +1 -0
- package/dist/chunk-VCTY3W6J.js +798 -0
- package/dist/chunk-VCTY3W6J.js.map +1 -0
- package/dist/chunk-VF3XSYTI.js +545 -0
- package/dist/chunk-VF3XSYTI.js.map +1 -0
- package/dist/{chunk-TLDB7WRY.js → chunk-YZPO4UHR.js} +28 -31
- package/dist/chunk-YZPO4UHR.js.map +1 -0
- package/dist/{chunk-GSW3OBHK.js → chunk-ZUXV7UWZ.js} +350 -724
- package/dist/chunk-ZUXV7UWZ.js.map +1 -0
- package/dist/cli.js +4 -2
- package/dist/cli.js.map +1 -1
- package/dist/{code-agent-session-CdxteG0y.d.ts → code-agent-session-CjZsVd19.d.ts} +1 -1
- package/dist/contract/index.d.ts +43 -29
- package/dist/contract/index.js +56 -19
- package/dist/contract/index.js.map +1 -1
- package/dist/{control-DbcDxouY.d.ts → control-6vuGfmDH.d.ts} +5 -5
- package/dist/control.d.ts +6 -6
- package/dist/cost-ledger-DWy3XdJc.d.ts +183 -0
- package/dist/{default-registry-DDfv22MQ.d.ts → default-registry-DaK8b3fv.d.ts} +2 -2
- package/dist/{emitter-BRchAAAx.d.ts → emitter-CjD7vUwv.d.ts} +2 -2
- package/dist/{failure-cluster-C48PiReX.d.ts → failure-cluster-DOAcSJ87.d.ts} +2 -2
- package/dist/{feedback-trajectory-pDcz1lQ1.d.ts → feedback-trajectory-BUnM58xL.d.ts} +3 -3
- package/dist/fuzz.d.ts +8 -16
- package/dist/fuzz.js +72 -42
- package/dist/fuzz.js.map +1 -1
- package/dist/{gepa-CQelRtuC.d.ts → gepa-eESocoDi.d.ts} +56 -6
- package/dist/hosted/index.d.ts +13 -10
- package/dist/{index-DbCXJfZ1.d.ts → index-PdX4VnPA.d.ts} +3 -3
- package/dist/index.d.ts +102 -57
- package/dist/index.js +328 -235
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-oMVxDTxl.d.ts → insight-report-DY4nDW9Q.d.ts} +1 -1
- package/dist/{integrity-C6PZ73iC.d.ts → integrity-DqlBiLyK.d.ts} +2 -2
- package/dist/{kind-factory-DWOvXjR_.d.ts → kind-factory-ClZmO25A.d.ts} +2 -2
- package/dist/llm-client-qoDd18Qz.d.ts +289 -0
- package/dist/meta-eval/index.d.ts +8 -7
- package/dist/meta-eval/index.js +1 -1
- package/dist/multishot/index.d.ts +9 -6
- package/dist/openapi.json +1 -1
- package/dist/pipelines/index.d.ts +16 -6
- package/dist/pipelines/index.js +119 -23
- package/dist/pipelines/index.js.map +1 -1
- package/dist/{policy-edit-Clb2v6Oa.d.ts → policy-edit-wG9uFEFm.d.ts} +13 -266
- package/dist/{pre-registration--vU0mMtD.d.ts → pre-registration-BWQhJ3vz.d.ts} +24 -5
- package/dist/{provenance-BbVagC68.d.ts → provenance-DpjwyseI.d.ts} +6 -6
- package/dist/{query-Ck190MOd.d.ts → query-CF7PG61p.d.ts} +5 -3
- package/dist/raw-provider-sink-C46HDghv.d.ts +132 -0
- package/dist/{release-report-CamNDe90.d.ts → release-report-C8G2i5Xi.d.ts} +2 -2
- package/dist/reporting.d.ts +10 -9
- package/dist/{researcher-Dwbo_Fxx.d.ts → researcher-C8XyxQsu.d.ts} +8 -8
- package/dist/rl.d.ts +18 -15
- package/dist/rl.js +2 -2
- package/dist/{rubric-predictive-validity-BIdf9h4R.d.ts → rubric-predictive-validity-p49lLVrE.d.ts} +1 -1
- package/dist/{run-campaign-UADIM77S.js → run-campaign-IM26A6PD.js} +4 -2
- package/dist/{run-record-CZmcpWPo.d.ts → run-record-BDH49H2E.d.ts} +1 -1
- package/dist/{runtime-trajectory-CC0jx9ql.d.ts → runtime-trajectory-DGBIUt4B.d.ts} +1 -1
- package/dist/{schema-SGWcK9wa.d.ts → schema-B3Q3l9Z_.d.ts} +2 -0
- package/dist/{semantic-concept-judge-CKjePUMh.d.ts → semantic-concept-judge-CXnPEJbf.d.ts} +24 -6
- package/dist/{statistics-oUbOJe-S.d.ts → statistics-KUnG73jH.d.ts} +1 -1
- package/dist/{storage-Dw_f7WMt.d.ts → storage-DrX3v_5B.d.ts} +12 -1
- package/dist/{store-9cAScOcb.d.ts → store-C1YxJDEK.d.ts} +1 -132
- package/dist/{store-BsVi7ncX.d.ts → store-DGqD0Pyo.d.ts} +1 -1
- package/dist/storyboard/index.d.ts +1 -1
- package/dist/{summary-report-DTNgQycC.d.ts → summary-report-C5bKFfm-.d.ts} +2 -2
- package/dist/{test-graded-scenario-mzYBKspu.d.ts → test-graded-scenario-B0ybnPY7.d.ts} +3 -3
- package/dist/traces.d.ts +25 -14
- package/dist/traces.js +16 -4
- package/dist/{types-Ca_63YSD.d.ts → types-BSw1rOUB.d.ts} +41 -39
- package/dist/{types-C7DGg5ex.d.ts → types-BkfcQnxV.d.ts} +15 -0
- package/dist/wire/index.d.ts +28 -19
- package/dist/wire/index.js +4 -2
- package/docs/distributed-driver.md +1 -1
- package/package.json +3 -3
- package/dist/chunk-3274WNK7.js.map +0 -1
- package/dist/chunk-4D5RVB3W.js.map +0 -1
- package/dist/chunk-7GKEAIAD.js +0 -205
- package/dist/chunk-7GKEAIAD.js.map +0 -1
- package/dist/chunk-CIUOICJT.js.map +0 -1
- package/dist/chunk-FAOEFFRT.js.map +0 -1
- package/dist/chunk-GSW3OBHK.js.map +0 -1
- package/dist/chunk-GY4SYVPJ.js.map +0 -1
- package/dist/chunk-LNQEP766.js.map +0 -1
- package/dist/chunk-MHNQWM4I.js.map +0 -1
- package/dist/chunk-MPHTT5HE.js +0 -74
- package/dist/chunk-MPHTT5HE.js.map +0 -1
- package/dist/chunk-NBSS5NDZ.js.map +0 -1
- package/dist/chunk-NYFUT3B3.js.map +0 -1
- package/dist/chunk-TLDB7WRY.js.map +0 -1
- package/dist/cost-ledger-DuSqlw5B.d.ts +0 -113
- /package/dist/{chunk-RPDDVKI7.js.map → chunk-4JLWXDYA.js.map} +0 -0
- /package/dist/{chunk-J6P6PK2R.js.map → chunk-FQNLDL4D.js.map} +0 -0
- /package/dist/{chunk-ONM6PEAE.js.map → chunk-GQCZRZ7L.js.map} +0 -0
- /package/dist/{chunk-3LXTCTWL.js.map → chunk-JSDVRFAP.js.map} +0 -0
- /package/dist/{run-campaign-UADIM77S.js.map → run-campaign-IM26A6PD.js.map} +0 -0
package/CHANGELOG.md
CHANGED
|
@@ -4,6 +4,37 @@ All notable changes to `@tangle-network/agent-eval` and its sibling `agent-eval-
|
|
|
4
4
|
|
|
5
5
|
---
|
|
6
6
|
|
|
7
|
+
## [0.117.0] — 2026-07-13 — durable cost and bounded behavioral evidence
|
|
8
|
+
|
|
9
|
+
### Added
|
|
10
|
+
|
|
11
|
+
- `createReferenceEquivalenceJudge()` and `runReferenceEquivalenceJudge()` score whether an answer preserves the meaning of one or more references, with the same cost and transport accounting as other judges.
|
|
12
|
+
|
|
13
|
+
### Changed
|
|
14
|
+
|
|
15
|
+
- `CostLedger.runPaidCall()` is now the single paid-call path across campaigns, proposers, judges, analysts, and distillation.
|
|
16
|
+
It durably reserves maximum spend before dispatch, records provider receipts, blocks unresolved crash state, and enforces the run ceiling before another paid call starts.
|
|
17
|
+
- `ToolSpan.argsCaptured` distinguishes a call with unavailable arguments from a captured no-argument call.
|
|
18
|
+
Repeated-call analysis, failure clustering, tool-use metrics, and per-step redundancy grading no longer compare uncaptured arguments.
|
|
19
|
+
Every OTLP export path uses one mapping that preserves this distinction.
|
|
20
|
+
|
|
21
|
+
### Breaking
|
|
22
|
+
|
|
23
|
+
- `CostLedger.record()` is removed because recording spend after a provider call cannot enforce a cost limit or survive a crash.
|
|
24
|
+
Use `CostLedger.runPaidCall()` for billable work, `CostLedger` receipt import for already-settled calls, or `costForUsage()` for pure estimates.
|
|
25
|
+
- `computeTraceMetrics()` now rejects mixed-trace input, and `BehavioralMetrics` adds required `traceId` and `tokenSequences` fields.
|
|
26
|
+
The convenience token trajectories now expose the longest proven-serial sequence instead of flattening parallel branches.
|
|
27
|
+
- `ToolUseMetrics` and `ToolStats` add required `callsWithCapturedArgs` fields.
|
|
28
|
+
`duplicateRate` now uses captured-argument calls as its denominator.
|
|
29
|
+
|
|
30
|
+
### Fixed
|
|
31
|
+
|
|
32
|
+
- Repeated-call findings now require a contiguous, time-bounded, serial episode within one agent branch instead of grouping identical or concurrent calls across an entire run.
|
|
33
|
+
- Behavioral token findings now analyze each trace and serial agent timeline independently, use numeric time ordering across accepted timestamp formats, and only attribute output decay to context that actually grew.
|
|
34
|
+
- Behavioral issue IDs remain stable across trace runs while evidence retains exact trace identities and sampled prevalence.
|
|
35
|
+
- Partial timing isolates only the uncertain interval, and same-named root spans retain independent structural identity.
|
|
36
|
+
- Multi-trace behavioral findings use pattern-level claims while each trace's exact values remain in its evidence reference.
|
|
37
|
+
|
|
7
38
|
## [0.116.0] — 2026-07-12 — evidence-linked AgentProfile optimization
|
|
8
39
|
|
|
9
40
|
### Added
|
package/dist/analyst/index.d.ts
CHANGED
|
@@ -1,21 +1,24 @@
|
|
|
1
1
|
import { M as MultiLayerVerifier, V as VerifyOptions, S as Severity } from '../multi-layer-verifier-BsqKuLyN.js';
|
|
2
|
-
import { R as RunCritic, S as SemanticConceptJudgeOptions, a as RunTrace, b as SemanticConceptJudgeInput, B as BehavioralMetrics } from '../semantic-concept-judge-
|
|
3
|
-
export { C as CreateAnalystAiConfig, D as DEFAULT_TRACE_ANALYST_KINDS, c as DiffPolicy, F as FAILURE_MODE_KIND_SPEC, d as FINDING_SUBJECT_GRAMMAR_PROMPT, e as FINDING_SUBJECT_KINDS, f as FINDING_SUBJECT_SYNTAX, g as FindingSubject, h as FindingSubjectKind, i as FindingSubjectStringSchema, j as FindingsDiff, k as FindingsStore, I as IMPROVEMENT_KIND_SPEC, K as KIND_EXPECTED_SUBJECTS, l as KNOWLEDGE_GAP_KIND_SPEC, m as KNOWLEDGE_POISONING_KIND_SPEC, P as PersistedFinding, n as SKILL_USAGE_ANALYST, o as SkillUsageAnalyst, p as SkillUsageRecord, q as SkillUsageReport, r as SkillUsageScanConfig, s as buildSkillUsageReport, t as createAnalystAi, u as defaultIsMaterial, v as diffFindings, w as emitSkillUsageFindings, x as findingSubjectGrammarPromptFor, y as parseFindingSubject, z as renderFindingSubject } from '../semantic-concept-judge-
|
|
4
|
-
import { b as JudgeFn, a as JudgeInput } from '../types-
|
|
5
|
-
import { A as Analyst,
|
|
6
|
-
export {
|
|
2
|
+
import { R as RunCritic, S as SemanticConceptJudgeOptions, a as RunTrace, b as SemanticConceptJudgeInput, B as BehavioralMetrics } from '../semantic-concept-judge-CXnPEJbf.js';
|
|
3
|
+
export { C as CreateAnalystAiConfig, D as DEFAULT_TRACE_ANALYST_KINDS, c as DiffPolicy, F as FAILURE_MODE_KIND_SPEC, d as FINDING_SUBJECT_GRAMMAR_PROMPT, e as FINDING_SUBJECT_KINDS, f as FINDING_SUBJECT_SYNTAX, g as FindingSubject, h as FindingSubjectKind, i as FindingSubjectStringSchema, j as FindingsDiff, k as FindingsStore, I as IMPROVEMENT_KIND_SPEC, K as KIND_EXPECTED_SUBJECTS, l as KNOWLEDGE_GAP_KIND_SPEC, m as KNOWLEDGE_POISONING_KIND_SPEC, P as PersistedFinding, n as SKILL_USAGE_ANALYST, o as SkillUsageAnalyst, p as SkillUsageRecord, q as SkillUsageReport, r as SkillUsageScanConfig, s as buildSkillUsageReport, t as createAnalystAi, u as defaultIsMaterial, v as diffFindings, w as emitSkillUsageFindings, x as findingSubjectGrammarPromptFor, y as parseFindingSubject, z as renderFindingSubject } from '../semantic-concept-judge-CXnPEJbf.js';
|
|
4
|
+
import { b as JudgeFn, a as JudgeInput } from '../types-BkfcQnxV.js';
|
|
5
|
+
import { A as Analyst, h as AnalystSeverity, c as AnalystFinding } from '../policy-edit-wG9uFEFm.js';
|
|
6
|
+
export { a as AnalystContext, g as AnalystCost, i as AnalystInputKind, j as AnalystRequirements, f as AnalystRunEvent, e as AnalystRunInputs, d as AnalystRunResult, b as AnalystRunSummary, k as ChatCallOpts, C as ChatClient, l as ChatRequest, m as ChatResponse, n as ChatTransport, o as CliBridgeTransportOpts, p as CreateChatClientOpts, D as DirectProviderTransportOpts, E as EvidenceRef, F as FindingToPolicyEditOptions, M as MockTransportOpts, q as POLICY_EDIT_AXES, r as POLICY_EDIT_CANDIDATE_RECORD_SCHEMA, s as POLICY_EDIT_TARGET_SURFACES, t as PolicyEdit, u as PolicyEditAdmission, v as PolicyEditAdmissionOptions, w as PolicyEditAxis, P as PolicyEditCandidateRecord, x as PolicyEditChange, y as PolicyEditExpectedGain, z as PolicyEditGainDirection, B as PolicyEditGainUnit, G as PolicyEditInit, H as PolicyEditRisk, I as PolicyEditSchemaVersion, J as PolicyEditSource, K as PolicyEditTarget, L as PolicyEditTargetSurface, N as PolicyEditValidationError, R as RouterTransportOpts, S as SandboxSdkTransportOpts, O as admitPolicyEdit, Q as applyPolicyEditToSurface, T as computeFindingId, U as computePolicyEditId, V as createChatClient, W as isPolicyEdit, X as makeFinding, Y as makePolicyEdit, Z as makePolicyEditCandidateRecord, _ as policyEditFromFinding, $ as policyEditsFromFindings, a0 as scorePolicyEditReadiness, a1 as validatePolicyEdit, a2 as validatePolicyEditCandidateRecord } from '../policy-edit-wG9uFEFm.js';
|
|
7
7
|
import { TCloud } from '@tangle-network/tcloud';
|
|
8
|
-
import { T as TraceAnalysisStore } from '../store-
|
|
9
|
-
export { a as AnalystHooks, A as AnalystRegistry, b as AnalystRegistryOptions, B as BudgetPolicy, D as DefaultAnalystRegistryOptions, R as RegistryRunOpts, c as buildDefaultAnalystRegistry } from '../default-registry-
|
|
10
|
-
export { A as ANALYST_SEVERITIES, C as CreateTraceAnalystKindOpts, R as RAW_FINDING_SCHEMA_PROMPT, a as RawAnalystFinding, b as RawAnalystFindingSchema, c as TraceAnalystGolden, T as TraceAnalystKindSpec, d as createTraceAnalystKind, p as parseRawFinding, r as renderPriorFindings } from '../kind-factory-
|
|
8
|
+
import { T as TraceAnalysisStore } from '../store-C1YxJDEK.js';
|
|
9
|
+
export { a as AnalystHooks, A as AnalystRegistry, b as AnalystRegistryOptions, B as BudgetPolicy, D as DefaultAnalystRegistryOptions, R as RegistryRunOpts, c as buildDefaultAnalystRegistry } from '../default-registry-DaK8b3fv.js';
|
|
10
|
+
export { A as ANALYST_SEVERITIES, C as CreateTraceAnalystKindOpts, R as RAW_FINDING_SCHEMA_PROMPT, a as RawAnalystFinding, b as RawAnalystFindingSchema, c as TraceAnalystGolden, T as TraceAnalystKindSpec, d as createTraceAnalystKind, p as parseRawFinding, r as renderPriorFindings } from '../kind-factory-ClZmO25A.js';
|
|
11
|
+
import { C as CostLedger } from '../cost-ledger-DWy3XdJc.js';
|
|
12
|
+
import { a as LlmClientOptions } from '../llm-client-qoDd18Qz.js';
|
|
11
13
|
import { AxFunction } from '@ax-llm/ax';
|
|
12
14
|
import '../verdict-C9MlYujm.js';
|
|
13
15
|
import 'zod';
|
|
14
|
-
import '../schema-
|
|
15
|
-
import '../store-
|
|
16
|
-
import '../run-record-
|
|
16
|
+
import '../schema-B3Q3l9Z_.js';
|
|
17
|
+
import '../store-DGqD0Pyo.js';
|
|
18
|
+
import '../run-record-BDH49H2E.js';
|
|
17
19
|
import '@tangle-network/agent-interface';
|
|
18
20
|
import '../errors-oeQrLqXC.js';
|
|
21
|
+
import '../raw-provider-sink-C46HDghv.js';
|
|
19
22
|
|
|
20
23
|
/**
|
|
21
24
|
* Adapter factories — lift each existing agent-eval primitive into the
|
|
@@ -186,6 +189,10 @@ interface StructureFindingsOptions {
|
|
|
186
189
|
model: string;
|
|
187
190
|
baseUrl: string;
|
|
188
191
|
apiKey?: string;
|
|
192
|
+
/** Optional ledger for direct use. */
|
|
193
|
+
costLedger?: CostLedger;
|
|
194
|
+
costPhase?: string;
|
|
195
|
+
maxTokens?: number;
|
|
189
196
|
/** Max reask attempts after a zero/invalid extraction. Default 1. */
|
|
190
197
|
maxReasks?: number;
|
|
191
198
|
/** Test seam: inject a fetch (no network in unit tests). */
|
package/dist/analyst/index.js
CHANGED
|
@@ -6,18 +6,19 @@ import {
|
|
|
6
6
|
SkillUsageAnalyst,
|
|
7
7
|
buildSkillUsageReport,
|
|
8
8
|
createAnalystAi,
|
|
9
|
-
createChatClient,
|
|
10
9
|
defaultIsMaterial,
|
|
11
10
|
diffFindings,
|
|
12
11
|
emitSkillUsageFindings,
|
|
13
12
|
runSemanticConceptJudge
|
|
14
|
-
} from "../chunk-
|
|
13
|
+
} from "../chunk-CCZIVI3F.js";
|
|
15
14
|
import {
|
|
16
15
|
behavioralAnalyst,
|
|
17
16
|
buildDefaultAnalystRegistry,
|
|
17
|
+
createChatClient,
|
|
18
18
|
deriveEfficiencyFindings
|
|
19
|
-
} from "../chunk-
|
|
20
|
-
import "../chunk-
|
|
19
|
+
} from "../chunk-VF3XSYTI.js";
|
|
20
|
+
import "../chunk-HHWE3POT.js";
|
|
21
|
+
import "../chunk-3YYRZDON.js";
|
|
21
22
|
import {
|
|
22
23
|
ANALYST_SEVERITIES,
|
|
23
24
|
AnalystRegistry,
|
|
@@ -64,11 +65,13 @@ import {
|
|
|
64
65
|
structureFindings,
|
|
65
66
|
validatePolicyEdit,
|
|
66
67
|
validatePolicyEditCandidateRecord
|
|
67
|
-
} from "../chunk-
|
|
68
|
-
import "../chunk-
|
|
68
|
+
} from "../chunk-MGEHEHSN.js";
|
|
69
|
+
import "../chunk-S2F4J57L.js";
|
|
69
70
|
import "../chunk-XJYR7XFV.js";
|
|
70
71
|
import "../chunk-VSMTAMNK.js";
|
|
71
|
-
import "../chunk-
|
|
72
|
+
import "../chunk-NJC7U437.js";
|
|
73
|
+
import "../chunk-VCTY3W6J.js";
|
|
74
|
+
import "../chunk-VI2UW6B6.js";
|
|
72
75
|
import "../chunk-PC4UYEBM.js";
|
|
73
76
|
import "../chunk-ONWEPEDO.js";
|
|
74
77
|
import "../chunk-PZ5AY32C.js";
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"sources":["../../src/analyst/adapters.ts"],"sourcesContent":["/**\n * Adapter factories — lift each existing agent-eval primitive into the\n * Analyst contract without re-implementing it.\n *\n * Five primitives, five factories. Each one:\n * - Builds an Analyst with a stable id (caller chooses; defaults\n * given), a sensible default `inputKind`, a version derived from\n * the wrapped primitive's version + an adapter revision, and an\n * `analyze()` that calls the primitive and lifts its output to\n * AnalystFinding[] using `makeFinding()`.\n * - Maps severities: the existing `Severity` ('critical' | 'major' |\n * 'minor' | 'info') projects onto AnalystSeverity ('critical' |\n * 'high' | 'medium' | 'low' | 'info'); 'major' → 'high', 'minor' →\n * 'medium'. Domain analysts that want finer-grained mapping override.\n *\n * Adapters never own state. Calling the same factory twice with the\n * same primitive instance is safe.\n */\n\nimport type {\n Finding as LayerFinding,\n Severity as LayerSeverity,\n MultiLayerVerifier,\n VerifyOptions,\n} from '../multi-layer-verifier'\nimport { RunCritic, type RunTrace } from '../run-critic'\nimport {\n runSemanticConceptJudge,\n SEMANTIC_CONCEPT_JUDGE_VERSION,\n type SemanticConceptJudgeInput,\n type SemanticConceptJudgeOptions,\n} from '../semantic-concept-judge'\nimport type { JudgeFn, JudgeInput, JudgeScore, TCloud } from '../types'\nimport type { Analyst, AnalystFinding, AnalystSeverity } from './types'\nimport { makeFinding } from './types'\n\nconst ADAPTER_REV = '1'\n\n// ── Severity bridges ───────────────────────────────────────────────\n\nexport function liftSeverity(s: LayerSeverity): AnalystSeverity {\n switch (s) {\n case 'critical':\n return 'critical'\n case 'major':\n return 'high'\n case 'minor':\n return 'medium'\n case 'info':\n return 'info'\n }\n}\n\n// ── 1. MultiLayerVerifier → Analyst ─────────────────────────────────\n\nexport interface VerifierAdapterOpts<Env> {\n id?: string\n area?: string\n verifier: MultiLayerVerifier<Env>\n /**\n * The verifier expects an `env` per run. Adapters take it from\n * `AnalystRunInputs.custom[<id>]` via the registry's 'custom' routing.\n */\n options?: Omit<VerifyOptions<Env>, 'env'>\n}\n\nexport function createVerifierAdapter<Env>(opts: VerifierAdapterOpts<Env>): Analyst<Env> {\n const id = opts.id ?? 'multi-layer-verifier'\n const area = opts.area ?? 'verification'\n return {\n id,\n description:\n \"Runs a MultiLayerVerifier and lifts each layer's findings into the analyst envelope.\",\n inputKind: 'custom',\n cost: { kind: 'deterministic' },\n version: `verifier-${ADAPTER_REV}`,\n async analyze(env, ctx) {\n const report = await opts.verifier.run({ env, ...opts.options })\n const out: AnalystFinding[] = []\n for (const layer of report.layers) {\n for (const finding of layer.findings) {\n out.push(liftLayerFinding(id, area, layer.layer, finding))\n }\n // Layer-level signal: a failed/error layer is itself a finding\n // even if it didn't emit per-finding rows.\n if (layer.status === 'fail' || layer.status === 'error' || layer.status === 'timeout') {\n out.push(\n makeFinding({\n analyst_id: id,\n area,\n subject: layer.layer,\n claim: `layer \"${layer.layer}\" ${layer.status}: ${layer.reason ?? 'no reason given'}`,\n severity:\n layer.status === 'error' ? 'high' : layer.status === 'timeout' ? 'medium' : 'high',\n confidence: 1,\n evidence_refs: [],\n metadata: {\n layer_status: layer.status,\n duration_ms: layer.durationMs,\n score: layer.score,\n diagnostics: layer.diagnostics,\n },\n }),\n )\n }\n }\n ctx.log?.('verifier complete', {\n layers: report.layers.length,\n blended: report.blendedScore,\n all_pass: report.allPass,\n })\n return out\n },\n }\n}\n\nfunction liftLayerFinding(\n analyst_id: string,\n area: string,\n layer: string,\n f: LayerFinding,\n): AnalystFinding {\n return makeFinding({\n analyst_id,\n area,\n subject: f.layer ?? layer,\n claim: f.message,\n severity: liftSeverity(f.severity),\n confidence: 0.85,\n evidence_refs: f.evidence\n ? [{ kind: 'artifact', uri: 'inline:evidence', excerpt: f.evidence }]\n : [],\n metadata: f.detail,\n })\n}\n\n// ── 2. RunCritic → Analyst ──────────────────────────────────────────\n\nexport interface RunCriticAdapterOpts {\n id?: string\n area?: string\n critic?: RunCritic\n /** Optional threshold below which a dimension is reported as a finding. Default 0.5. */\n threshold?: number\n}\n\nexport function createRunCriticAdapter(opts: RunCriticAdapterOpts = {}): Analyst<RunTrace> {\n const id = opts.id ?? 'run-critic'\n const area = opts.area ?? 'run-quality'\n const critic = opts.critic ?? new RunCritic()\n const threshold = opts.threshold ?? 0.5\n return {\n id,\n description:\n 'Scores a single run across success / grounding / drift / tool-quality and surfaces below-threshold dimensions.',\n inputKind: 'custom',\n cost: { kind: 'deterministic' },\n version: `run-critic-${ADAPTER_REV}`,\n async analyze(trace) {\n const score = critic.scoreTrace(trace)\n const out: AnalystFinding[] = []\n const dims: Array<[keyof typeof score, AnalystSeverity, string]> = [\n ['success', 'critical', 'run did not complete successfully'],\n ['goalProgress', 'high', 'goal progress is low'],\n ['repoGroundedness', 'high', 'output is poorly grounded in the repository'],\n ['toolUseQuality', 'medium', 'tool use quality is low'],\n ['patchQuality', 'medium', 'no real patch/edit evidence'],\n ['testReality', 'high', 'no real test/build evidence'],\n ['finalGate', 'critical', 'final gate is blocking'],\n ]\n for (const [dim, sev, msg] of dims) {\n const value = score[dim] as number\n if (typeof value === 'number' && value < threshold) {\n out.push(\n makeFinding({\n analyst_id: id,\n area,\n subject: dim,\n claim: msg,\n rationale: `${dim}=${value.toFixed(2)} below threshold ${threshold}`,\n severity: sev,\n confidence: 1,\n evidence_refs: [],\n metadata: { dimension: dim, value, threshold, run_id: trace.run.runId },\n }),\n )\n }\n }\n // Drift penalty is high → surface as a finding (inverse threshold).\n if (score.driftPenalty > 1 - threshold) {\n out.push(\n makeFinding({\n analyst_id: id,\n area,\n subject: 'drift',\n claim: 'agent output drifted from repository signal',\n rationale: `driftPenalty=${score.driftPenalty.toFixed(2)}`,\n severity: 'medium',\n confidence: 0.9,\n evidence_refs: [],\n metadata: { drift_penalty: score.driftPenalty, notes: score.notes },\n }),\n )\n }\n return out\n },\n }\n}\n\n// ── 3. JudgeFn → Analyst ────────────────────────────────────────────\n\nexport interface JudgeAdapterOpts {\n id?: string\n area?: string\n judge: JudgeFn\n /** TCloud handle the JudgeFn calls. */\n tcloud: TCloud\n /** Optional cost classification — most judges call an LLM. */\n cost?: Analyst['cost']\n /** Optional threshold below which a JudgeScore becomes a finding. Default 6 (on 0-10 scale). */\n threshold?: number\n}\n\nexport function createJudgeAdapter(opts: JudgeAdapterOpts): Analyst<JudgeInput> {\n const id = opts.id ?? 'judge'\n const area = opts.area ?? 'judge'\n const threshold = opts.threshold ?? 6\n return {\n id,\n description:\n 'Wraps an agent-eval JudgeFn into an analyst; below-threshold dimensions surface as findings.',\n inputKind: 'judge-input',\n cost: opts.cost ?? { kind: 'llm' },\n version: `judge-${ADAPTER_REV}`,\n async analyze(input) {\n const scores = await opts.judge(opts.tcloud, input)\n return scores\n .filter((s) => normalize10(s.score) < threshold)\n .map((s) => liftJudgeScore(id, area, s))\n },\n }\n}\n\nfunction normalize10(s: number): number {\n // JudgeScore convention is 0-10 but some judges emit 0-1. Coerce to 0-10.\n return s <= 1 ? s * 10 : s\n}\n\nfunction liftJudgeScore(analyst_id: string, area: string, s: JudgeScore): AnalystFinding {\n const score10 = normalize10(s.score)\n const severity: AnalystSeverity =\n score10 < 3 ? 'critical' : score10 < 5 ? 'high' : score10 < 7 ? 'medium' : 'low'\n return makeFinding({\n analyst_id,\n area,\n subject: s.dimension,\n claim: `${s.judgeName}/${s.dimension} scored ${score10.toFixed(1)}/10`,\n rationale: s.reasoning,\n severity,\n confidence: 0.8,\n evidence_refs: s.evidence\n ? [{ kind: 'artifact', uri: 'inline:evidence', excerpt: s.evidence }]\n : [],\n // Provenance: this finding IS a judge verdict (an acceptance score), not an\n // observation of behavior. The steer firewall (assertNoJudgeVerdict) rejects\n // it from steering — even when it cites an artifact above — because letting a\n // verdict steer the next attempt is the held-out judge leaking into the loop.\n derived_from_judge: true,\n metadata: { judge_name: s.judgeName, dimension: s.dimension, score_10: score10 },\n })\n}\n\n// ── 4. SemanticConceptJudge → Analyst ──────────────────────────────\n\nexport interface SemanticConceptJudgeAdapterOpts {\n id?: string\n area?: string\n options?: SemanticConceptJudgeOptions\n}\n\nexport function createSemanticConceptJudgeAdapter(\n opts: SemanticConceptJudgeAdapterOpts = {},\n): Analyst<SemanticConceptJudgeInput> {\n const id = opts.id ?? 'semantic-concept-judge'\n const area = opts.area ?? 'concept-coverage'\n return {\n id,\n description:\n 'Runs the semantic-concept judge and surfaces missing / weak concepts as findings.',\n inputKind: 'custom',\n cost: { kind: 'llm', models: opts.options?.model ? [opts.options.model] : undefined },\n version: `${SEMANTIC_CONCEPT_JUDGE_VERSION}-adapter-${ADAPTER_REV}`,\n async analyze(input) {\n const result = await runSemanticConceptJudge(input, opts.options)\n if (!result.available) {\n return [\n makeFinding({\n analyst_id: id,\n area,\n claim: 'semantic-concept judge unavailable',\n rationale: result.error,\n severity: 'info',\n confidence: 1,\n evidence_refs: [],\n metadata: { reason: result.error },\n }),\n ]\n }\n const out: AnalystFinding[] = []\n for (const f of result.findings) {\n // Only surface gaps: missing concepts or low scores. Concepts at\n // 7+/10 with present=true are not findings — they're successes.\n if (f.present && f.score >= 7) continue\n out.push(\n makeFinding({\n analyst_id: id,\n area,\n subject: f.concept,\n claim: f.present\n ? `concept \"${f.concept}\" is weak (${f.score}/10)`\n : `concept \"${f.concept}\" is missing`,\n rationale: f.evidence,\n severity: liftSeverity(f.severity),\n confidence: 0.85,\n evidence_refs: [{ kind: 'artifact', uri: 'inline:evidence', excerpt: f.evidence }],\n metadata: {\n concept: f.concept,\n present: f.present,\n score_10: f.score,\n cost_usd: result.costUsd ?? undefined,\n },\n }),\n )\n }\n return out\n },\n }\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AAoCA,IAAM,cAAc;AAIb,SAAS,aAAa,GAAmC;AAC9D,UAAQ,GAAG;AAAA,IACT,KAAK;AACH,aAAO;AAAA,IACT,KAAK;AACH,aAAO;AAAA,IACT,KAAK;AACH,aAAO;AAAA,IACT,KAAK;AACH,aAAO;AAAA,EACX;AACF;AAeO,SAAS,sBAA2B,MAA8C;AACvF,QAAM,KAAK,KAAK,MAAM;AACtB,QAAM,OAAO,KAAK,QAAQ;AAC1B,SAAO;AAAA,IACL;AAAA,IACA,aACE;AAAA,IACF,WAAW;AAAA,IACX,MAAM,EAAE,MAAM,gBAAgB;AAAA,IAC9B,SAAS,YAAY,WAAW;AAAA,IAChC,MAAM,QAAQ,KAAK,KAAK;AACtB,YAAM,SAAS,MAAM,KAAK,SAAS,IAAI,EAAE,KAAK,GAAG,KAAK,QAAQ,CAAC;AAC/D,YAAM,MAAwB,CAAC;AAC/B,iBAAW,SAAS,OAAO,QAAQ;AACjC,mBAAW,WAAW,MAAM,UAAU;AACpC,cAAI,KAAK,iBAAiB,IAAI,MAAM,MAAM,OAAO,OAAO,CAAC;AAAA,QAC3D;AAGA,YAAI,MAAM,WAAW,UAAU,MAAM,WAAW,WAAW,MAAM,WAAW,WAAW;AACrF,cAAI;AAAA,YACF,YAAY;AAAA,cACV,YAAY;AAAA,cACZ;AAAA,cACA,SAAS,MAAM;AAAA,cACf,OAAO,UAAU,MAAM,KAAK,KAAK,MAAM,MAAM,KAAK,MAAM,UAAU,iBAAiB;AAAA,cACnF,UACE,MAAM,WAAW,UAAU,SAAS,MAAM,WAAW,YAAY,WAAW;AAAA,cAC9E,YAAY;AAAA,cACZ,eAAe,CAAC;AAAA,cAChB,UAAU;AAAA,gBACR,cAAc,MAAM;AAAA,gBACpB,aAAa,MAAM;AAAA,gBACnB,OAAO,MAAM;AAAA,gBACb,aAAa,MAAM;AAAA,cACrB;AAAA,YACF,CAAC;AAAA,UACH;AAAA,QACF;AAAA,MACF;AACA,UAAI,MAAM,qBAAqB;AAAA,QAC7B,QAAQ,OAAO,OAAO;AAAA,QACtB,SAAS,OAAO;AAAA,QAChB,UAAU,OAAO;AAAA,MACnB,CAAC;AACD,aAAO;AAAA,IACT;AAAA,EACF;AACF;AAEA,SAAS,iBACP,YACA,MACA,OACA,GACgB;AAChB,SAAO,YAAY;AAAA,IACjB;AAAA,IACA;AAAA,IACA,SAAS,EAAE,SAAS;AAAA,IACpB,OAAO,EAAE;AAAA,IACT,UAAU,aAAa,EAAE,QAAQ;AAAA,IACjC,YAAY;AAAA,IACZ,eAAe,EAAE,WACb,CAAC,EAAE,MAAM,YAAY,KAAK,mBAAmB,SAAS,EAAE,SAAS,CAAC,IAClE,CAAC;AAAA,IACL,UAAU,EAAE;AAAA,EACd,CAAC;AACH;AAYO,SAAS,uBAAuB,OAA6B,CAAC,GAAsB;AACzF,QAAM,KAAK,KAAK,MAAM;AACtB,QAAM,OAAO,KAAK,QAAQ;AAC1B,QAAM,SAAS,KAAK,UAAU,IAAI,UAAU;AAC5C,QAAM,YAAY,KAAK,aAAa;AACpC,SAAO;AAAA,IACL;AAAA,IACA,aACE;AAAA,IACF,WAAW;AAAA,IACX,MAAM,EAAE,MAAM,gBAAgB;AAAA,IAC9B,SAAS,cAAc,WAAW;AAAA,IAClC,MAAM,QAAQ,OAAO;AACnB,YAAM,QAAQ,OAAO,WAAW,KAAK;AACrC,YAAM,MAAwB,CAAC;AAC/B,YAAM,OAA6D;AAAA,QACjE,CAAC,WAAW,YAAY,mCAAmC;AAAA,QAC3D,CAAC,gBAAgB,QAAQ,sBAAsB;AAAA,QAC/C,CAAC,oBAAoB,QAAQ,6CAA6C;AAAA,QAC1E,CAAC,kBAAkB,UAAU,yBAAyB;AAAA,QACtD,CAAC,gBAAgB,UAAU,6BAA6B;AAAA,QACxD,CAAC,eAAe,QAAQ,6BAA6B;AAAA,QACrD,CAAC,aAAa,YAAY,wBAAwB;AAAA,MACpD;AACA,iBAAW,CAAC,KAAK,KAAK,GAAG,KAAK,MAAM;AAClC,cAAM,QAAQ,MAAM,GAAG;AACvB,YAAI,OAAO,UAAU,YAAY,QAAQ,WAAW;AAClD,cAAI;AAAA,YACF,YAAY;AAAA,cACV,YAAY;AAAA,cACZ;AAAA,cACA,SAAS;AAAA,cACT,OAAO;AAAA,cACP,WAAW,GAAG,GAAG,IAAI,MAAM,QAAQ,CAAC,CAAC,oBAAoB,SAAS;AAAA,cAClE,UAAU;AAAA,cACV,YAAY;AAAA,cACZ,eAAe,CAAC;AAAA,cAChB,UAAU,EAAE,WAAW,KAAK,OAAO,WAAW,QAAQ,MAAM,IAAI,MAAM;AAAA,YACxE,CAAC;AAAA,UACH;AAAA,QACF;AAAA,MACF;AAEA,UAAI,MAAM,eAAe,IAAI,WAAW;AACtC,YAAI;AAAA,UACF,YAAY;AAAA,YACV,YAAY;AAAA,YACZ;AAAA,YACA,SAAS;AAAA,YACT,OAAO;AAAA,YACP,WAAW,gBAAgB,MAAM,aAAa,QAAQ,CAAC,CAAC;AAAA,YACxD,UAAU;AAAA,YACV,YAAY;AAAA,YACZ,eAAe,CAAC;AAAA,YAChB,UAAU,EAAE,eAAe,MAAM,cAAc,OAAO,MAAM,MAAM;AAAA,UACpE,CAAC;AAAA,QACH;AAAA,MACF;AACA,aAAO;AAAA,IACT;AAAA,EACF;AACF;AAgBO,SAAS,mBAAmB,MAA6C;AAC9E,QAAM,KAAK,KAAK,MAAM;AACtB,QAAM,OAAO,KAAK,QAAQ;AAC1B,QAAM,YAAY,KAAK,aAAa;AACpC,SAAO;AAAA,IACL;AAAA,IACA,aACE;AAAA,IACF,WAAW;AAAA,IACX,MAAM,KAAK,QAAQ,EAAE,MAAM,MAAM;AAAA,IACjC,SAAS,SAAS,WAAW;AAAA,IAC7B,MAAM,QAAQ,OAAO;AACnB,YAAM,SAAS,MAAM,KAAK,MAAM,KAAK,QAAQ,KAAK;AAClD,aAAO,OACJ,OAAO,CAAC,MAAM,YAAY,EAAE,KAAK,IAAI,SAAS,EAC9C,IAAI,CAAC,MAAM,eAAe,IAAI,MAAM,CAAC,CAAC;AAAA,IAC3C;AAAA,EACF;AACF;AAEA,SAAS,YAAY,GAAmB;AAEtC,SAAO,KAAK,IAAI,IAAI,KAAK;AAC3B;AAEA,SAAS,eAAe,YAAoB,MAAc,GAA+B;AACvF,QAAM,UAAU,YAAY,EAAE,KAAK;AACnC,QAAM,WACJ,UAAU,IAAI,aAAa,UAAU,IAAI,SAAS,UAAU,IAAI,WAAW;AAC7E,SAAO,YAAY;AAAA,IACjB;AAAA,IACA;AAAA,IACA,SAAS,EAAE;AAAA,IACX,OAAO,GAAG,EAAE,SAAS,IAAI,EAAE,SAAS,WAAW,QAAQ,QAAQ,CAAC,CAAC;AAAA,IACjE,WAAW,EAAE;AAAA,IACb;AAAA,IACA,YAAY;AAAA,IACZ,eAAe,EAAE,WACb,CAAC,EAAE,MAAM,YAAY,KAAK,mBAAmB,SAAS,EAAE,SAAS,CAAC,IAClE,CAAC;AAAA;AAAA;AAAA;AAAA;AAAA,IAKL,oBAAoB;AAAA,IACpB,UAAU,EAAE,YAAY,EAAE,WAAW,WAAW,EAAE,WAAW,UAAU,QAAQ;AAAA,EACjF,CAAC;AACH;AAUO,SAAS,kCACd,OAAwC,CAAC,GACL;AACpC,QAAM,KAAK,KAAK,MAAM;AACtB,QAAM,OAAO,KAAK,QAAQ;AAC1B,SAAO;AAAA,IACL;AAAA,IACA,aACE;AAAA,IACF,WAAW;AAAA,IACX,MAAM,EAAE,MAAM,OAAO,QAAQ,KAAK,SAAS,QAAQ,CAAC,KAAK,QAAQ,KAAK,IAAI,OAAU;AAAA,IACpF,SAAS,GAAG,8BAA8B,YAAY,WAAW;AAAA,IACjE,MAAM,QAAQ,OAAO;AACnB,YAAM,SAAS,MAAM,wBAAwB,OAAO,KAAK,OAAO;AAChE,UAAI,CAAC,OAAO,WAAW;AACrB,eAAO;AAAA,UACL,YAAY;AAAA,YACV,YAAY;AAAA,YACZ;AAAA,YACA,OAAO;AAAA,YACP,WAAW,OAAO;AAAA,YAClB,UAAU;AAAA,YACV,YAAY;AAAA,YACZ,eAAe,CAAC;AAAA,YAChB,UAAU,EAAE,QAAQ,OAAO,MAAM;AAAA,UACnC,CAAC;AAAA,QACH;AAAA,MACF;AACA,YAAM,MAAwB,CAAC;AAC/B,iBAAW,KAAK,OAAO,UAAU;AAG/B,YAAI,EAAE,WAAW,EAAE,SAAS,EAAG;AAC/B,YAAI;AAAA,UACF,YAAY;AAAA,YACV,YAAY;AAAA,YACZ;AAAA,YACA,SAAS,EAAE;AAAA,YACX,OAAO,EAAE,UACL,YAAY,EAAE,OAAO,cAAc,EAAE,KAAK,SAC1C,YAAY,EAAE,OAAO;AAAA,YACzB,WAAW,EAAE;AAAA,YACb,UAAU,aAAa,EAAE,QAAQ;AAAA,YACjC,YAAY;AAAA,YACZ,eAAe,CAAC,EAAE,MAAM,YAAY,KAAK,mBAAmB,SAAS,EAAE,SAAS,CAAC;AAAA,YACjF,UAAU;AAAA,cACR,SAAS,EAAE;AAAA,cACX,SAAS,EAAE;AAAA,cACX,UAAU,EAAE;AAAA,cACZ,UAAU,OAAO,WAAW;AAAA,YAC9B;AAAA,UACF,CAAC;AAAA,QACH;AAAA,MACF;AACA,aAAO;AAAA,IACT;AAAA,EACF;AACF;","names":[]}
|
|
1
|
+
{"version":3,"sources":["../../src/analyst/adapters.ts"],"sourcesContent":["/**\n * Adapter factories — lift each existing agent-eval primitive into the\n * Analyst contract without re-implementing it.\n *\n * Five primitives, five factories. Each one:\n * - Builds an Analyst with a stable id (caller chooses; defaults\n * given), a sensible default `inputKind`, a version derived from\n * the wrapped primitive's version + an adapter revision, and an\n * `analyze()` that calls the primitive and lifts its output to\n * AnalystFinding[] using `makeFinding()`.\n * - Maps severities: the existing `Severity` ('critical' | 'major' |\n * 'minor' | 'info') projects onto AnalystSeverity ('critical' |\n * 'high' | 'medium' | 'low' | 'info'); 'major' → 'high', 'minor' →\n * 'medium'. Domain analysts that want finer-grained mapping override.\n *\n * Adapters never own state. Calling the same factory twice with the\n * same primitive instance is safe.\n */\n\nimport type {\n Finding as LayerFinding,\n Severity as LayerSeverity,\n MultiLayerVerifier,\n VerifyOptions,\n} from '../multi-layer-verifier'\nimport { RunCritic, type RunTrace } from '../run-critic'\nimport {\n runSemanticConceptJudge,\n SEMANTIC_CONCEPT_JUDGE_VERSION,\n type SemanticConceptJudgeInput,\n type SemanticConceptJudgeOptions,\n} from '../semantic-concept-judge'\nimport type { JudgeFn, JudgeInput, JudgeScore, TCloud } from '../types'\nimport type { Analyst, AnalystFinding, AnalystSeverity } from './types'\nimport { makeFinding } from './types'\n\nconst ADAPTER_REV = '1'\n\n// ── Severity bridges ───────────────────────────────────────────────\n\nexport function liftSeverity(s: LayerSeverity): AnalystSeverity {\n switch (s) {\n case 'critical':\n return 'critical'\n case 'major':\n return 'high'\n case 'minor':\n return 'medium'\n case 'info':\n return 'info'\n }\n}\n\n// ── 1. MultiLayerVerifier → Analyst ─────────────────────────────────\n\nexport interface VerifierAdapterOpts<Env> {\n id?: string\n area?: string\n verifier: MultiLayerVerifier<Env>\n /**\n * The verifier expects an `env` per run. Adapters take it from\n * `AnalystRunInputs.custom[<id>]` via the registry's 'custom' routing.\n */\n options?: Omit<VerifyOptions<Env>, 'env'>\n}\n\nexport function createVerifierAdapter<Env>(opts: VerifierAdapterOpts<Env>): Analyst<Env> {\n const id = opts.id ?? 'multi-layer-verifier'\n const area = opts.area ?? 'verification'\n return {\n id,\n description:\n \"Runs a MultiLayerVerifier and lifts each layer's findings into the analyst envelope.\",\n inputKind: 'custom',\n cost: { kind: 'deterministic' },\n version: `verifier-${ADAPTER_REV}`,\n async analyze(env, ctx) {\n const report = await opts.verifier.run({ env, ...opts.options })\n const out: AnalystFinding[] = []\n for (const layer of report.layers) {\n for (const finding of layer.findings) {\n out.push(liftLayerFinding(id, area, layer.layer, finding))\n }\n // Layer-level signal: a failed/error layer is itself a finding\n // even if it didn't emit per-finding rows.\n if (layer.status === 'fail' || layer.status === 'error' || layer.status === 'timeout') {\n out.push(\n makeFinding({\n analyst_id: id,\n area,\n subject: layer.layer,\n claim: `layer \"${layer.layer}\" ${layer.status}: ${layer.reason ?? 'no reason given'}`,\n severity:\n layer.status === 'error' ? 'high' : layer.status === 'timeout' ? 'medium' : 'high',\n confidence: 1,\n evidence_refs: [],\n metadata: {\n layer_status: layer.status,\n duration_ms: layer.durationMs,\n score: layer.score,\n diagnostics: layer.diagnostics,\n },\n }),\n )\n }\n }\n ctx.log?.('verifier complete', {\n layers: report.layers.length,\n blended: report.blendedScore,\n all_pass: report.allPass,\n })\n return out\n },\n }\n}\n\nfunction liftLayerFinding(\n analyst_id: string,\n area: string,\n layer: string,\n f: LayerFinding,\n): AnalystFinding {\n return makeFinding({\n analyst_id,\n area,\n subject: f.layer ?? layer,\n claim: f.message,\n severity: liftSeverity(f.severity),\n confidence: 0.85,\n evidence_refs: f.evidence\n ? [{ kind: 'artifact', uri: 'inline:evidence', excerpt: f.evidence }]\n : [],\n metadata: f.detail,\n })\n}\n\n// ── 2. RunCritic → Analyst ──────────────────────────────────────────\n\nexport interface RunCriticAdapterOpts {\n id?: string\n area?: string\n critic?: RunCritic\n /** Optional threshold below which a dimension is reported as a finding. Default 0.5. */\n threshold?: number\n}\n\nexport function createRunCriticAdapter(opts: RunCriticAdapterOpts = {}): Analyst<RunTrace> {\n const id = opts.id ?? 'run-critic'\n const area = opts.area ?? 'run-quality'\n const critic = opts.critic ?? new RunCritic()\n const threshold = opts.threshold ?? 0.5\n return {\n id,\n description:\n 'Scores a single run across success / grounding / drift / tool-quality and surfaces below-threshold dimensions.',\n inputKind: 'custom',\n cost: { kind: 'deterministic' },\n version: `run-critic-${ADAPTER_REV}`,\n async analyze(trace) {\n const score = critic.scoreTrace(trace)\n const out: AnalystFinding[] = []\n const dims: Array<[keyof typeof score, AnalystSeverity, string]> = [\n ['success', 'critical', 'run did not complete successfully'],\n ['goalProgress', 'high', 'goal progress is low'],\n ['repoGroundedness', 'high', 'output is poorly grounded in the repository'],\n ['toolUseQuality', 'medium', 'tool use quality is low'],\n ['patchQuality', 'medium', 'no real patch/edit evidence'],\n ['testReality', 'high', 'no real test/build evidence'],\n ['finalGate', 'critical', 'final gate is blocking'],\n ]\n for (const [dim, sev, msg] of dims) {\n const value = score[dim] as number\n if (typeof value === 'number' && value < threshold) {\n out.push(\n makeFinding({\n analyst_id: id,\n area,\n subject: dim,\n claim: msg,\n rationale: `${dim}=${value.toFixed(2)} below threshold ${threshold}`,\n severity: sev,\n confidence: 1,\n evidence_refs: [],\n metadata: { dimension: dim, value, threshold, run_id: trace.run.runId },\n }),\n )\n }\n }\n // Drift penalty is high → surface as a finding (inverse threshold).\n if (score.driftPenalty > 1 - threshold) {\n out.push(\n makeFinding({\n analyst_id: id,\n area,\n subject: 'drift',\n claim: 'agent output drifted from repository signal',\n rationale: `driftPenalty=${score.driftPenalty.toFixed(2)}`,\n severity: 'medium',\n confidence: 0.9,\n evidence_refs: [],\n metadata: { drift_penalty: score.driftPenalty, notes: score.notes },\n }),\n )\n }\n return out\n },\n }\n}\n\n// ── 3. JudgeFn → Analyst ────────────────────────────────────────────\n\nexport interface JudgeAdapterOpts {\n id?: string\n area?: string\n judge: JudgeFn\n /** TCloud handle the JudgeFn calls. */\n tcloud: TCloud\n /** Optional cost classification — most judges call an LLM. */\n cost?: Analyst['cost']\n /** Optional threshold below which a JudgeScore becomes a finding. Default 6 (on 0-10 scale). */\n threshold?: number\n}\n\nexport function createJudgeAdapter(opts: JudgeAdapterOpts): Analyst<JudgeInput> {\n const id = opts.id ?? 'judge'\n const area = opts.area ?? 'judge'\n const threshold = opts.threshold ?? 6\n return {\n id,\n description:\n 'Wraps an agent-eval JudgeFn into an analyst; below-threshold dimensions surface as findings.',\n inputKind: 'judge-input',\n cost: opts.cost ?? { kind: 'llm' },\n version: `judge-${ADAPTER_REV}`,\n async analyze(input) {\n const scores = await opts.judge(opts.tcloud, input)\n return scores\n .filter((s) => normalize10(s.score) < threshold)\n .map((s) => liftJudgeScore(id, area, s))\n },\n }\n}\n\nfunction normalize10(s: number): number {\n // JudgeScore convention is 0-10 but some judges emit 0-1. Coerce to 0-10.\n return s <= 1 ? s * 10 : s\n}\n\nfunction liftJudgeScore(analyst_id: string, area: string, s: JudgeScore): AnalystFinding {\n const score10 = normalize10(s.score)\n const severity: AnalystSeverity =\n score10 < 3 ? 'critical' : score10 < 5 ? 'high' : score10 < 7 ? 'medium' : 'low'\n return makeFinding({\n analyst_id,\n area,\n subject: s.dimension,\n claim: `${s.judgeName}/${s.dimension} scored ${score10.toFixed(1)}/10`,\n rationale: s.reasoning,\n severity,\n confidence: 0.8,\n evidence_refs: s.evidence\n ? [{ kind: 'artifact', uri: 'inline:evidence', excerpt: s.evidence }]\n : [],\n // Provenance: this finding IS a judge verdict (an acceptance score), not an\n // observation of behavior. The steer firewall (assertNoJudgeVerdict) rejects\n // it from steering — even when it cites an artifact above — because letting a\n // verdict steer the next attempt is the held-out judge leaking into the loop.\n derived_from_judge: true,\n metadata: { judge_name: s.judgeName, dimension: s.dimension, score_10: score10 },\n })\n}\n\n// ── 4. SemanticConceptJudge → Analyst ──────────────────────────────\n\nexport interface SemanticConceptJudgeAdapterOpts {\n id?: string\n area?: string\n options?: SemanticConceptJudgeOptions\n}\n\nexport function createSemanticConceptJudgeAdapter(\n opts: SemanticConceptJudgeAdapterOpts = {},\n): Analyst<SemanticConceptJudgeInput> {\n const id = opts.id ?? 'semantic-concept-judge'\n const area = opts.area ?? 'concept-coverage'\n return {\n id,\n description:\n 'Runs the semantic-concept judge and surfaces missing / weak concepts as findings.',\n inputKind: 'custom',\n cost: { kind: 'llm', models: opts.options?.model ? [opts.options.model] : undefined },\n version: `${SEMANTIC_CONCEPT_JUDGE_VERSION}-adapter-${ADAPTER_REV}`,\n async analyze(input) {\n const result = await runSemanticConceptJudge(input, opts.options)\n if (!result.available) {\n return [\n makeFinding({\n analyst_id: id,\n area,\n claim: 'semantic-concept judge unavailable',\n rationale: result.error,\n severity: 'info',\n confidence: 1,\n evidence_refs: [],\n metadata: { reason: result.error },\n }),\n ]\n }\n const out: AnalystFinding[] = []\n for (const f of result.findings) {\n // Only surface gaps: missing concepts or low scores. Concepts at\n // 7+/10 with present=true are not findings — they're successes.\n if (f.present && f.score >= 7) continue\n out.push(\n makeFinding({\n analyst_id: id,\n area,\n subject: f.concept,\n claim: f.present\n ? `concept \"${f.concept}\" is weak (${f.score}/10)`\n : `concept \"${f.concept}\" is missing`,\n rationale: f.evidence,\n severity: liftSeverity(f.severity),\n confidence: 0.85,\n evidence_refs: [{ kind: 'artifact', uri: 'inline:evidence', excerpt: f.evidence }],\n metadata: {\n concept: f.concept,\n present: f.present,\n score_10: f.score,\n cost_usd: result.costUsd ?? undefined,\n },\n }),\n )\n }\n return out\n },\n }\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AAoCA,IAAM,cAAc;AAIb,SAAS,aAAa,GAAmC;AAC9D,UAAQ,GAAG;AAAA,IACT,KAAK;AACH,aAAO;AAAA,IACT,KAAK;AACH,aAAO;AAAA,IACT,KAAK;AACH,aAAO;AAAA,IACT,KAAK;AACH,aAAO;AAAA,EACX;AACF;AAeO,SAAS,sBAA2B,MAA8C;AACvF,QAAM,KAAK,KAAK,MAAM;AACtB,QAAM,OAAO,KAAK,QAAQ;AAC1B,SAAO;AAAA,IACL;AAAA,IACA,aACE;AAAA,IACF,WAAW;AAAA,IACX,MAAM,EAAE,MAAM,gBAAgB;AAAA,IAC9B,SAAS,YAAY,WAAW;AAAA,IAChC,MAAM,QAAQ,KAAK,KAAK;AACtB,YAAM,SAAS,MAAM,KAAK,SAAS,IAAI,EAAE,KAAK,GAAG,KAAK,QAAQ,CAAC;AAC/D,YAAM,MAAwB,CAAC;AAC/B,iBAAW,SAAS,OAAO,QAAQ;AACjC,mBAAW,WAAW,MAAM,UAAU;AACpC,cAAI,KAAK,iBAAiB,IAAI,MAAM,MAAM,OAAO,OAAO,CAAC;AAAA,QAC3D;AAGA,YAAI,MAAM,WAAW,UAAU,MAAM,WAAW,WAAW,MAAM,WAAW,WAAW;AACrF,cAAI;AAAA,YACF,YAAY;AAAA,cACV,YAAY;AAAA,cACZ;AAAA,cACA,SAAS,MAAM;AAAA,cACf,OAAO,UAAU,MAAM,KAAK,KAAK,MAAM,MAAM,KAAK,MAAM,UAAU,iBAAiB;AAAA,cACnF,UACE,MAAM,WAAW,UAAU,SAAS,MAAM,WAAW,YAAY,WAAW;AAAA,cAC9E,YAAY;AAAA,cACZ,eAAe,CAAC;AAAA,cAChB,UAAU;AAAA,gBACR,cAAc,MAAM;AAAA,gBACpB,aAAa,MAAM;AAAA,gBACnB,OAAO,MAAM;AAAA,gBACb,aAAa,MAAM;AAAA,cACrB;AAAA,YACF,CAAC;AAAA,UACH;AAAA,QACF;AAAA,MACF;AACA,UAAI,MAAM,qBAAqB;AAAA,QAC7B,QAAQ,OAAO,OAAO;AAAA,QACtB,SAAS,OAAO;AAAA,QAChB,UAAU,OAAO;AAAA,MACnB,CAAC;AACD,aAAO;AAAA,IACT;AAAA,EACF;AACF;AAEA,SAAS,iBACP,YACA,MACA,OACA,GACgB;AAChB,SAAO,YAAY;AAAA,IACjB;AAAA,IACA;AAAA,IACA,SAAS,EAAE,SAAS;AAAA,IACpB,OAAO,EAAE;AAAA,IACT,UAAU,aAAa,EAAE,QAAQ;AAAA,IACjC,YAAY;AAAA,IACZ,eAAe,EAAE,WACb,CAAC,EAAE,MAAM,YAAY,KAAK,mBAAmB,SAAS,EAAE,SAAS,CAAC,IAClE,CAAC;AAAA,IACL,UAAU,EAAE;AAAA,EACd,CAAC;AACH;AAYO,SAAS,uBAAuB,OAA6B,CAAC,GAAsB;AACzF,QAAM,KAAK,KAAK,MAAM;AACtB,QAAM,OAAO,KAAK,QAAQ;AAC1B,QAAM,SAAS,KAAK,UAAU,IAAI,UAAU;AAC5C,QAAM,YAAY,KAAK,aAAa;AACpC,SAAO;AAAA,IACL;AAAA,IACA,aACE;AAAA,IACF,WAAW;AAAA,IACX,MAAM,EAAE,MAAM,gBAAgB;AAAA,IAC9B,SAAS,cAAc,WAAW;AAAA,IAClC,MAAM,QAAQ,OAAO;AACnB,YAAM,QAAQ,OAAO,WAAW,KAAK;AACrC,YAAM,MAAwB,CAAC;AAC/B,YAAM,OAA6D;AAAA,QACjE,CAAC,WAAW,YAAY,mCAAmC;AAAA,QAC3D,CAAC,gBAAgB,QAAQ,sBAAsB;AAAA,QAC/C,CAAC,oBAAoB,QAAQ,6CAA6C;AAAA,QAC1E,CAAC,kBAAkB,UAAU,yBAAyB;AAAA,QACtD,CAAC,gBAAgB,UAAU,6BAA6B;AAAA,QACxD,CAAC,eAAe,QAAQ,6BAA6B;AAAA,QACrD,CAAC,aAAa,YAAY,wBAAwB;AAAA,MACpD;AACA,iBAAW,CAAC,KAAK,KAAK,GAAG,KAAK,MAAM;AAClC,cAAM,QAAQ,MAAM,GAAG;AACvB,YAAI,OAAO,UAAU,YAAY,QAAQ,WAAW;AAClD,cAAI;AAAA,YACF,YAAY;AAAA,cACV,YAAY;AAAA,cACZ;AAAA,cACA,SAAS;AAAA,cACT,OAAO;AAAA,cACP,WAAW,GAAG,GAAG,IAAI,MAAM,QAAQ,CAAC,CAAC,oBAAoB,SAAS;AAAA,cAClE,UAAU;AAAA,cACV,YAAY;AAAA,cACZ,eAAe,CAAC;AAAA,cAChB,UAAU,EAAE,WAAW,KAAK,OAAO,WAAW,QAAQ,MAAM,IAAI,MAAM;AAAA,YACxE,CAAC;AAAA,UACH;AAAA,QACF;AAAA,MACF;AAEA,UAAI,MAAM,eAAe,IAAI,WAAW;AACtC,YAAI;AAAA,UACF,YAAY;AAAA,YACV,YAAY;AAAA,YACZ;AAAA,YACA,SAAS;AAAA,YACT,OAAO;AAAA,YACP,WAAW,gBAAgB,MAAM,aAAa,QAAQ,CAAC,CAAC;AAAA,YACxD,UAAU;AAAA,YACV,YAAY;AAAA,YACZ,eAAe,CAAC;AAAA,YAChB,UAAU,EAAE,eAAe,MAAM,cAAc,OAAO,MAAM,MAAM;AAAA,UACpE,CAAC;AAAA,QACH;AAAA,MACF;AACA,aAAO;AAAA,IACT;AAAA,EACF;AACF;AAgBO,SAAS,mBAAmB,MAA6C;AAC9E,QAAM,KAAK,KAAK,MAAM;AACtB,QAAM,OAAO,KAAK,QAAQ;AAC1B,QAAM,YAAY,KAAK,aAAa;AACpC,SAAO;AAAA,IACL;AAAA,IACA,aACE;AAAA,IACF,WAAW;AAAA,IACX,MAAM,KAAK,QAAQ,EAAE,MAAM,MAAM;AAAA,IACjC,SAAS,SAAS,WAAW;AAAA,IAC7B,MAAM,QAAQ,OAAO;AACnB,YAAM,SAAS,MAAM,KAAK,MAAM,KAAK,QAAQ,KAAK;AAClD,aAAO,OACJ,OAAO,CAAC,MAAM,YAAY,EAAE,KAAK,IAAI,SAAS,EAC9C,IAAI,CAAC,MAAM,eAAe,IAAI,MAAM,CAAC,CAAC;AAAA,IAC3C;AAAA,EACF;AACF;AAEA,SAAS,YAAY,GAAmB;AAEtC,SAAO,KAAK,IAAI,IAAI,KAAK;AAC3B;AAEA,SAAS,eAAe,YAAoB,MAAc,GAA+B;AACvF,QAAM,UAAU,YAAY,EAAE,KAAK;AACnC,QAAM,WACJ,UAAU,IAAI,aAAa,UAAU,IAAI,SAAS,UAAU,IAAI,WAAW;AAC7E,SAAO,YAAY;AAAA,IACjB;AAAA,IACA;AAAA,IACA,SAAS,EAAE;AAAA,IACX,OAAO,GAAG,EAAE,SAAS,IAAI,EAAE,SAAS,WAAW,QAAQ,QAAQ,CAAC,CAAC;AAAA,IACjE,WAAW,EAAE;AAAA,IACb;AAAA,IACA,YAAY;AAAA,IACZ,eAAe,EAAE,WACb,CAAC,EAAE,MAAM,YAAY,KAAK,mBAAmB,SAAS,EAAE,SAAS,CAAC,IAClE,CAAC;AAAA;AAAA;AAAA;AAAA;AAAA,IAKL,oBAAoB;AAAA,IACpB,UAAU,EAAE,YAAY,EAAE,WAAW,WAAW,EAAE,WAAW,UAAU,QAAQ;AAAA,EACjF,CAAC;AACH;AAUO,SAAS,kCACd,OAAwC,CAAC,GACL;AACpC,QAAM,KAAK,KAAK,MAAM;AACtB,QAAM,OAAO,KAAK,QAAQ;AAC1B,SAAO;AAAA,IACL;AAAA,IACA,aACE;AAAA,IACF,WAAW;AAAA,IACX,MAAM,EAAE,MAAM,OAAO,QAAQ,KAAK,SAAS,QAAQ,CAAC,KAAK,QAAQ,KAAK,IAAI,OAAU;AAAA,IACpF,SAAS,GAAG,8BAA8B,YAAY,WAAW;AAAA,IACjE,MAAM,QAAQ,OAAO;AACnB,YAAM,SAAS,MAAM,wBAAwB,OAAO,KAAK,OAAO;AAChE,UAAI,CAAC,OAAO,WAAW;AACrB,eAAO;AAAA,UACL,YAAY;AAAA,YACV,YAAY;AAAA,YACZ;AAAA,YACA,OAAO;AAAA,YACP,WAAW,OAAO;AAAA,YAClB,UAAU;AAAA,YACV,YAAY;AAAA,YACZ,eAAe,CAAC;AAAA,YAChB,UAAU,EAAE,QAAQ,OAAO,MAAM;AAAA,UACnC,CAAC;AAAA,QACH;AAAA,MACF;AACA,YAAM,MAAwB,CAAC;AAC/B,iBAAW,KAAK,OAAO,UAAU;AAG/B,YAAI,EAAE,WAAW,EAAE,SAAS,EAAG;AAC/B,YAAI;AAAA,UACF,YAAY;AAAA,YACV,YAAY;AAAA,YACZ;AAAA,YACA,SAAS,EAAE;AAAA,YACX,OAAO,EAAE,UACL,YAAY,EAAE,OAAO,cAAc,EAAE,KAAK,SAC1C,YAAY,EAAE,OAAO;AAAA,YACzB,WAAW,EAAE;AAAA,YACb,UAAU,aAAa,EAAE,QAAQ;AAAA,YACjC,YAAY;AAAA,YACZ,eAAe,CAAC,EAAE,MAAM,YAAY,KAAK,mBAAmB,SAAS,EAAE,SAAS,CAAC;AAAA,YACjF,UAAU;AAAA,cACR,SAAS,EAAE;AAAA,cACX,SAAS,EAAE;AAAA,cACX,UAAU,EAAE;AAAA,cACZ,UAAU,OAAO,WAAW;AAAA,YAC9B;AAAA,UACF,CAAC;AAAA,QACH;AAAA,MACF;AACA,aAAO;AAAA,IACT;AAAA,EACF;AACF;","names":[]}
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import { AxAIService } from '@ax-llm/ax';
|
|
2
|
-
import { T as TraceAnalysisStore } from './store-
|
|
2
|
+
import { T as TraceAnalysisStore } from './store-C1YxJDEK.js';
|
|
3
3
|
|
|
4
4
|
interface AnalyzeTracesInput {
|
|
5
5
|
/** The user-facing question. Domain framing belongs here, not in the
|
|
@@ -1,7 +1,7 @@
|
|
|
1
|
-
import { A as AnalystRegistry } from './default-registry-
|
|
1
|
+
import { A as AnalystRegistry } from './default-registry-DaK8b3fv.js';
|
|
2
2
|
import { D as DatasetScenario } from './dataset-NENEzRgk.js';
|
|
3
|
-
import { R as RunRecord } from './run-record-
|
|
4
|
-
import { I as InsightReport } from './insight-report-
|
|
3
|
+
import { R as RunRecord } from './run-record-BDH49H2E.js';
|
|
4
|
+
import { I as InsightReport } from './insight-report-DY4nDW9Q.js';
|
|
5
5
|
|
|
6
6
|
/**
|
|
7
7
|
* # `analyzeRuns()` — turn a set of agent runs into an actionable decision packet.
|
|
@@ -1,5 +1,5 @@
|
|
|
1
|
-
import { T as TraceStore } from './store-
|
|
2
|
-
import { S as Span, e as TraceEvent } from './schema-
|
|
1
|
+
import { T as TraceStore } from './store-DGqD0Pyo.js';
|
|
2
|
+
import { S as Span, e as TraceEvent } from './schema-B3Q3l9Z_.js';
|
|
3
3
|
|
|
4
4
|
/**
|
|
5
5
|
* Tool-use metrics — derived purely from trace data.
|
|
@@ -13,9 +13,11 @@ import { S as Span, e as TraceEvent } from './schema-SGWcK9wa.js';
|
|
|
13
13
|
interface ToolUseMetrics {
|
|
14
14
|
runId: string;
|
|
15
15
|
totalCalls: number;
|
|
16
|
+
/** Calls whose arguments were captured and can be compared for duplication. */
|
|
17
|
+
callsWithCapturedArgs: number;
|
|
16
18
|
byTool: Record<string, ToolStats>;
|
|
17
19
|
errorRate: number;
|
|
18
|
-
/** Ratio of calls
|
|
20
|
+
/** Ratio of captured-argument calls already seen with the same tool name and arguments. */
|
|
19
21
|
duplicateRate: number;
|
|
20
22
|
/** Ratio of error calls followed by ≥1 retry on same tool. */
|
|
21
23
|
retryRate: number;
|
|
@@ -24,6 +26,7 @@ interface ToolUseMetrics {
|
|
|
24
26
|
}
|
|
25
27
|
interface ToolStats {
|
|
26
28
|
calls: number;
|
|
29
|
+
callsWithCapturedArgs: number;
|
|
27
30
|
errors: number;
|
|
28
31
|
avgLatencyMs: number;
|
|
29
32
|
duplicates: number;
|
|
@@ -1,10 +1,10 @@
|
|
|
1
|
-
import { c as CalibrationReport } from '../calibration-
|
|
1
|
+
import { c as CalibrationReport } from '../calibration-C8MTS7cw.js';
|
|
2
2
|
import { O as OffPolicyEstimate, a as OffPolicyOptions, b as OffPolicyTrajectory } from '../off-policy-DiwuKKg7.js';
|
|
3
|
-
import { d as CodeAgentSessionSource, a as CodeAgentSessionIntakeOptions, c as CodeAgentSessionMetrics, C as CodeAgentSessionDiagnostic } from '../code-agent-session-
|
|
4
|
-
import { R as RunRecord, a as RunSplitTag } from '../run-record-
|
|
5
|
-
import { T as TraceStore } from '../store-
|
|
6
|
-
import { R as RuntimeTrajectoryRecord, P as ProjectRuntimeTrajectoryEvidenceOptions, a as RuntimeTrajectoryEvidenceProjection } from '../runtime-trajectory-
|
|
7
|
-
import '../schema-
|
|
3
|
+
import { d as CodeAgentSessionSource, a as CodeAgentSessionIntakeOptions, c as CodeAgentSessionMetrics, C as CodeAgentSessionDiagnostic } from '../code-agent-session-CjZsVd19.js';
|
|
4
|
+
import { R as RunRecord, a as RunSplitTag } from '../run-record-BDH49H2E.js';
|
|
5
|
+
import { T as TraceStore } from '../store-DGqD0Pyo.js';
|
|
6
|
+
import { R as RuntimeTrajectoryRecord, P as ProjectRuntimeTrajectoryEvidenceOptions, a as RuntimeTrajectoryEvidenceProjection } from '../runtime-trajectory-DGBIUt4B.js';
|
|
7
|
+
import '../schema-B3Q3l9Z_.js';
|
|
8
8
|
import '../outcome-store-rnXLEqSn.js';
|
|
9
9
|
import '@tangle-network/agent-interface';
|
|
10
10
|
import '../errors-oeQrLqXC.js';
|
|
@@ -11,13 +11,13 @@ import {
|
|
|
11
11
|
import {
|
|
12
12
|
projectRuntimeTrajectoryEvidence
|
|
13
13
|
} from "../chunk-T4SQEITX.js";
|
|
14
|
-
import "../chunk-VI2UW6B6.js";
|
|
15
14
|
import {
|
|
16
15
|
offPolicyEstimateAll
|
|
17
16
|
} from "../chunk-DTJ6QUQB.js";
|
|
18
17
|
import {
|
|
19
18
|
confidenceInterval
|
|
20
19
|
} from "../chunk-PJQFMIOX.js";
|
|
20
|
+
import "../chunk-VI2UW6B6.js";
|
|
21
21
|
import {
|
|
22
22
|
ValidationError
|
|
23
23
|
} from "../chunk-ONWEPEDO.js";
|
|
@@ -1,11 +1,14 @@
|
|
|
1
|
-
export { B as BENCHMARK_SPLIT_SEED, a as BenchmarkAdapter, b as BenchmarkDatasetItem, k as BenchmarkDistribution, c as BenchmarkEvaluation, d as BenchmarkFamily, l as BenchmarkMetricCalibrationOptions, m as BenchmarkMetricCalibrationResult, n as BenchmarkReport, e as BenchmarkResponder, o as BenchmarkRunOptions, p as BenchmarkRunResult, f as BenchmarkScenario, q as BenchmarkSliceSummary, g as BenchmarkSource, h as BenchmarkTaskKind, r as BuildStandardRetrievalItemsOptions, R as RetrievalIdAdapterOptions, S as StandardRetrievalArtifact, s as StandardRetrievalDocument, t as StandardRetrievalEvaluationOptions, u as StandardRetrievalPayload, v as StandardRetrievalQrel, w as StandardRetrievalQuery, x as StandardRetrievalResult, y as buildStandardRetrievalItems, z as calibrateBenchmarkMetric, A as createRetrievalIdBenchmarkAdapter, i as deterministicSplit, C as evaluateStandardRetrieval, D as normalizeRetrievedDocumentIds, E as parseBeirCorpusJsonl, F as parseBeirQueriesJsonl, G as parseJsonlRows, H as parseQrels, I as parseTsvRows, J as renderBenchmarkReportMarkdown, K as retrievalMetricsAtCutoff, L as routing, M as runBenchmarkAdapter, N as summarizeBenchmarkCampaign } from '../index-
|
|
2
|
-
import '../types-
|
|
3
|
-
import '../policy-edit-
|
|
4
|
-
import '../run-record-
|
|
1
|
+
export { B as BENCHMARK_SPLIT_SEED, a as BenchmarkAdapter, b as BenchmarkDatasetItem, k as BenchmarkDistribution, c as BenchmarkEvaluation, d as BenchmarkFamily, l as BenchmarkMetricCalibrationOptions, m as BenchmarkMetricCalibrationResult, n as BenchmarkReport, e as BenchmarkResponder, o as BenchmarkRunOptions, p as BenchmarkRunResult, f as BenchmarkScenario, q as BenchmarkSliceSummary, g as BenchmarkSource, h as BenchmarkTaskKind, r as BuildStandardRetrievalItemsOptions, R as RetrievalIdAdapterOptions, S as StandardRetrievalArtifact, s as StandardRetrievalDocument, t as StandardRetrievalEvaluationOptions, u as StandardRetrievalPayload, v as StandardRetrievalQrel, w as StandardRetrievalQuery, x as StandardRetrievalResult, y as buildStandardRetrievalItems, z as calibrateBenchmarkMetric, A as createRetrievalIdBenchmarkAdapter, i as deterministicSplit, C as evaluateStandardRetrieval, D as normalizeRetrievedDocumentIds, E as parseBeirCorpusJsonl, F as parseBeirQueriesJsonl, G as parseJsonlRows, H as parseQrels, I as parseTsvRows, J as renderBenchmarkReportMarkdown, K as retrievalMetricsAtCutoff, L as routing, M as runBenchmarkAdapter, N as summarizeBenchmarkCampaign } from '../index-PdX4VnPA.js';
|
|
2
|
+
import '../types-BSw1rOUB.js';
|
|
3
|
+
import '../policy-edit-wG9uFEFm.js';
|
|
4
|
+
import '../run-record-BDH49H2E.js';
|
|
5
5
|
import '@tangle-network/agent-interface';
|
|
6
6
|
import '../errors-oeQrLqXC.js';
|
|
7
|
-
import '../schema-
|
|
8
|
-
import '../store-
|
|
9
|
-
import '../types-
|
|
7
|
+
import '../schema-B3Q3l9Z_.js';
|
|
8
|
+
import '../store-C1YxJDEK.js';
|
|
9
|
+
import '../types-BkfcQnxV.js';
|
|
10
|
+
import '../cost-ledger-DWy3XdJc.js';
|
|
10
11
|
import '@tangle-network/tcloud';
|
|
11
|
-
import '../
|
|
12
|
+
import '../llm-client-qoDd18Qz.js';
|
|
13
|
+
import '../raw-provider-sink-C46HDghv.js';
|
|
14
|
+
import '../storage-DrX3v_5B.js';
|
package/dist/benchmarks/index.js
CHANGED
|
@@ -16,22 +16,23 @@ import {
|
|
|
16
16
|
routing_exports,
|
|
17
17
|
runBenchmarkAdapter,
|
|
18
18
|
summarizeBenchmarkCampaign
|
|
19
|
-
} from "../chunk-
|
|
20
|
-
import "../chunk-
|
|
21
|
-
import "../chunk-
|
|
22
|
-
import "../chunk-
|
|
23
|
-
import "../chunk-
|
|
24
|
-
import "../chunk-
|
|
25
|
-
import "../chunk-CIUOICJT.js";
|
|
19
|
+
} from "../chunk-JSDVRFAP.js";
|
|
20
|
+
import "../chunk-ZUXV7UWZ.js";
|
|
21
|
+
import "../chunk-HQPHZGL6.js";
|
|
22
|
+
import "../chunk-IDZTTFRR.js";
|
|
23
|
+
import "../chunk-3YYRZDON.js";
|
|
24
|
+
import "../chunk-MGEHEHSN.js";
|
|
26
25
|
import "../chunk-ARU2PZFM.js";
|
|
27
26
|
import "../chunk-PJQFMIOX.js";
|
|
28
|
-
import "../chunk-
|
|
27
|
+
import "../chunk-4JLWXDYA.js";
|
|
29
28
|
import "../chunk-GGE4NNQT.js";
|
|
30
|
-
import "../chunk-
|
|
29
|
+
import "../chunk-S2F4J57L.js";
|
|
31
30
|
import "../chunk-5UF54T55.js";
|
|
32
31
|
import "../chunk-XJYR7XFV.js";
|
|
33
32
|
import "../chunk-VSMTAMNK.js";
|
|
34
|
-
import "../chunk-
|
|
33
|
+
import "../chunk-NJC7U437.js";
|
|
34
|
+
import "../chunk-VCTY3W6J.js";
|
|
35
|
+
import "../chunk-VI2UW6B6.js";
|
|
35
36
|
import "../chunk-PC4UYEBM.js";
|
|
36
37
|
import "../chunk-ONWEPEDO.js";
|
|
37
38
|
import "../chunk-PZ5AY32C.js";
|
|
@@ -1,7 +1,7 @@
|
|
|
1
|
-
import { S as SandboxDriver, H as HarnessConfig, a as SandboxHarnessResult, T as TestGradedScenario, b as TestGradedRunResult } from '../test-graded-scenario-
|
|
2
|
-
import { T as TraceEmitter } from '../emitter-
|
|
3
|
-
import { R as Run } from '../schema-
|
|
4
|
-
import { T as TraceStore } from '../store-
|
|
1
|
+
import { S as SandboxDriver, H as HarnessConfig, a as SandboxHarnessResult, T as TestGradedScenario, b as TestGradedRunResult } from '../test-graded-scenario-B0ybnPY7.js';
|
|
2
|
+
import { T as TraceEmitter } from '../emitter-CjD7vUwv.js';
|
|
3
|
+
import { R as Run } from '../schema-B3Q3l9Z_.js';
|
|
4
|
+
import { T as TraceStore } from '../store-DGqD0Pyo.js';
|
|
5
5
|
|
|
6
6
|
/**
|
|
7
7
|
* BuilderSession — ties a builder-of-builders workflow together.
|
|
@@ -1,5 +1,5 @@
|
|
|
1
|
-
import { T as TraceStore } from './store-
|
|
2
|
-
import { R as Run } from './schema-
|
|
1
|
+
import { T as TraceStore } from './store-DGqD0Pyo.js';
|
|
2
|
+
import { R as Run } from './schema-B3Q3l9Z_.js';
|
|
3
3
|
import { O as OutcomeFilter, b as OutcomeStore } from './outcome-store-rnXLEqSn.js';
|
|
4
4
|
|
|
5
5
|
/**
|
package/dist/campaign/index.d.ts
CHANGED
|
@@ -1,34 +1,37 @@
|
|
|
1
|
-
import {
|
|
2
|
-
export {
|
|
3
|
-
import { P as PairedArmsComparison, S as SignedManifest, B as BackendIntegrityReport, C as CompletionRequirement, R as RuntimeEventLike, a as CompletionVerdict, b as ProducedState, c as CorrectnessChecker } from '../pre-registration
|
|
4
|
-
export { L as LlmJudgeDimension, d as LlmJudgeOptions, l as llmJudge } from '../pre-registration
|
|
5
|
-
import {
|
|
6
|
-
|
|
7
|
-
|
|
8
|
-
import {
|
|
9
|
-
export { f as
|
|
10
|
-
import { a as PairedBootstrapResult, E as EProcessState } from '../statistics-
|
|
11
|
-
import { C as CampaignStorage } from '../storage-
|
|
12
|
-
export { f as fsCampaignStorage, i as inMemoryCampaignStorage } from '../storage-
|
|
13
|
-
export { A as AxisEvidence, a as AxisVerdict, B as BuildEvidenceVectorOptions, l as BuildLoopProvenanceArgs, D as DefaultProductionGateOptions, m as EmitLoopProvenanceArgs, n as EmitLoopProvenanceResult, E as EvidenceVector, b as EvolutionaryProposerOptions, H as HeldOutGateOptions, o as LoopProvenanceBackend, q as LoopProvenanceCandidate, L as LoopProvenanceRecord, O as ObjectiveSource, c as ParetoSignificanceGateOptions, P as PowerPreflight, s as PowerPreflightOptions, d as PromotionObjective, e as PromotionPolicy, R as RunEvalOptions, f as buildEvidenceVector, t as buildLoopProvenanceRecord, g as composeGate, h as defaultProductionGate, u as emitLoopProvenance, i as evolutionaryProposer, j as heldOutGate, v as loopProvenanceSpans, p as paretoPolicy, k as paretoSignificanceGate, w as powerPreflight, x as provenanceRecordPath, y as provenanceSpansPath, r as runEval } from '../provenance-
|
|
1
|
+
import { v as PolicyEditAdmissionOptions, u as PolicyEditAdmission, t as PolicyEdit, c as AnalystFinding, F as FindingToPolicyEditOptions } from '../policy-edit-wG9uFEFm.js';
|
|
2
|
+
export { r as POLICY_EDIT_CANDIDATE_RECORD_SCHEMA, P as PolicyEditCandidateRecord, a2 as validatePolicyEditCandidateRecord } from '../policy-edit-wG9uFEFm.js';
|
|
3
|
+
import { P as PairedArmsComparison, S as SignedManifest, B as BackendIntegrityReport, C as CompletionRequirement, R as RuntimeEventLike, a as CompletionVerdict, b as ProducedState, c as CorrectnessChecker } from '../pre-registration-BWQhJ3vz.js';
|
|
4
|
+
export { L as LlmJudgeDimension, d as LlmJudgeOptions, l as llmJudge } from '../pre-registration-BWQhJ3vz.js';
|
|
5
|
+
import { C as CampaignRunPlan, P as PlanCampaignRunOptions, h as RunCampaignOptions, i as RunImprovementLoopOptions } from '../gepa-eESocoDi.js';
|
|
6
|
+
export { o as CampaignRunPlanCell, p as GepaProposerConstraints, G as GepaProposerOptions, O as OpenAutoPrOptions, q as OpenAutoPrResult, e as ReferenceEquivalenceJudgeOptions, g as ReferenceEquivalenceScenario, a as RunImprovementLoopResult, R as RunOptimizationOptions, s as RunOptimizationResult, t as countSentenceEdits, j as createReferenceEquivalenceJudge, u as defaultRenderDiff, v as extractH2Sections, k as gepaProposer, w as openAutoPr, x as planCampaignRun, r as runCampaign, l as runImprovementLoop, y as runOptimization } from '../gepa-eESocoDi.js';
|
|
7
|
+
import { A as AnalyzeTracesOptions, a as AnalyzeTracesInput, b as AnalyzeTracesResult } from '../analyst-C8HHvfJp.js';
|
|
8
|
+
import { S as Scenario, M as MutableSurface, D as DispatchContext, b as JudgeConfig, G as Gate, o as GenerationRecord, J as JudgeScore, L as LabeledScenarioStore, s as LabeledScenarioWrite, t as LabeledScenarioSampleArgs, u as LabeledScenarioRecord, v as LabelTrust, c as SurfaceProposer, w as ProposedCandidate, x as ProposeContext, j as CodeSurface, y as LabeledScenarioSource, C as CampaignResult } from '../types-BSw1rOUB.js';
|
|
9
|
+
export { e as CampaignAggregates, f as CampaignArtifactWriter, g as CampaignCellResult, h as CampaignCostMeter, z as CampaignTokenUsage, i as CampaignTraceWriter, k as DispatchFn, l as GateContext, d as GateDecision, m as GateResult, n as GenerationCandidate, A as JudgeAggregate, a as JudgeDimension, p as Mutator, O as OptimizationProposer, q as OptimizerConfig, P as ParetoParent, R as RedactionStatus, B as ScenarioAggregate, E as ScoredSurfaceOutcome, r as SessionScript, T as TraceSpan, F as isProposedCandidate, H as labelTrustRank } from '../types-BSw1rOUB.js';
|
|
10
|
+
import { a as PairedBootstrapResult, E as EProcessState } from '../statistics-KUnG73jH.js';
|
|
11
|
+
import { C as CampaignStorage } from '../storage-DrX3v_5B.js';
|
|
12
|
+
export { c as createRunCostLedger, f as fsCampaignStorage, i as inMemoryCampaignStorage } from '../storage-DrX3v_5B.js';
|
|
13
|
+
export { A as AxisEvidence, a as AxisVerdict, B as BuildEvidenceVectorOptions, l as BuildLoopProvenanceArgs, D as DefaultProductionGateOptions, m as EmitLoopProvenanceArgs, n as EmitLoopProvenanceResult, E as EvidenceVector, b as EvolutionaryProposerOptions, H as HeldOutGateOptions, o as LoopProvenanceBackend, q as LoopProvenanceCandidate, L as LoopProvenanceRecord, O as ObjectiveSource, c as ParetoSignificanceGateOptions, P as PowerPreflight, s as PowerPreflightOptions, d as PromotionObjective, e as PromotionPolicy, R as RunEvalOptions, f as buildEvidenceVector, t as buildLoopProvenanceRecord, g as composeGate, h as defaultProductionGate, u as emitLoopProvenance, i as evolutionaryProposer, j as heldOutGate, v as loopProvenanceSpans, p as paretoPolicy, k as paretoSignificanceGate, w as powerPreflight, x as provenanceRecordPath, y as provenanceSpansPath, r as runEval } from '../provenance-DpjwyseI.js';
|
|
14
|
+
import { a as LlmClientOptions } from '../llm-client-qoDd18Qz.js';
|
|
14
15
|
import { AgentProfile } from '@tangle-network/agent-interface';
|
|
15
16
|
import { A as AgentEvalError, V as ValidationError } from '../errors-oeQrLqXC.js';
|
|
16
|
-
import { a as RunSplitTag, R as RunRecord } from '../run-record-
|
|
17
|
-
import {
|
|
18
|
-
import '../
|
|
19
|
-
import '../
|
|
17
|
+
import { a as RunSplitTag, R as RunRecord } from '../run-record-BDH49H2E.js';
|
|
18
|
+
import { C as CostLedger, b as CostLedgerSummary, M as MaximumCharge, c as CostReceiptInput } from '../cost-ledger-DWy3XdJc.js';
|
|
19
|
+
import { T as TraceAnalystKindSpec } from '../kind-factory-ClZmO25A.js';
|
|
20
|
+
import '../store-C1YxJDEK.js';
|
|
21
|
+
import '../types-BkfcQnxV.js';
|
|
20
22
|
import '@tangle-network/tcloud';
|
|
23
|
+
import 'zod';
|
|
24
|
+
import '../raw-provider-sink-C46HDghv.js';
|
|
21
25
|
import '../verdict-C9MlYujm.js';
|
|
22
|
-
import '@ax-llm/ax';
|
|
23
26
|
import '../dataset-NENEzRgk.js';
|
|
24
|
-
import '../store-
|
|
25
|
-
import '../schema-
|
|
27
|
+
import '../store-DGqD0Pyo.js';
|
|
28
|
+
import '../schema-B3Q3l9Z_.js';
|
|
29
|
+
import '@ax-llm/ax';
|
|
26
30
|
import '../judge-calibration-7C-IDmKr.js';
|
|
27
31
|
import '../hosted/index.js';
|
|
28
|
-
import '../insight-report-
|
|
29
|
-
import '../summary-report-
|
|
30
|
-
import '../failure-cluster-
|
|
31
|
-
import 'zod';
|
|
32
|
+
import '../insight-report-DY4nDW9Q.js';
|
|
33
|
+
import '../summary-report-C5bKFfm-.js';
|
|
34
|
+
import '../failure-cluster-DOAcSJ87.js';
|
|
32
35
|
|
|
33
36
|
/**
|
|
34
37
|
* Lineage DAG — a git-graph of improvement candidates.
|
|
@@ -1468,12 +1471,12 @@ declare function fapoEscalationEntry<TScenario extends Scenario, TArtifact>(conf
|
|
|
1468
1471
|
* It runs `runCampaign` once per profile (reusing its seeds, reps, bootstrap
|
|
1469
1472
|
* CIs, resumability, and the `LabeledScenarioStore` capture flywheel), maps
|
|
1470
1473
|
* every cell to a validated `RunRecord` carrying the real `tokenUsage` the
|
|
1471
|
-
* dispatch
|
|
1474
|
+
* dispatch committed via `ctx.cost.runPaidCall`, and runs `assertRealBackend`
|
|
1472
1475
|
* BY CONSTRUCTION before returning — so a stub-backend run fails loudly instead
|
|
1473
1476
|
* of reporting a clean 0/N leaderboard.
|
|
1474
1477
|
*
|
|
1475
1478
|
* Dispatch contract: a dispatch that calls an LLM MUST report usage via
|
|
1476
|
-
* `ctx.cost.
|
|
1479
|
+
* `ctx.cost.runPaidCall({ execute, receipt })`.
|
|
1477
1480
|
* A dispatch that reports zero tokens is indistinguishable from a stub and the
|
|
1478
1481
|
* integrity guard treats it as one.
|
|
1479
1482
|
*/
|
|
@@ -1485,8 +1488,8 @@ declare class ProfileMatrixError extends AgentEvalError {
|
|
|
1485
1488
|
constructor(message: string);
|
|
1486
1489
|
}
|
|
1487
1490
|
/** Dispatch for one cell: render `profile` against `scenario`, returning the
|
|
1488
|
-
* artifact the judges score.
|
|
1489
|
-
*
|
|
1491
|
+
* artifact the judges score. Run LLM work through `ctx.cost.runPaidCall` —
|
|
1492
|
+
* the integrity check depends on its receipt. */
|
|
1490
1493
|
type ProfileDispatchFn<TScenario extends Scenario, TArtifact> = (profile: AgentProfile, scenario: TScenario, ctx: DispatchContext) => Promise<TArtifact>;
|
|
1491
1494
|
interface RunProfileMatrixOptions<TScenario extends Scenario, TArtifact> {
|
|
1492
1495
|
/** Axis 3 — the agent-under-test configurations. Each is one column. */
|
|
@@ -1637,7 +1640,7 @@ interface PlaybackContext extends DispatchContext {
|
|
|
1637
1640
|
* `SandboxPlaybackDriver` (real API / sandbox workspace) and
|
|
1638
1641
|
* `PlaywrightPlaybackDriver` (real UI) — because they depend on runtime /
|
|
1639
1642
|
* browser infra the substrate must not import. The driver MUST report LLM
|
|
1640
|
-
* usage
|
|
1643
|
+
* usage through `ctx.cost.runPaidCall` so the backend-integrity check sees real
|
|
1641
1644
|
* tokens (a run that never reports tokens reads as a stub).
|
|
1642
1645
|
*/
|
|
1643
1646
|
interface PlaybackDriver<TStory extends UserStory = UserStory> {
|
|
@@ -1950,10 +1953,14 @@ interface ProposePatchesArgs {
|
|
|
1950
1953
|
/** How many candidate patches to propose. */
|
|
1951
1954
|
count: number;
|
|
1952
1955
|
signal: AbortSignal;
|
|
1956
|
+
costLedger?: CostLedger;
|
|
1957
|
+
costPhase?: string;
|
|
1953
1958
|
}
|
|
1954
1959
|
interface SkillOptProposerOptions {
|
|
1955
1960
|
llm: LlmClientOptions;
|
|
1956
1961
|
model: string;
|
|
1962
|
+
/** Optional ledger for direct proposer use. Campaign context takes precedence. */
|
|
1963
|
+
costLedger?: CostLedger;
|
|
1957
1964
|
/** What the skill document governs — orients the prompt. */
|
|
1958
1965
|
target: string;
|
|
1959
1966
|
/** Default ops-per-patch cap when used as a bare `SurfaceProposer`. The
|
|
@@ -2077,6 +2084,8 @@ interface RunSkillOptResult {
|
|
|
2077
2084
|
/** Total cost across every scoring campaign (train evidence + holdout
|
|
2078
2085
|
* acceptance) the hill-climb ran. */
|
|
2079
2086
|
totalCostUsd: number;
|
|
2087
|
+
/** Run-wide spend, including scoring, proposals, and judges. */
|
|
2088
|
+
cost: CostLedgerSummary;
|
|
2080
2089
|
}
|
|
2081
2090
|
/**
|
|
2082
2091
|
* SkillOpt sequential hill-climb: each epoch reflects on train-scenario weaknesses, proposes bounded patches, accepts the first patch that strictly improves the held-out composite, and anneals the edit budget on consecutive rejections.
|
|
@@ -2191,6 +2200,11 @@ interface HaloProposerOptions {
|
|
|
2191
2200
|
model?: string;
|
|
2192
2201
|
/** Model used to APPLY halo's findings to the prompt surface. Default = `model`. */
|
|
2193
2202
|
applyModel?: string;
|
|
2203
|
+
/** Optional ledger for direct proposer use. Campaign context takes precedence. */
|
|
2204
|
+
costLedger?: CostLedger;
|
|
2205
|
+
analysisMaximumCharge?: MaximumCharge;
|
|
2206
|
+
analysisReceipt?: (report: string) => CostReceiptInput;
|
|
2207
|
+
applyMaxTokens?: number;
|
|
2194
2208
|
/** The real halo binary. Default 'halo' (from `pip install halo-engine`). */
|
|
2195
2209
|
haloBin?: string;
|
|
2196
2210
|
/** Resolve the OTLP traces (JSONL string) halo should analyze for THIS
|
|
@@ -2323,6 +2337,8 @@ interface PolicyEditHistoryGenerationContext {
|
|
|
2323
2337
|
interface LlmPolicyEditProposerOptions {
|
|
2324
2338
|
llm: LlmClientOptions;
|
|
2325
2339
|
model: string;
|
|
2340
|
+
/** Optional ledger for direct proposer use. Campaign context takes precedence. */
|
|
2341
|
+
costLedger?: CostLedger;
|
|
2326
2342
|
/** Plain-language description of the JSON surface being improved. */
|
|
2327
2343
|
target: string;
|
|
2328
2344
|
/** PolicyEdit target surface every authored edit must retain. */
|
|
@@ -2396,6 +2412,8 @@ declare function projectPolicyEditHistory(history: readonly GenerationRecord[],
|
|
|
2396
2412
|
*/
|
|
2397
2413
|
|
|
2398
2414
|
interface MemoryCurationProposerOptions {
|
|
2415
|
+
/** Optional ledger for direct proposer use. Campaign context takes precedence. */
|
|
2416
|
+
costLedger?: CostLedger;
|
|
2399
2417
|
/** Top-K lessons retained in the surface memory block. Default 12. */
|
|
2400
2418
|
maxEntries?: number;
|
|
2401
2419
|
/** Heading rendered above the lessons inside the block. Default below. */
|
|
@@ -2408,6 +2426,7 @@ interface MemoryCurationProposerOptions {
|
|
|
2408
2426
|
baseUrl: string;
|
|
2409
2427
|
apiKey?: string;
|
|
2410
2428
|
model: string;
|
|
2429
|
+
maxTokens?: number;
|
|
2411
2430
|
fetchImpl?: LlmClientOptions['fetch'];
|
|
2412
2431
|
};
|
|
2413
2432
|
}
|
|
@@ -2496,6 +2515,11 @@ interface TraceAnalystProposerOptions {
|
|
|
2496
2515
|
/** Model used to APPLY findings to the prompt surface. Default = `model`.
|
|
2497
2516
|
* Keep this EQUAL to haloProposer's `applyModel` for an apples-to-apples run. */
|
|
2498
2517
|
applyModel?: string;
|
|
2518
|
+
/** Optional ledger for direct proposer use. Campaign context takes precedence. */
|
|
2519
|
+
costLedger?: CostLedger;
|
|
2520
|
+
analysisMaximumCharge?: MaximumCharge;
|
|
2521
|
+
analysisReceipt?: (report: string) => CostReceiptInput;
|
|
2522
|
+
applyMaxTokens?: number;
|
|
2499
2523
|
/** Ax provider name. Default 'openai' — works for any OpenAI-compatible base
|
|
2500
2524
|
* via `apiURL`. Use 'deepseek' to hit DeepSeek's native provider. */
|
|
2501
2525
|
provider?: string;
|