@tangle-network/agent-eval 0.80.0 → 0.82.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +27 -0
- package/README.md +53 -152
- package/dist/adapters/http.d.ts +2 -2
- package/dist/adapters/langchain.d.ts +2 -2
- package/dist/adapters/otel.d.ts +4 -4
- package/dist/{agent-profile-aSEaJ9Pl.d.ts → agent-profile-D0PBIWlV.d.ts} +14 -2
- package/dist/analyst/index.d.ts +10 -10
- package/dist/analyst/index.js +3 -3
- package/dist/{analyst-t7zZS3TV.d.ts → analyst-C8HHvfJp.d.ts} +1 -1
- package/dist/belief-state/index.d.ts +344 -8
- package/dist/belief-state/index.js +1518 -142
- package/dist/belief-state/index.js.map +1 -1
- package/dist/benchmarks/index.d.ts +2 -2
- package/dist/campaign/index.d.ts +40 -120
- package/dist/campaign/index.js +129 -238
- package/dist/campaign/index.js.map +1 -1
- package/dist/{chunk-RPLZ4OIB.js → chunk-BABOZOSN.js} +7 -4
- package/dist/{chunk-RPLZ4OIB.js.map → chunk-BABOZOSN.js.map} +1 -1
- package/dist/{chunk-IHDHUN2X.js → chunk-CVVHBFGN.js} +3 -3
- package/dist/chunk-CVVHBFGN.js.map +1 -0
- package/dist/{chunk-B26KI423.js → chunk-FZWAFVAA.js} +2 -2
- package/dist/{chunk-ITBRCT73.js → chunk-IDVBLYCY.js} +2 -10
- package/dist/chunk-IDVBLYCY.js.map +1 -0
- package/dist/{chunk-5LVWPNS5.js → chunk-L5G7OUKD.js} +4 -4
- package/dist/{chunk-5LVWPNS5.js.map → chunk-L5G7OUKD.js.map} +1 -1
- package/dist/{chunk-GXHLRXDI.js → chunk-OTYQPHPL.js} +4 -4
- package/dist/{chunk-6REHLN5J.js → chunk-QS3RBQPI.js} +2 -2
- package/dist/{chunk-GWGO2K6Y.js → chunk-RBNA5AZT.js} +2 -2
- package/dist/chunk-S42AWHMP.js +697 -0
- package/dist/chunk-S42AWHMP.js.map +1 -0
- package/dist/chunk-VI2UW6B6.js +162 -0
- package/dist/chunk-VI2UW6B6.js.map +1 -0
- package/dist/{chunk-CF67I6QY.js → chunk-VIDQF3F5.js} +2 -2
- package/dist/{chunk-XXNIODOM.js → chunk-WJL2NJXN.js} +3 -3
- package/dist/{chunk-LB2UOI5F.js → chunk-YGYXHNAQ.js} +47 -160
- package/dist/chunk-YGYXHNAQ.js.map +1 -0
- package/dist/{chunk-KX6F6NCG.js → chunk-Z7VFTS2J.js} +2 -2
- package/dist/{chunk-ZPSKPT3V.js → chunk-ZZ2HOPME.js} +5 -2
- package/dist/chunk-ZZ2HOPME.js.map +1 -0
- package/dist/cli.js +2 -2
- package/dist/code-agent-session-BRXmavYv.d.ts +80 -0
- package/dist/contract/index.d.ts +45 -15
- package/dist/contract/index.js +32 -7
- package/dist/contract/index.js.map +1 -1
- package/dist/{control-CehLtoET.d.ts → control-GeE8OhpN.d.ts} +1 -1
- package/dist/control.d.ts +2 -2
- package/dist/hosted/index.d.ts +4 -4
- package/dist/{index-B1RKber3.d.ts → index-DE3RXAXD.d.ts} +1 -1
- package/dist/index.d.ts +78 -287
- package/dist/index.js +87 -410
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-dlpEzQDi.d.ts → insight-report-3ADTfClO.d.ts} +1 -1
- package/dist/{kind-factory-DqV2t1Xk.d.ts → kind-factory-CVecZZG_.d.ts} +2 -2
- package/dist/{llm-client-DbjLfz-K.d.ts → llm-client-CuUg2Mn3.d.ts} +1 -1
- package/dist/meta-eval/index.d.ts +2 -2
- package/dist/openapi.json +1 -1
- package/dist/pipelines/index.js +2 -2
- package/dist/{provenance-jG-Gngg8.d.ts → provenance-DPpNIOJD.d.ts} +4 -4
- package/dist/{registry-BK0Zee01.d.ts → registry-DrEQ3Luj.d.ts} +1 -1
- package/dist/{release-report-CXXZlR8g.d.ts → release-report-hlNtD12q.d.ts} +2 -2
- package/dist/reporting.d.ts +5 -5
- package/dist/reporting.js +3 -3
- package/dist/{researcher-rInLj9De.d.ts → researcher-BLPHBbNV.d.ts} +3 -3
- package/dist/rl.d.ts +7 -7
- package/dist/rl.js +4 -4
- package/dist/{rubric-predictive-validity-CLPuwiUw.d.ts → rubric-predictive-validity-CnEl9Jc8.d.ts} +1 -1
- package/dist/{run-campaign-OVEZF24D.js → run-campaign-4Y5V5CN3.js} +3 -3
- package/dist/{run-improvement-loop-BAl_aVOZ.d.ts → run-improvement-loop-CNqQckTj.d.ts} +3 -3
- package/dist/{run-record-sItO5ftF.d.ts → run-record-De9VarXR.d.ts} +1 -1
- package/dist/{semantic-concept-judge-qXEUV2w7.d.ts → semantic-concept-judge-DIEgr_6v.d.ts} +6 -6
- package/dist/{statistics-B7yCbi9i.d.ts → statistics-CnC1FMbx.d.ts} +3 -7
- package/dist/{store-GmBE2pZZ.d.ts → store-C1YxJDEK.d.ts} +1 -1
- package/dist/{summary-report-BTaXq1TS.d.ts → summary-report-Db0dDSWP.d.ts} +1 -1
- package/dist/traces.d.ts +5 -5
- package/dist/{types-DRvV0zRo.d.ts → types-Cu3u_x59.d.ts} +3 -3
- package/dist/{types-4mm2msnR.d.ts → types-D7lLRYe9.d.ts} +1 -1
- package/dist/wire/index.js +2 -2
- package/dist/workflow/index.d.ts +7 -7
- package/dist/workflow/index.js +1 -1
- package/docs/concepts.md +1 -0
- package/docs/research/belief-state-agent-eval-roadmap.md +39 -7
- package/docs/self-improvement-map.md +111 -0
- package/package.json +2 -2
- package/dist/chunk-IHDHUN2X.js.map +0 -1
- package/dist/chunk-ITBRCT73.js.map +0 -1
- package/dist/chunk-LB2UOI5F.js.map +0 -1
- package/dist/chunk-ZPSKPT3V.js.map +0 -1
- /package/dist/{chunk-B26KI423.js.map → chunk-FZWAFVAA.js.map} +0 -0
- /package/dist/{chunk-GXHLRXDI.js.map → chunk-OTYQPHPL.js.map} +0 -0
- /package/dist/{chunk-6REHLN5J.js.map → chunk-QS3RBQPI.js.map} +0 -0
- /package/dist/{chunk-GWGO2K6Y.js.map → chunk-RBNA5AZT.js.map} +0 -0
- /package/dist/{chunk-CF67I6QY.js.map → chunk-VIDQF3F5.js.map} +0 -0
- /package/dist/{chunk-XXNIODOM.js.map → chunk-WJL2NJXN.js.map} +0 -0
- /package/dist/{chunk-KX6F6NCG.js.map → chunk-Z7VFTS2J.js.map} +0 -0
- /package/dist/{run-campaign-OVEZF24D.js.map → run-campaign-4Y5V5CN3.js.map} +0 -0
package/dist/campaign/index.d.ts
CHANGED
|
@@ -1,26 +1,26 @@
|
|
|
1
|
-
import { A as AnalyzeTracesOptions, a as AnalyzeTracesInput, b as AnalyzeTracesResult } from '../analyst-
|
|
2
|
-
import { S as Scenario, M as MutableSurface, b as DispatchContext, a as JudgeConfig, I as ImprovementDriver, q as ProposeContext, J as JudgeScore, L as LabeledScenarioStore, r as LabeledScenarioWrite, s as LabeledScenarioSampleArgs, t as LabeledScenarioRecord, u as LabelTrust, v as LabeledScenarioSource, g as CampaignResult, i as CodeSurface } from '../types-
|
|
3
|
-
export { C as CampaignAggregates, d as CampaignArtifactWriter, e as CampaignCellResult, f as CampaignCostMeter, w as CampaignTokenUsage, h as CampaignTraceWriter, D as DispatchFn, G as Gate, j as GateContext, c as GateDecision, k as GateResult, l as GenerationCandidate, m as GenerationRecord, x as JudgeAggregate, n as JudgeDimension, o as Mutator, O as OptimizerConfig, P as ParetoParent, y as ProposedCandidate, R as RedactionStatus, z as ScenarioAggregate, p as SessionScript, T as TraceSpan, A as isProposedCandidate, B as labelTrustRank } from '../types-
|
|
4
|
-
import {
|
|
5
|
-
export {
|
|
6
|
-
export { A as AxisEvidence, a as AxisVerdict, B as BuildEvidenceVectorOptions, k as BuildLoopProvenanceArgs, D as DefaultProductionGateOptions, l as EmitLoopProvenanceArgs, m as EmitLoopProvenanceResult, E as EvidenceVector, b as EvolutionaryDriverOptions, H as HeldOutGateOptions, n as LoopProvenanceBackend, o as LoopProvenanceCandidate, L as LoopProvenanceRecord, O as ObjectiveSource, P as ParetoSignificanceGateOptions, c as PromotionObjective, d as PromotionPolicy, R as RunEvalOptions, e as buildEvidenceVector, q as buildLoopProvenanceRecord, f as composeGate, g as defaultProductionGate, s as emitLoopProvenance, h as evolutionaryDriver, i as heldOutGate, t as loopProvenanceSpans, p as paretoPolicy, j as paretoSignificanceGate, u as provenanceRecordPath, v as provenanceSpansPath, r as runEval, w as surfaceContentHash } from '../provenance-
|
|
7
|
-
import { L as LlmClientOptions } from '../llm-client-
|
|
8
|
-
import { c as TraceAnalystKindSpec } from '../kind-factory-
|
|
9
|
-
import { c as AnalystFinding } from '../types-
|
|
10
|
-
import { a as PairedBootstrapResult } from '../statistics-
|
|
11
|
-
import { A as AgentProfile, B as BackendIntegrityReport, C as CompletionRequirement, R as RuntimeEventLike, a as CompletionVerdict, P as ProducedState, b as CorrectnessChecker } from '../agent-profile-
|
|
1
|
+
import { A as AnalyzeTracesOptions, a as AnalyzeTracesInput, b as AnalyzeTracesResult } from '../analyst-C8HHvfJp.js';
|
|
2
|
+
import { S as Scenario, M as MutableSurface, b as DispatchContext, a as JudgeConfig, I as ImprovementDriver, q as ProposeContext, J as JudgeScore, L as LabeledScenarioStore, r as LabeledScenarioWrite, s as LabeledScenarioSampleArgs, t as LabeledScenarioRecord, u as LabelTrust, v as LabeledScenarioSource, g as CampaignResult, i as CodeSurface } from '../types-D7lLRYe9.js';
|
|
3
|
+
export { C as CampaignAggregates, d as CampaignArtifactWriter, e as CampaignCellResult, f as CampaignCostMeter, w as CampaignTokenUsage, h as CampaignTraceWriter, D as DispatchFn, G as Gate, j as GateContext, c as GateDecision, k as GateResult, l as GenerationCandidate, m as GenerationRecord, x as JudgeAggregate, n as JudgeDimension, o as Mutator, O as OptimizerConfig, P as ParetoParent, y as ProposedCandidate, R as RedactionStatus, z as ScenarioAggregate, p as SessionScript, T as TraceSpan, A as isProposedCandidate, B as labelTrustRank } from '../types-D7lLRYe9.js';
|
|
4
|
+
import { b as RunCampaignOptions, c as RunImprovementLoopOptions, C as CampaignStorage } from '../run-improvement-loop-CNqQckTj.js';
|
|
5
|
+
export { e as GepaDriverConstraints, G as GepaDriverOptions, O as OpenAutoPrOptions, h as OpenAutoPrResult, a as RunImprovementLoopResult, R as RunOptimizationOptions, j as RunOptimizationResult, k as countSentenceEdits, l as defaultRenderDiff, m as extractH2Sections, f as fsCampaignStorage, g as gepaDriver, i as inMemoryCampaignStorage, o as openAutoPr, r as runCampaign, d as runImprovementLoop, n as runOptimization, s as surfaceHash } from '../run-improvement-loop-CNqQckTj.js';
|
|
6
|
+
export { A as AxisEvidence, a as AxisVerdict, B as BuildEvidenceVectorOptions, k as BuildLoopProvenanceArgs, D as DefaultProductionGateOptions, l as EmitLoopProvenanceArgs, m as EmitLoopProvenanceResult, E as EvidenceVector, b as EvolutionaryDriverOptions, H as HeldOutGateOptions, n as LoopProvenanceBackend, o as LoopProvenanceCandidate, L as LoopProvenanceRecord, O as ObjectiveSource, P as ParetoSignificanceGateOptions, c as PromotionObjective, d as PromotionPolicy, R as RunEvalOptions, e as buildEvidenceVector, q as buildLoopProvenanceRecord, f as composeGate, g as defaultProductionGate, s as emitLoopProvenance, h as evolutionaryDriver, i as heldOutGate, t as loopProvenanceSpans, p as paretoPolicy, j as paretoSignificanceGate, u as provenanceRecordPath, v as provenanceSpansPath, r as runEval, w as surfaceContentHash } from '../provenance-DPpNIOJD.js';
|
|
7
|
+
import { L as LlmClientOptions } from '../llm-client-CuUg2Mn3.js';
|
|
8
|
+
import { c as TraceAnalystKindSpec } from '../kind-factory-CVecZZG_.js';
|
|
9
|
+
import { c as AnalystFinding } from '../types-Cu3u_x59.js';
|
|
10
|
+
import { a as PairedBootstrapResult } from '../statistics-CnC1FMbx.js';
|
|
11
|
+
import { A as AgentProfile, B as BackendIntegrityReport, C as CompletionRequirement, R as RuntimeEventLike, a as CompletionVerdict, P as ProducedState, b as CorrectnessChecker } from '../agent-profile-D0PBIWlV.js';
|
|
12
12
|
import { A as AgentEvalError } from '../errors-Dwqw-T_m.js';
|
|
13
|
-
import {
|
|
13
|
+
import { a as RunSplitTag, R as RunRecord } from '../run-record-De9VarXR.js';
|
|
14
14
|
import '@ax-llm/ax';
|
|
15
|
-
import '../store-
|
|
15
|
+
import '../store-C1YxJDEK.js';
|
|
16
16
|
import '../red-team-DW9Ca_tj.js';
|
|
17
17
|
import '../dataset-B2kL-fSM.js';
|
|
18
18
|
import '../store-CKUAgsJz.js';
|
|
19
19
|
import '../schema-m0gsnbt3.js';
|
|
20
20
|
import '../pareto-E-pembql.js';
|
|
21
21
|
import '../hosted/index.js';
|
|
22
|
-
import '../insight-report-
|
|
23
|
-
import '../summary-report-
|
|
22
|
+
import '../insight-report-3ADTfClO.js';
|
|
23
|
+
import '../summary-report-Db0dDSWP.js';
|
|
24
24
|
import '../failure-cluster-CL7IVgkJ.js';
|
|
25
25
|
import '../judge-calibration-DilmB3Ml.js';
|
|
26
26
|
import '../raw-provider-sink-C46HDghv.js';
|
|
@@ -172,79 +172,23 @@ interface AceDriverOptions {
|
|
|
172
172
|
}
|
|
173
173
|
declare function aceDriver(opts?: AceDriverOptions): ImprovementDriver;
|
|
174
174
|
|
|
175
|
-
/**
|
|
176
|
-
* Driver selection guide — "which `ImprovementDriver` do I pick, and why?"
|
|
177
|
-
*
|
|
178
|
-
* The substrate ships seven drivers with overlapping shapes. This is the
|
|
179
|
-
* decision table (data, not behavior): each entry says what a driver mutates,
|
|
180
|
-
* how it proposes changes, when to reach for it, and its relative cost.
|
|
181
|
-
* `selectDriver()` turns a goal + surface into a ranked recommendation.
|
|
182
|
-
*
|
|
183
|
-
* Import the actual driver functions from `@tangle-network/agent-eval/campaign`
|
|
184
|
-
* (gepaDriver, skillOptDriver, aceDriver, memoryCurationDriver, haloDriver,
|
|
185
|
-
* traceAnalystDriver, evolutionaryDriver); this module only helps you choose.
|
|
186
|
-
*/
|
|
187
|
-
type DriverName = 'gepa' | 'skillOpt' | 'ace' | 'memoryCuration' | 'halo' | 'traceAnalyst' | 'evolutionary';
|
|
188
|
-
/** The mutable surface a driver targets. */
|
|
189
|
-
type DriverSurface = 'prompt' | 'skill-doc' | 'playbook' | 'memory' | 'any';
|
|
190
|
-
/** How a driver turns evidence into the next candidate. */
|
|
191
|
-
type DriverStrategy = 'reflective-rewrite' | 'anchored-patch' | 'append-only' | 'dedup-curate' | 'analysis-edit' | 'population-mutate';
|
|
192
|
-
/** What a caller is trying to do this run. */
|
|
193
|
-
type DriverGoal = 'explore' | 'refine' | 'accumulate' | 'benchmark';
|
|
194
|
-
interface DriverGuideEntry {
|
|
195
|
-
/** One-line description of the mechanism. */
|
|
196
|
-
summary: string;
|
|
197
|
-
/** The surface the driver edits. */
|
|
198
|
-
surface: DriverSurface;
|
|
199
|
-
/** How it proposes the next candidate. */
|
|
200
|
-
strategy: DriverStrategy;
|
|
201
|
-
/** When to reach for this driver. */
|
|
202
|
-
whenUse: string;
|
|
203
|
-
/** Relative LLM cost per generation. */
|
|
204
|
-
cost: 'low' | 'medium' | 'high';
|
|
205
|
-
/** True when the driver shells out to an external engine (extra setup). */
|
|
206
|
-
external?: boolean;
|
|
207
|
-
}
|
|
208
|
-
declare const DRIVER_GUIDE: Record<DriverName, DriverGuideEntry>;
|
|
209
|
-
interface SelectDriverCriteria {
|
|
210
|
-
/** What you're trying to do this run. */
|
|
211
|
-
goal: DriverGoal;
|
|
212
|
-
/** Restrict to drivers that edit this surface (optional). */
|
|
213
|
-
surface?: DriverSurface;
|
|
214
|
-
}
|
|
215
|
-
interface DriverRecommendation {
|
|
216
|
-
name: DriverName;
|
|
217
|
-
entry: DriverGuideEntry;
|
|
218
|
-
reason: string;
|
|
219
|
-
}
|
|
220
|
-
/**
|
|
221
|
-
* Rank the drivers for a goal (and optional surface filter), best first.
|
|
222
|
-
* Returns the recommendation list, not instances — import the chosen driver
|
|
223
|
-
* function yourself. Always returns at least the goal's primary driver.
|
|
224
|
-
*/
|
|
225
|
-
declare function selectDriver(criteria: SelectDriverCriteria): DriverRecommendation[];
|
|
226
|
-
|
|
227
175
|
/**
|
|
228
176
|
* @experimental
|
|
229
177
|
*
|
|
230
178
|
* `haloDriver` — wraps the REAL halo-engine (Inference.net's hierarchical
|
|
231
179
|
* agentic trace analyzer, `pip install halo-engine`, repo context-labs/halo)
|
|
232
180
|
* as an agent-eval `ImprovementDriver`, so HALO competes head-to-head with
|
|
233
|
-
* `gepaDriver` — and with our own `traceAnalystDriver` — inside
|
|
234
|
-
*
|
|
181
|
+
* `gepaDriver` — and with our own `traceAnalystDriver` — inside `compareDrivers`
|
|
182
|
+
* on identical traces / scenarios / held-out scoring.
|
|
235
183
|
*
|
|
236
|
-
* It PRESERVES halo's actual working usage — `
|
|
237
|
-
* published CLI (`halo <traces.jsonl> -p <prompt> -m <model
|
|
238
|
-
*
|
|
239
|
-
*
|
|
240
|
-
*
|
|
241
|
-
*
|
|
242
|
-
* `gepaDriver` (which also mutates the prompt). The analysis is HALO's; only
|
|
243
|
-
* the surface-application is ours, and it is identical in spirit to how HALO's
|
|
244
|
-
* own loop feeds findings to a coding agent.
|
|
184
|
+
* It PRESERVES halo's actual working usage — `analyze` shells out to the
|
|
185
|
+
* published CLI (`halo <traces.jsonl> -p <prompt> -m <model>`) and uses its real
|
|
186
|
+
* RLM findings verbatim. We do NOT reimplement its analysis; that would make the
|
|
187
|
+
* benchmark meaningless. The materialize/apply pipeline is the shared
|
|
188
|
+
* `analysisEditDriver` — identical to `traceAnalystDriver`, which is what makes
|
|
189
|
+
* the comparison apples-to-apples.
|
|
245
190
|
*
|
|
246
191
|
* Fail-loud: no traces → throw; halo errors → throw; empty findings → throw.
|
|
247
|
-
* Never fabricate a candidate (that would silently flatter or penalize HALO).
|
|
248
192
|
*/
|
|
249
193
|
|
|
250
194
|
interface HaloDriverOptions {
|
|
@@ -259,11 +203,8 @@ interface HaloDriverOptions {
|
|
|
259
203
|
applyModel?: string;
|
|
260
204
|
/** The real halo binary. Default 'halo' (from `pip install halo-engine`). */
|
|
261
205
|
haloBin?: string;
|
|
262
|
-
/**
|
|
263
|
-
*
|
|
264
|
-
* generation — wired by the bench to the captured AppWorld OTLP for the
|
|
265
|
-
* current surface. Returning empty throws (halo has nothing to analyze).
|
|
266
|
-
*/
|
|
206
|
+
/** Resolve the OTLP traces (JSONL string) halo should analyze for THIS
|
|
207
|
+
* generation. Returning empty throws (halo has nothing to analyze). */
|
|
267
208
|
resolveTraces: (ctx: ProposeContext) => string | Promise<string>;
|
|
268
209
|
/** halo's analysis prompt (`-p`). Default targets the failure taxonomy. */
|
|
269
210
|
analysisPrompt?: string;
|
|
@@ -481,28 +422,18 @@ declare function parseSkillPatchResponse(raw: string, maxPatches: number, editBu
|
|
|
481
422
|
*
|
|
482
423
|
* `traceAnalystDriver` — wraps agent-eval's OWN trace-analyst engine
|
|
483
424
|
* (`AnalystRegistry` over the agentic OTLP reader) as an `ImprovementDriver`.
|
|
484
|
-
* It is the symmetric opponent to `haloDriver`: both
|
|
485
|
-
*
|
|
486
|
-
* LLM edit, so a `compareDrivers` lift delta isolates a single
|
|
487
|
-
* ANALYSIS QUALITY. The benchmark answers "is our HALO clone as good
|
|
488
|
-
* real HALO?" as a held-out lift CI, not a vibe.
|
|
489
|
-
*
|
|
490
|
-
* The fairness contract (the only thing that makes the head-to-head honest):
|
|
491
|
-
* - SAME input: both engines read the identical `traces.jsonl` (haloDriver
|
|
492
|
-
* hands it to the halo CLI; this driver wraps it in an `OtlpFileTraceStore`).
|
|
493
|
-
* - SAME application: the apply-step here is byte-for-byte the apply-step in
|
|
494
|
-
* `haloDriver` (same `APPLY_SYSTEM`, same one-shot `callLlm` prompt edit).
|
|
495
|
-
* - ONLY difference: who produced the findings — the real halo-engine vs our
|
|
496
|
-
* `AnalystRegistry` (whose actor prompt is a near-verbatim port of HALO's).
|
|
425
|
+
* It is the symmetric opponent to `haloDriver`: both run the SAME shared
|
|
426
|
+
* `analysisEditDriver` pipeline (materialize identical traces → apply via one
|
|
427
|
+
* identical LLM edit), so a `compareDrivers` lift delta isolates a single
|
|
428
|
+
* variable — ANALYSIS QUALITY. The benchmark answers "is our HALO clone as good
|
|
429
|
+
* as the real HALO?" as a held-out lift CI, not a vibe.
|
|
497
430
|
*
|
|
498
431
|
* Findings come from the REGISTRY (structured `AnalystFinding[]` carrying
|
|
499
|
-
* area / severity / recommended_action),
|
|
500
|
-
*
|
|
501
|
-
* the unstructured escape hatch.
|
|
432
|
+
* area / severity / recommended_action), rendered into the report the shared
|
|
433
|
+
* apply step consumes.
|
|
502
434
|
*
|
|
503
435
|
* Fail-loud: no traces → throw; analyst run errors → throw; zero findings →
|
|
504
|
-
* throw. Never fabricate a candidate
|
|
505
|
-
* our engine relative to HALO).
|
|
436
|
+
* throw. Never fabricate a candidate.
|
|
506
437
|
*/
|
|
507
438
|
|
|
508
439
|
interface TraceAnalystDriverOptions {
|
|
@@ -519,26 +450,15 @@ interface TraceAnalystDriverOptions {
|
|
|
519
450
|
/** Ax provider name. Default 'openai' — works for any OpenAI-compatible base
|
|
520
451
|
* via `apiURL`. Use 'deepseek' to hit DeepSeek's native provider. */
|
|
521
452
|
provider?: string;
|
|
522
|
-
/** Which analyst kinds to run. Default = the full shipped suite
|
|
523
|
-
* (`DEFAULT_TRACE_ANALYST_KINDS`: failure-mode, knowledge-gap,
|
|
524
|
-
* knowledge-poisoning, improvement). Narrow it for cost-parity runs. */
|
|
453
|
+
/** Which analyst kinds to run. Default = the full shipped suite. */
|
|
525
454
|
kinds?: readonly TraceAnalystKindSpec[];
|
|
526
|
-
/**
|
|
527
|
-
*
|
|
528
|
-
* generation — identical contract to `haloDriver.resolveTraces`, wired by
|
|
529
|
-
* the bench to the captured AppWorld OTLP for the current surface. Returning
|
|
530
|
-
* empty throws (the analyst has nothing to read).
|
|
531
|
-
*/
|
|
455
|
+
/** Resolve the OTLP traces (JSONL string) the analyst should read for THIS
|
|
456
|
+
* generation — identical contract to `haloDriver.resolveTraces`. */
|
|
532
457
|
resolveTraces: (ctx: ProposeContext) => string | Promise<string>;
|
|
533
|
-
/**
|
|
534
|
-
*
|
|
535
|
-
* over `kinds`, reading the resolved traces as an `OtlpFileTraceStore`. A
|
|
536
|
-
* consumer may inject a pre-built registry / alternate engine here; the
|
|
537
|
-
* unit suite injects canned findings to exercise the apply path without
|
|
538
|
-
* driving the agentic loop.
|
|
539
|
-
*/
|
|
458
|
+
/** Override the findings producer. Default: the shipped `AnalystRegistry`
|
|
459
|
+
* over `kinds`. The unit suite injects canned findings here. */
|
|
540
460
|
analyze?: (tracePath: string, ctx: ProposeContext) => Promise<ReadonlyArray<AnalystFinding>>;
|
|
541
|
-
/** Test seam: inject a fetch for the apply-step `callLlm
|
|
461
|
+
/** Test seam: inject a fetch for the apply-step `callLlm`. */
|
|
542
462
|
fetchImpl?: LlmClientOptions['fetch'];
|
|
543
463
|
}
|
|
544
464
|
/** Wrap agent-eval's trace-analyst registry as an ImprovementDriver (prompt-tier). */
|
|
@@ -1272,4 +1192,4 @@ declare function gitWorktreeAdapter(opts: GitWorktreeAdapterOptions): WorktreeAd
|
|
|
1272
1192
|
* as a ref under the adapter's worktree dir. */
|
|
1273
1193
|
declare function resolveWorktreePath(surface: CodeSurface, worktreeDir?: string): string;
|
|
1274
1194
|
|
|
1275
|
-
export { type AcceptedEdit, type AceDriverOptions, type AnalystArtifact, type AnalystScenario, type ApplySkillPatchResult, type BuildAnalystSurfaceDispatchOptions, type CampaignBreakdown, CampaignResult, CampaignStorage, CodeSurface, type CompareDriversOptions,
|
|
1195
|
+
export { type AcceptedEdit, type AceDriverOptions, type AnalystArtifact, type AnalystScenario, type ApplySkillPatchResult, type BuildAnalystSurfaceDispatchOptions, type CampaignBreakdown, CampaignResult, CampaignStorage, CodeSurface, type CompareDriversOptions, type DimensionRegression, DispatchContext, type DriverComparison, type DriverEntry, type DriverPairwise, type DriverScore, type FailureModeRecallJudgeOptions, FsLabeledScenarioStore, type FsLabeledScenarioStoreOptions, type GitWorktreeAdapterOptions, type HaloDriverOptions, type HeldoutSignificance, type HeldoutSignificanceOptions, ImprovementDriver, JudgeConfig, JudgeScore, LabelTrust, LabeledScenarioRecord, LabeledScenarioSampleArgs, LabeledScenarioSource, LabeledScenarioStore, LabeledScenarioStoreError, LabeledScenarioWrite, type MemoryCurationDriverOptions, MutableSurface, type OptimizerEntryConfig, type PairedHoldout, type PlaybackContext, type PlaybackDriver, type PlaybackStep, type ProfileDispatchFn, ProfileMatrixError, type ProfileSummary, ProposeContext, type ProposePatchesArgs, type RejectedEdit, RunCampaignOptions, RunImprovementLoopOptions, type RunProfileMatrixOptions, type RunProfileMatrixResult, type RunSkillOptOptions, type RunSkillOptResult, Scenario, type ScenarioRollup, type ScoreboardRenderOptions, type ScoreboardRow, type ScoreboardSummary, type SkillOptDriver, type SkillOptDriverOptions, type SkillOptEpochRecord, type SkillOptEvidence, type SkillPatch, type SkillPatchOp, SkillPatchParseError, type SkillPatchRejection, type TraceAnalystDriverOptions, type UserStory, type UserStoryVerdict, type Worktree, type WorktreeAdapter, WorktreeAdapterError, aceDriver, applySkillPatch, buildAnalystSurfaceDispatch, campaignBreakdown, campaignMeanComposite, compareDrivers, detectScale, dimensionRegressions, failureModeRecallJudge, gepaParetoEntry, gepaReflectionEntry, gitWorktreeAdapter, haloDriver, heldoutSignificance, makePlaybackDispatch, memoryCurationDriver, pairHoldout, parseSkillPatchResponse, patchEditCount, renderScoreboardMarkdown, resolveWorktreePath, runProfileMatrix, runSkillOpt, scoreUserStory, scoreboardSummary, skillOptDriver, skillOptEntry, traceAnalystDriver, userStoryScoreboard };
|