@tangle-network/agent-eval 0.79.0 → 0.81.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (107) hide show
  1. package/CHANGELOG.md +14 -0
  2. package/README.md +101 -169
  3. package/dist/adapters/http.d.ts +2 -2
  4. package/dist/adapters/langchain.d.ts +2 -2
  5. package/dist/adapters/otel.d.ts +4 -4
  6. package/dist/{agent-profile-aSEaJ9Pl.d.ts → agent-profile-D0PBIWlV.d.ts} +14 -2
  7. package/dist/analyst/index.d.ts +10 -10
  8. package/dist/analyst/index.js +3 -3
  9. package/dist/{analyst-t7zZS3TV.d.ts → analyst-C8HHvfJp.d.ts} +1 -1
  10. package/dist/belief-state/index.d.ts +524 -0
  11. package/dist/belief-state/index.js +1862 -0
  12. package/dist/belief-state/index.js.map +1 -0
  13. package/dist/benchmarks/index.d.ts +2 -2
  14. package/dist/calibration-Cpr3WaX3.d.ts +101 -0
  15. package/dist/campaign/index.d.ts +40 -120
  16. package/dist/campaign/index.js +129 -238
  17. package/dist/campaign/index.js.map +1 -1
  18. package/dist/chunk-4DIJWVUT.js +131 -0
  19. package/dist/chunk-4DIJWVUT.js.map +1 -0
  20. package/dist/{chunk-RPLZ4OIB.js → chunk-BABOZOSN.js} +7 -4
  21. package/dist/{chunk-RPLZ4OIB.js.map → chunk-BABOZOSN.js.map} +1 -1
  22. package/dist/{chunk-IHDHUN2X.js → chunk-CVVHBFGN.js} +3 -3
  23. package/dist/chunk-CVVHBFGN.js.map +1 -0
  24. package/dist/{chunk-B26KI423.js → chunk-FZWAFVAA.js} +2 -2
  25. package/dist/{chunk-ITBRCT73.js → chunk-IDVBLYCY.js} +2 -10
  26. package/dist/chunk-IDVBLYCY.js.map +1 -0
  27. package/dist/{chunk-5LVWPNS5.js → chunk-L5G7OUKD.js} +4 -4
  28. package/dist/{chunk-5LVWPNS5.js.map → chunk-L5G7OUKD.js.map} +1 -1
  29. package/dist/chunk-NPCTHQIO.js +91 -0
  30. package/dist/chunk-NPCTHQIO.js.map +1 -0
  31. package/dist/{chunk-GXHLRXDI.js → chunk-OTYQPHPL.js} +4 -4
  32. package/dist/{chunk-6REHLN5J.js → chunk-QS3RBQPI.js} +2 -2
  33. package/dist/{chunk-GWGO2K6Y.js → chunk-RBNA5AZT.js} +2 -2
  34. package/dist/chunk-S42AWHMP.js +697 -0
  35. package/dist/chunk-S42AWHMP.js.map +1 -0
  36. package/dist/chunk-VI2UW6B6.js +162 -0
  37. package/dist/chunk-VI2UW6B6.js.map +1 -0
  38. package/dist/{chunk-CF67I6QY.js → chunk-VIDQF3F5.js} +2 -2
  39. package/dist/{chunk-XXNIODOM.js → chunk-WJL2NJXN.js} +3 -3
  40. package/dist/{chunk-LB2UOI5F.js → chunk-YGYXHNAQ.js} +47 -160
  41. package/dist/chunk-YGYXHNAQ.js.map +1 -0
  42. package/dist/{chunk-KX6F6NCG.js → chunk-Z7VFTS2J.js} +2 -2
  43. package/dist/{chunk-ZPSKPT3V.js → chunk-ZZ2HOPME.js} +5 -2
  44. package/dist/chunk-ZZ2HOPME.js.map +1 -0
  45. package/dist/cli.js +2 -2
  46. package/dist/code-agent-session-BRXmavYv.d.ts +80 -0
  47. package/dist/contract/index.d.ts +132 -18
  48. package/dist/contract/index.js +139 -6
  49. package/dist/contract/index.js.map +1 -1
  50. package/dist/{control-CehLtoET.d.ts → control-GeE8OhpN.d.ts} +1 -1
  51. package/dist/control.d.ts +2 -2
  52. package/dist/governance/index.d.ts +1 -1
  53. package/dist/hosted/index.d.ts +4 -4
  54. package/dist/{index-B1RKber3.d.ts → index-DE3RXAXD.d.ts} +1 -1
  55. package/dist/index.d.ts +79 -288
  56. package/dist/index.js +87 -410
  57. package/dist/index.js.map +1 -1
  58. package/dist/{insight-report-dlpEzQDi.d.ts → insight-report-3ADTfClO.d.ts} +1 -1
  59. package/dist/{kind-factory-DqV2t1Xk.d.ts → kind-factory-CVecZZG_.d.ts} +2 -2
  60. package/dist/{llm-client-DbjLfz-K.d.ts → llm-client-CuUg2Mn3.d.ts} +1 -1
  61. package/dist/meta-eval/index.d.ts +6 -99
  62. package/dist/meta-eval/index.js +7 -76
  63. package/dist/meta-eval/index.js.map +1 -1
  64. package/dist/off-policy-DiwuKKg7.d.ts +132 -0
  65. package/dist/openapi.json +1 -1
  66. package/dist/{outcome-store-D6KWmYvj.d.ts → outcome-store-rnXLEqSn.d.ts} +1 -1
  67. package/dist/pipelines/index.js +2 -2
  68. package/dist/{provenance-CEAJI9rm.d.ts → provenance-B9Q4886D.d.ts} +4 -4
  69. package/dist/{registry-BmEuU94S.d.ts → registry-DrEQ3Luj.d.ts} +2 -2
  70. package/dist/{release-report-CXXZlR8g.d.ts → release-report-hlNtD12q.d.ts} +2 -2
  71. package/dist/reporting.d.ts +6 -6
  72. package/dist/reporting.js +3 -3
  73. package/dist/{researcher-rInLj9De.d.ts → researcher-BLPHBbNV.d.ts} +3 -3
  74. package/dist/rl.d.ts +11 -141
  75. package/dist/rl.js +10 -124
  76. package/dist/rl.js.map +1 -1
  77. package/dist/{rubric-predictive-validity-CWyWWLBg.d.ts → rubric-predictive-validity-CnEl9Jc8.d.ts} +2 -2
  78. package/dist/{run-campaign-OVEZF24D.js → run-campaign-4Y5V5CN3.js} +3 -3
  79. package/dist/{run-improvement-loop-Bgu4C59E.d.ts → run-improvement-loop-D6PZOoQL.d.ts} +2 -2
  80. package/dist/{run-record-sItO5ftF.d.ts → run-record-De9VarXR.d.ts} +1 -1
  81. package/dist/{semantic-concept-judge-Du4ZVyef.d.ts → semantic-concept-judge-DIEgr_6v.d.ts} +6 -6
  82. package/dist/{statistics-B7yCbi9i.d.ts → statistics-CnC1FMbx.d.ts} +3 -7
  83. package/dist/{store-GmBE2pZZ.d.ts → store-C1YxJDEK.d.ts} +1 -1
  84. package/dist/{summary-report-BTaXq1TS.d.ts → summary-report-Db0dDSWP.d.ts} +1 -1
  85. package/dist/traces.d.ts +5 -5
  86. package/dist/{types-DRvV0zRo.d.ts → types-Cu3u_x59.d.ts} +3 -3
  87. package/dist/{types-QHG0KnkF.d.ts → types-D7lLRYe9.d.ts} +2 -2
  88. package/dist/wire/index.js +2 -2
  89. package/dist/workflow/index.d.ts +7 -7
  90. package/dist/workflow/index.js +1 -1
  91. package/docs/concepts.md +1 -0
  92. package/docs/research/belief-state-agent-eval-roadmap.md +590 -0
  93. package/docs/research/research-roadmap.md +1 -0
  94. package/docs/self-improvement-map.md +111 -0
  95. package/package.json +7 -2
  96. package/dist/chunk-IHDHUN2X.js.map +0 -1
  97. package/dist/chunk-ITBRCT73.js.map +0 -1
  98. package/dist/chunk-LB2UOI5F.js.map +0 -1
  99. package/dist/chunk-ZPSKPT3V.js.map +0 -1
  100. /package/dist/{chunk-B26KI423.js.map → chunk-FZWAFVAA.js.map} +0 -0
  101. /package/dist/{chunk-GXHLRXDI.js.map → chunk-OTYQPHPL.js.map} +0 -0
  102. /package/dist/{chunk-6REHLN5J.js.map → chunk-QS3RBQPI.js.map} +0 -0
  103. /package/dist/{chunk-GWGO2K6Y.js.map → chunk-RBNA5AZT.js.map} +0 -0
  104. /package/dist/{chunk-CF67I6QY.js.map → chunk-VIDQF3F5.js.map} +0 -0
  105. /package/dist/{chunk-XXNIODOM.js.map → chunk-WJL2NJXN.js.map} +0 -0
  106. /package/dist/{chunk-KX6F6NCG.js.map → chunk-Z7VFTS2J.js.map} +0 -0
  107. /package/dist/{run-campaign-OVEZF24D.js.map → run-campaign-4Y5V5CN3.js.map} +0 -0
@@ -1,26 +1,26 @@
1
- import { A as AnalyzeTracesOptions, a as AnalyzeTracesInput, b as AnalyzeTracesResult } from '../analyst-t7zZS3TV.js';
2
- import { S as Scenario, M as MutableSurface, b as DispatchContext, a as JudgeConfig, I as ImprovementDriver, q as ProposeContext, J as JudgeScore, L as LabeledScenarioStore, r as LabeledScenarioWrite, s as LabeledScenarioSampleArgs, t as LabeledScenarioRecord, u as LabelTrust, v as LabeledScenarioSource, f as CampaignResult, h as CodeSurface } from '../types-QHG0KnkF.js';
3
- export { C as CampaignAggregates, c as CampaignArtifactWriter, d as CampaignCellResult, e as CampaignCostMeter, w as CampaignTokenUsage, g as CampaignTraceWriter, D as DispatchFn, G as Gate, i as GateContext, j as GateDecision, k as GateResult, l as GenerationCandidate, m as GenerationRecord, x as JudgeAggregate, n as JudgeDimension, o as Mutator, O as OptimizerConfig, P as ParetoParent, y as ProposedCandidate, R as RedactionStatus, z as ScenarioAggregate, p as SessionScript, T as TraceSpan, A as isProposedCandidate, B as labelTrustRank } from '../types-QHG0KnkF.js';
4
- import { a as RunCampaignOptions, b as RunImprovementLoopOptions, C as CampaignStorage } from '../run-improvement-loop-Bgu4C59E.js';
5
- export { d as GepaDriverConstraints, G as GepaDriverOptions, O as OpenAutoPrOptions, e as OpenAutoPrResult, R as RunImprovementLoopResult, h as RunOptimizationOptions, j as RunOptimizationResult, k as countSentenceEdits, l as defaultRenderDiff, m as extractH2Sections, f as fsCampaignStorage, g as gepaDriver, i as inMemoryCampaignStorage, o as openAutoPr, r as runCampaign, c as runImprovementLoop, n as runOptimization, s as surfaceHash } from '../run-improvement-loop-Bgu4C59E.js';
6
- export { A as AxisEvidence, a as AxisVerdict, B as BuildEvidenceVectorOptions, k as BuildLoopProvenanceArgs, D as DefaultProductionGateOptions, l as EmitLoopProvenanceArgs, m as EmitLoopProvenanceResult, E as EvidenceVector, b as EvolutionaryDriverOptions, H as HeldOutGateOptions, n as LoopProvenanceBackend, o as LoopProvenanceCandidate, L as LoopProvenanceRecord, O as ObjectiveSource, P as ParetoSignificanceGateOptions, c as PromotionObjective, d as PromotionPolicy, R as RunEvalOptions, e as buildEvidenceVector, q as buildLoopProvenanceRecord, f as composeGate, g as defaultProductionGate, s as emitLoopProvenance, h as evolutionaryDriver, i as heldOutGate, t as loopProvenanceSpans, p as paretoPolicy, j as paretoSignificanceGate, u as provenanceRecordPath, v as provenanceSpansPath, r as runEval, w as surfaceContentHash } from '../provenance-CEAJI9rm.js';
7
- import { L as LlmClientOptions } from '../llm-client-DbjLfz-K.js';
8
- import { c as TraceAnalystKindSpec } from '../kind-factory-DqV2t1Xk.js';
9
- import { c as AnalystFinding } from '../types-DRvV0zRo.js';
10
- import { a as PairedBootstrapResult } from '../statistics-B7yCbi9i.js';
11
- import { A as AgentProfile, B as BackendIntegrityReport, C as CompletionRequirement, R as RuntimeEventLike, a as CompletionVerdict, P as ProducedState, b as CorrectnessChecker } from '../agent-profile-aSEaJ9Pl.js';
1
+ import { A as AnalyzeTracesOptions, a as AnalyzeTracesInput, b as AnalyzeTracesResult } from '../analyst-C8HHvfJp.js';
2
+ import { S as Scenario, M as MutableSurface, b as DispatchContext, a as JudgeConfig, I as ImprovementDriver, q as ProposeContext, J as JudgeScore, L as LabeledScenarioStore, r as LabeledScenarioWrite, s as LabeledScenarioSampleArgs, t as LabeledScenarioRecord, u as LabelTrust, v as LabeledScenarioSource, g as CampaignResult, i as CodeSurface } from '../types-D7lLRYe9.js';
3
+ export { C as CampaignAggregates, d as CampaignArtifactWriter, e as CampaignCellResult, f as CampaignCostMeter, w as CampaignTokenUsage, h as CampaignTraceWriter, D as DispatchFn, G as Gate, j as GateContext, c as GateDecision, k as GateResult, l as GenerationCandidate, m as GenerationRecord, x as JudgeAggregate, n as JudgeDimension, o as Mutator, O as OptimizerConfig, P as ParetoParent, y as ProposedCandidate, R as RedactionStatus, z as ScenarioAggregate, p as SessionScript, T as TraceSpan, A as isProposedCandidate, B as labelTrustRank } from '../types-D7lLRYe9.js';
4
+ import { a as RunCampaignOptions, b as RunImprovementLoopOptions, C as CampaignStorage } from '../run-improvement-loop-D6PZOoQL.js';
5
+ export { d as GepaDriverConstraints, G as GepaDriverOptions, O as OpenAutoPrOptions, e as OpenAutoPrResult, R as RunImprovementLoopResult, h as RunOptimizationOptions, j as RunOptimizationResult, k as countSentenceEdits, l as defaultRenderDiff, m as extractH2Sections, f as fsCampaignStorage, g as gepaDriver, i as inMemoryCampaignStorage, o as openAutoPr, r as runCampaign, c as runImprovementLoop, n as runOptimization, s as surfaceHash } from '../run-improvement-loop-D6PZOoQL.js';
6
+ export { A as AxisEvidence, a as AxisVerdict, B as BuildEvidenceVectorOptions, k as BuildLoopProvenanceArgs, D as DefaultProductionGateOptions, l as EmitLoopProvenanceArgs, m as EmitLoopProvenanceResult, E as EvidenceVector, b as EvolutionaryDriverOptions, H as HeldOutGateOptions, n as LoopProvenanceBackend, o as LoopProvenanceCandidate, L as LoopProvenanceRecord, O as ObjectiveSource, P as ParetoSignificanceGateOptions, c as PromotionObjective, d as PromotionPolicy, R as RunEvalOptions, e as buildEvidenceVector, q as buildLoopProvenanceRecord, f as composeGate, g as defaultProductionGate, s as emitLoopProvenance, h as evolutionaryDriver, i as heldOutGate, t as loopProvenanceSpans, p as paretoPolicy, j as paretoSignificanceGate, u as provenanceRecordPath, v as provenanceSpansPath, r as runEval, w as surfaceContentHash } from '../provenance-B9Q4886D.js';
7
+ import { L as LlmClientOptions } from '../llm-client-CuUg2Mn3.js';
8
+ import { c as TraceAnalystKindSpec } from '../kind-factory-CVecZZG_.js';
9
+ import { c as AnalystFinding } from '../types-Cu3u_x59.js';
10
+ import { a as PairedBootstrapResult } from '../statistics-CnC1FMbx.js';
11
+ import { A as AgentProfile, B as BackendIntegrityReport, C as CompletionRequirement, R as RuntimeEventLike, a as CompletionVerdict, P as ProducedState, b as CorrectnessChecker } from '../agent-profile-D0PBIWlV.js';
12
12
  import { A as AgentEvalError } from '../errors-Dwqw-T_m.js';
13
- import { b as RunSplitTag, R as RunRecord } from '../run-record-sItO5ftF.js';
13
+ import { a as RunSplitTag, R as RunRecord } from '../run-record-De9VarXR.js';
14
14
  import '@ax-llm/ax';
15
- import '../store-GmBE2pZZ.js';
15
+ import '../store-C1YxJDEK.js';
16
16
  import '../red-team-DW9Ca_tj.js';
17
17
  import '../dataset-B2kL-fSM.js';
18
18
  import '../store-CKUAgsJz.js';
19
19
  import '../schema-m0gsnbt3.js';
20
20
  import '../pareto-E-pembql.js';
21
21
  import '../hosted/index.js';
22
- import '../insight-report-dlpEzQDi.js';
23
- import '../summary-report-BTaXq1TS.js';
22
+ import '../insight-report-3ADTfClO.js';
23
+ import '../summary-report-Db0dDSWP.js';
24
24
  import '../failure-cluster-CL7IVgkJ.js';
25
25
  import '../judge-calibration-DilmB3Ml.js';
26
26
  import '../raw-provider-sink-C46HDghv.js';
@@ -172,79 +172,23 @@ interface AceDriverOptions {
172
172
  }
173
173
  declare function aceDriver(opts?: AceDriverOptions): ImprovementDriver;
174
174
 
175
- /**
176
- * Driver selection guide — "which `ImprovementDriver` do I pick, and why?"
177
- *
178
- * The substrate ships seven drivers with overlapping shapes. This is the
179
- * decision table (data, not behavior): each entry says what a driver mutates,
180
- * how it proposes changes, when to reach for it, and its relative cost.
181
- * `selectDriver()` turns a goal + surface into a ranked recommendation.
182
- *
183
- * Import the actual driver functions from `@tangle-network/agent-eval/campaign`
184
- * (gepaDriver, skillOptDriver, aceDriver, memoryCurationDriver, haloDriver,
185
- * traceAnalystDriver, evolutionaryDriver); this module only helps you choose.
186
- */
187
- type DriverName = 'gepa' | 'skillOpt' | 'ace' | 'memoryCuration' | 'halo' | 'traceAnalyst' | 'evolutionary';
188
- /** The mutable surface a driver targets. */
189
- type DriverSurface = 'prompt' | 'skill-doc' | 'playbook' | 'memory' | 'any';
190
- /** How a driver turns evidence into the next candidate. */
191
- type DriverStrategy = 'reflective-rewrite' | 'anchored-patch' | 'append-only' | 'dedup-curate' | 'analysis-edit' | 'population-mutate';
192
- /** What a caller is trying to do this run. */
193
- type DriverGoal = 'explore' | 'refine' | 'accumulate' | 'benchmark';
194
- interface DriverGuideEntry {
195
- /** One-line description of the mechanism. */
196
- summary: string;
197
- /** The surface the driver edits. */
198
- surface: DriverSurface;
199
- /** How it proposes the next candidate. */
200
- strategy: DriverStrategy;
201
- /** When to reach for this driver. */
202
- whenUse: string;
203
- /** Relative LLM cost per generation. */
204
- cost: 'low' | 'medium' | 'high';
205
- /** True when the driver shells out to an external engine (extra setup). */
206
- external?: boolean;
207
- }
208
- declare const DRIVER_GUIDE: Record<DriverName, DriverGuideEntry>;
209
- interface SelectDriverCriteria {
210
- /** What you're trying to do this run. */
211
- goal: DriverGoal;
212
- /** Restrict to drivers that edit this surface (optional). */
213
- surface?: DriverSurface;
214
- }
215
- interface DriverRecommendation {
216
- name: DriverName;
217
- entry: DriverGuideEntry;
218
- reason: string;
219
- }
220
- /**
221
- * Rank the drivers for a goal (and optional surface filter), best first.
222
- * Returns the recommendation list, not instances — import the chosen driver
223
- * function yourself. Always returns at least the goal's primary driver.
224
- */
225
- declare function selectDriver(criteria: SelectDriverCriteria): DriverRecommendation[];
226
-
227
175
  /**
228
176
  * @experimental
229
177
  *
230
178
  * `haloDriver` — wraps the REAL halo-engine (Inference.net's hierarchical
231
179
  * agentic trace analyzer, `pip install halo-engine`, repo context-labs/halo)
232
180
  * as an agent-eval `ImprovementDriver`, so HALO competes head-to-head with
233
- * `gepaDriver` — and with our own `traceAnalystDriver` — inside
234
- * `compareDrivers` on identical traces / scenarios / held-out scoring.
181
+ * `gepaDriver` — and with our own `traceAnalystDriver` — inside `compareDrivers`
182
+ * on identical traces / scenarios / held-out scoring.
235
183
  *
236
- * It PRESERVES halo's actual working usage — `propose()` shells out to the
237
- * published CLI (`halo <traces.jsonl> -p <prompt> -m <model> --base-url
238
- * --api-key`) and uses its real RLM findings verbatim. We do NOT reimplement
239
- * its analysis; that would make the benchmark meaningless. The only adaptation
240
- * is applying HALO's findings to the current prompt surface via one LLM edit —
241
- * exactly what makes the comparison prompt-tier apples-to-apples with
242
- * `gepaDriver` (which also mutates the prompt). The analysis is HALO's; only
243
- * the surface-application is ours, and it is identical in spirit to how HALO's
244
- * own loop feeds findings to a coding agent.
184
+ * It PRESERVES halo's actual working usage — `analyze` shells out to the
185
+ * published CLI (`halo <traces.jsonl> -p <prompt> -m <model>`) and uses its real
186
+ * RLM findings verbatim. We do NOT reimplement its analysis; that would make the
187
+ * benchmark meaningless. The materialize/apply pipeline is the shared
188
+ * `analysisEditDriver` identical to `traceAnalystDriver`, which is what makes
189
+ * the comparison apples-to-apples.
245
190
  *
246
191
  * Fail-loud: no traces → throw; halo errors → throw; empty findings → throw.
247
- * Never fabricate a candidate (that would silently flatter or penalize HALO).
248
192
  */
249
193
 
250
194
  interface HaloDriverOptions {
@@ -259,11 +203,8 @@ interface HaloDriverOptions {
259
203
  applyModel?: string;
260
204
  /** The real halo binary. Default 'halo' (from `pip install halo-engine`). */
261
205
  haloBin?: string;
262
- /**
263
- * Resolve the OTLP traces (JSONL string) halo should analyze for THIS
264
- * generation — wired by the bench to the captured AppWorld OTLP for the
265
- * current surface. Returning empty throws (halo has nothing to analyze).
266
- */
206
+ /** Resolve the OTLP traces (JSONL string) halo should analyze for THIS
207
+ * generation. Returning empty throws (halo has nothing to analyze). */
267
208
  resolveTraces: (ctx: ProposeContext) => string | Promise<string>;
268
209
  /** halo's analysis prompt (`-p`). Default targets the failure taxonomy. */
269
210
  analysisPrompt?: string;
@@ -481,28 +422,18 @@ declare function parseSkillPatchResponse(raw: string, maxPatches: number, editBu
481
422
  *
482
423
  * `traceAnalystDriver` — wraps agent-eval's OWN trace-analyst engine
483
424
  * (`AnalystRegistry` over the agentic OTLP reader) as an `ImprovementDriver`.
484
- * It is the symmetric opponent to `haloDriver`: both consume the SAME OTLP
485
- * corpus and apply their findings to the prompt surface via one IDENTICAL
486
- * LLM edit, so a `compareDrivers` lift delta isolates a single variable —
487
- * ANALYSIS QUALITY. The benchmark answers "is our HALO clone as good as the
488
- * real HALO?" as a held-out lift CI, not a vibe.
489
- *
490
- * The fairness contract (the only thing that makes the head-to-head honest):
491
- * - SAME input: both engines read the identical `traces.jsonl` (haloDriver
492
- * hands it to the halo CLI; this driver wraps it in an `OtlpFileTraceStore`).
493
- * - SAME application: the apply-step here is byte-for-byte the apply-step in
494
- * `haloDriver` (same `APPLY_SYSTEM`, same one-shot `callLlm` prompt edit).
495
- * - ONLY difference: who produced the findings — the real halo-engine vs our
496
- * `AnalystRegistry` (whose actor prompt is a near-verbatim port of HALO's).
425
+ * It is the symmetric opponent to `haloDriver`: both run the SAME shared
426
+ * `analysisEditDriver` pipeline (materialize identical traces apply via one
427
+ * identical LLM edit), so a `compareDrivers` lift delta isolates a single
428
+ * variable — ANALYSIS QUALITY. The benchmark answers "is our HALO clone as good
429
+ * as the real HALO?" as a held-out lift CI, not a vibe.
497
430
  *
498
431
  * Findings come from the REGISTRY (structured `AnalystFinding[]` carrying
499
- * area / severity / recommended_action), NOT bare `analyzeTraces` (which emits
500
- * `string[]`). The registry is the productized engine; raw `analyzeTraces` is
501
- * the unstructured escape hatch.
432
+ * area / severity / recommended_action), rendered into the report the shared
433
+ * apply step consumes.
502
434
  *
503
435
  * Fail-loud: no traces → throw; analyst run errors → throw; zero findings →
504
- * throw. Never fabricate a candidate (that would silently flatter or penalize
505
- * our engine relative to HALO).
436
+ * throw. Never fabricate a candidate.
506
437
  */
507
438
 
508
439
  interface TraceAnalystDriverOptions {
@@ -519,26 +450,15 @@ interface TraceAnalystDriverOptions {
519
450
  /** Ax provider name. Default 'openai' — works for any OpenAI-compatible base
520
451
  * via `apiURL`. Use 'deepseek' to hit DeepSeek's native provider. */
521
452
  provider?: string;
522
- /** Which analyst kinds to run. Default = the full shipped suite
523
- * (`DEFAULT_TRACE_ANALYST_KINDS`: failure-mode, knowledge-gap,
524
- * knowledge-poisoning, improvement). Narrow it for cost-parity runs. */
453
+ /** Which analyst kinds to run. Default = the full shipped suite. */
525
454
  kinds?: readonly TraceAnalystKindSpec[];
526
- /**
527
- * Resolve the OTLP traces (JSONL string) the analyst should read for THIS
528
- * generation — identical contract to `haloDriver.resolveTraces`, wired by
529
- * the bench to the captured AppWorld OTLP for the current surface. Returning
530
- * empty throws (the analyst has nothing to read).
531
- */
455
+ /** Resolve the OTLP traces (JSONL string) the analyst should read for THIS
456
+ * generation identical contract to `haloDriver.resolveTraces`. */
532
457
  resolveTraces: (ctx: ProposeContext) => string | Promise<string>;
533
- /**
534
- * Override the findings producer. Default: the shipped `AnalystRegistry`
535
- * over `kinds`, reading the resolved traces as an `OtlpFileTraceStore`. A
536
- * consumer may inject a pre-built registry / alternate engine here; the
537
- * unit suite injects canned findings to exercise the apply path without
538
- * driving the agentic loop.
539
- */
458
+ /** Override the findings producer. Default: the shipped `AnalystRegistry`
459
+ * over `kinds`. The unit suite injects canned findings here. */
540
460
  analyze?: (tracePath: string, ctx: ProposeContext) => Promise<ReadonlyArray<AnalystFinding>>;
541
- /** Test seam: inject a fetch for the apply-step `callLlm` (no network in unit tests). */
461
+ /** Test seam: inject a fetch for the apply-step `callLlm`. */
542
462
  fetchImpl?: LlmClientOptions['fetch'];
543
463
  }
544
464
  /** Wrap agent-eval's trace-analyst registry as an ImprovementDriver (prompt-tier). */
@@ -1272,4 +1192,4 @@ declare function gitWorktreeAdapter(opts: GitWorktreeAdapterOptions): WorktreeAd
1272
1192
  * as a ref under the adapter's worktree dir. */
1273
1193
  declare function resolveWorktreePath(surface: CodeSurface, worktreeDir?: string): string;
1274
1194
 
1275
- export { type AcceptedEdit, type AceDriverOptions, type AnalystArtifact, type AnalystScenario, type ApplySkillPatchResult, type BuildAnalystSurfaceDispatchOptions, type CampaignBreakdown, CampaignResult, CampaignStorage, CodeSurface, type CompareDriversOptions, DRIVER_GUIDE, type DimensionRegression, DispatchContext, type DriverComparison, type DriverEntry, type DriverGoal, type DriverGuideEntry, type DriverName, type DriverPairwise, type DriverRecommendation, type DriverScore, type DriverStrategy, type DriverSurface, type FailureModeRecallJudgeOptions, FsLabeledScenarioStore, type FsLabeledScenarioStoreOptions, type GitWorktreeAdapterOptions, type HaloDriverOptions, type HeldoutSignificance, type HeldoutSignificanceOptions, ImprovementDriver, JudgeConfig, JudgeScore, LabelTrust, LabeledScenarioRecord, LabeledScenarioSampleArgs, LabeledScenarioSource, LabeledScenarioStore, LabeledScenarioStoreError, LabeledScenarioWrite, type MemoryCurationDriverOptions, MutableSurface, type OptimizerEntryConfig, type PairedHoldout, type PlaybackContext, type PlaybackDriver, type PlaybackStep, type ProfileDispatchFn, ProfileMatrixError, type ProfileSummary, ProposeContext, type ProposePatchesArgs, type RejectedEdit, RunCampaignOptions, RunImprovementLoopOptions, type RunProfileMatrixOptions, type RunProfileMatrixResult, type RunSkillOptOptions, type RunSkillOptResult, Scenario, type ScenarioRollup, type ScoreboardRenderOptions, type ScoreboardRow, type ScoreboardSummary, type SelectDriverCriteria, type SkillOptDriver, type SkillOptDriverOptions, type SkillOptEpochRecord, type SkillOptEvidence, type SkillPatch, type SkillPatchOp, SkillPatchParseError, type SkillPatchRejection, type TraceAnalystDriverOptions, type UserStory, type UserStoryVerdict, type Worktree, type WorktreeAdapter, WorktreeAdapterError, aceDriver, applySkillPatch, buildAnalystSurfaceDispatch, campaignBreakdown, campaignMeanComposite, compareDrivers, detectScale, dimensionRegressions, failureModeRecallJudge, gepaParetoEntry, gepaReflectionEntry, gitWorktreeAdapter, haloDriver, heldoutSignificance, makePlaybackDispatch, memoryCurationDriver, pairHoldout, parseSkillPatchResponse, patchEditCount, renderScoreboardMarkdown, resolveWorktreePath, runProfileMatrix, runSkillOpt, scoreUserStory, scoreboardSummary, selectDriver, skillOptDriver, skillOptEntry, traceAnalystDriver, userStoryScoreboard };
1195
+ export { type AcceptedEdit, type AceDriverOptions, type AnalystArtifact, type AnalystScenario, type ApplySkillPatchResult, type BuildAnalystSurfaceDispatchOptions, type CampaignBreakdown, CampaignResult, CampaignStorage, CodeSurface, type CompareDriversOptions, type DimensionRegression, DispatchContext, type DriverComparison, type DriverEntry, type DriverPairwise, type DriverScore, type FailureModeRecallJudgeOptions, FsLabeledScenarioStore, type FsLabeledScenarioStoreOptions, type GitWorktreeAdapterOptions, type HaloDriverOptions, type HeldoutSignificance, type HeldoutSignificanceOptions, ImprovementDriver, JudgeConfig, JudgeScore, LabelTrust, LabeledScenarioRecord, LabeledScenarioSampleArgs, LabeledScenarioSource, LabeledScenarioStore, LabeledScenarioStoreError, LabeledScenarioWrite, type MemoryCurationDriverOptions, MutableSurface, type OptimizerEntryConfig, type PairedHoldout, type PlaybackContext, type PlaybackDriver, type PlaybackStep, type ProfileDispatchFn, ProfileMatrixError, type ProfileSummary, ProposeContext, type ProposePatchesArgs, type RejectedEdit, RunCampaignOptions, RunImprovementLoopOptions, type RunProfileMatrixOptions, type RunProfileMatrixResult, type RunSkillOptOptions, type RunSkillOptResult, Scenario, type ScenarioRollup, type ScoreboardRenderOptions, type ScoreboardRow, type ScoreboardSummary, type SkillOptDriver, type SkillOptDriverOptions, type SkillOptEpochRecord, type SkillOptEvidence, type SkillPatch, type SkillPatchOp, SkillPatchParseError, type SkillPatchRejection, type TraceAnalystDriverOptions, type UserStory, type UserStoryVerdict, type Worktree, type WorktreeAdapter, WorktreeAdapterError, aceDriver, applySkillPatch, buildAnalystSurfaceDispatch, campaignBreakdown, campaignMeanComposite, compareDrivers, detectScale, dimensionRegressions, failureModeRecallJudge, gepaParetoEntry, gepaReflectionEntry, gitWorktreeAdapter, haloDriver, heldoutSignificance, makePlaybackDispatch, memoryCurationDriver, pairHoldout, parseSkillPatchResponse, patchEditCount, renderScoreboardMarkdown, resolveWorktreePath, runProfileMatrix, runSkillOpt, scoreUserStory, scoreboardSummary, skillOptDriver, skillOptEntry, traceAnalystDriver, userStoryScoreboard };