@tangle-network/agent-eval 0.115.2 → 0.116.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (78) hide show
  1. package/CHANGELOG.md +34 -0
  2. package/dist/analyst/index.d.ts +8 -10
  3. package/dist/analyst/index.js +28 -23
  4. package/dist/analyst/index.js.map +1 -1
  5. package/dist/{analyst-C8HHvfJp.d.ts → analyst-CFBc14Wc.d.ts} +1 -1
  6. package/dist/{analyze-runs-BYHg6Irm.d.ts → analyze-runs-0rz_m29H.d.ts} +3 -3
  7. package/dist/belief-state/index.d.ts +3 -3
  8. package/dist/benchmarks/index.d.ts +7 -3
  9. package/dist/benchmarks/index.js +5 -5
  10. package/dist/campaign/index.d.ts +212 -23
  11. package/dist/campaign/index.js +20 -5
  12. package/dist/{chunk-N6MTC3GK.js → chunk-3274WNK7.js} +428 -94
  13. package/dist/chunk-3274WNK7.js.map +1 -0
  14. package/dist/{chunk-DRPIZQIT.js → chunk-4D5RVB3W.js} +2 -2
  15. package/dist/{chunk-LVTGFSHF.js → chunk-7GKEAIAD.js} +2 -2
  16. package/dist/{chunk-DWLIGZBX.js → chunk-CIUOICJT.js} +748 -3
  17. package/dist/chunk-CIUOICJT.js.map +1 -0
  18. package/dist/{chunk-5NVBGKPH.js → chunk-GSW3OBHK.js} +1284 -182
  19. package/dist/chunk-GSW3OBHK.js.map +1 -0
  20. package/dist/{chunk-FUCQVFMU.js → chunk-GY4SYVPJ.js} +12 -3
  21. package/dist/chunk-GY4SYVPJ.js.map +1 -0
  22. package/dist/chunk-MPHTT5HE.js +74 -0
  23. package/dist/chunk-MPHTT5HE.js.map +1 -0
  24. package/dist/{chunk-I2HNIE6N.js → chunk-NBSS5NDZ.js} +4 -4
  25. package/dist/{chunk-QG5F6463.js → chunk-ONM6PEAE.js} +2 -2
  26. package/dist/cli.js +2 -2
  27. package/dist/{code-agent-session-D-g04tcy.d.ts → code-agent-session-CdxteG0y.d.ts} +1 -1
  28. package/dist/contract/index.d.ts +19 -19
  29. package/dist/contract/index.js +6 -4
  30. package/dist/contract/index.js.map +1 -1
  31. package/dist/{control-CcBiAEnn.d.ts → control-DbcDxouY.d.ts} +1 -1
  32. package/dist/control.d.ts +2 -2
  33. package/dist/{default-registry-DltpYR5u.d.ts → default-registry-DDfv22MQ.d.ts} +2 -1
  34. package/dist/{gepa-dne9JDPL.d.ts → gepa-CQelRtuC.d.ts} +10 -8
  35. package/dist/hosted/index.d.ts +8 -4
  36. package/dist/{index-BTEpx9He.d.ts → index-DbCXJfZ1.d.ts} +2 -2
  37. package/dist/index.d.ts +27 -30
  38. package/dist/index.js +28 -22
  39. package/dist/index.js.map +1 -1
  40. package/dist/{insight-report-IwwvqZZv.d.ts → insight-report-oMVxDTxl.d.ts} +1 -1
  41. package/dist/{integrity-qemeBAyx.d.ts → integrity-C6PZ73iC.d.ts} +1 -1
  42. package/dist/kind-factory-DWOvXjR_.d.ts +171 -0
  43. package/dist/meta-eval/index.d.ts +2 -2
  44. package/dist/multishot/index.d.ts +6 -2
  45. package/dist/openapi.json +1 -1
  46. package/dist/policy-edit-Clb2v6Oa.d.ts +708 -0
  47. package/dist/{pre-registration-D8h7ZxNL.d.ts → pre-registration--vU0mMtD.d.ts} +4 -4
  48. package/dist/{provenance-Bibyg1U9.d.ts → provenance-BbVagC68.d.ts} +26 -14
  49. package/dist/{release-report-CCtzajxP.d.ts → release-report-CamNDe90.d.ts} +2 -2
  50. package/dist/reporting.d.ts +4 -4
  51. package/dist/{researcher-Dq-EtpbE.d.ts → researcher-Dwbo_Fxx.d.ts} +5 -5
  52. package/dist/rl.d.ts +11 -9
  53. package/dist/rl.js +2 -2
  54. package/dist/{rubric-predictive-validity-DYTLjGWu.d.ts → rubric-predictive-validity-BIdf9h4R.d.ts} +1 -1
  55. package/dist/{run-record-B7RTi_ix.d.ts → run-record-CZmcpWPo.d.ts} +1 -1
  56. package/dist/{runtime-trajectory-Dws7Kpgi.d.ts → runtime-trajectory-CC0jx9ql.d.ts} +1 -1
  57. package/dist/{semantic-concept-judge-DxJmRkyJ.d.ts → semantic-concept-judge-CKjePUMh.d.ts} +3 -3
  58. package/dist/{store-C1YxJDEK.d.ts → store-9cAScOcb.d.ts} +132 -1
  59. package/dist/{summary-report-BJ5aNwZ1.d.ts → summary-report-DTNgQycC.d.ts} +1 -1
  60. package/dist/traces.d.ts +6 -8
  61. package/dist/{types-C5gJrOVT.d.ts → types-Ca_63YSD.d.ts} +59 -2
  62. package/dist/wire/index.js +2 -2
  63. package/docs/design/loop-taxonomy.md +1 -2
  64. package/package.json +1 -1
  65. package/dist/chunk-5NVBGKPH.js.map +0 -1
  66. package/dist/chunk-AN5UYSVD.js +0 -761
  67. package/dist/chunk-AN5UYSVD.js.map +0 -1
  68. package/dist/chunk-DWLIGZBX.js.map +0 -1
  69. package/dist/chunk-FUCQVFMU.js.map +0 -1
  70. package/dist/chunk-N6MTC3GK.js.map +0 -1
  71. package/dist/kind-factory-DcNg13sZ.d.ts +0 -508
  72. package/dist/llm-client-DyqEH4jH.d.ts +0 -265
  73. package/dist/policy-edit-RLn8GWof.d.ts +0 -103
  74. package/dist/raw-provider-sink-C46HDghv.d.ts +0 -132
  75. /package/dist/{chunk-DRPIZQIT.js.map → chunk-4D5RVB3W.js.map} +0 -0
  76. /package/dist/{chunk-LVTGFSHF.js.map → chunk-7GKEAIAD.js.map} +0 -0
  77. /package/dist/{chunk-I2HNIE6N.js.map → chunk-NBSS5NDZ.js.map} +0 -0
  78. /package/dist/{chunk-QG5F6463.js.map → chunk-ONM6PEAE.js.map} +0 -0
@@ -1,10 +1,10 @@
1
1
  import { A as AgentEvalError } from './errors-oeQrLqXC.js';
2
- import { R as RunRecord } from './run-record-B7RTi_ix.js';
2
+ import { R as RunRecord } from './run-record-CZmcpWPo.js';
3
3
  import { P as PairedBootstrapOptions, M as McNemarResult, R as RiskDifferenceResult, a as PairedBootstrapResult } from './statistics-oUbOJe-S.js';
4
- import { C as ChatClient } from './kind-factory-DcNg13sZ.js';
5
- import { a as JudgeDimension, S as Scenario, b as JudgeConfig } from './types-C5gJrOVT.js';
4
+ import { C as ChatClient } from './policy-edit-Clb2v6Oa.js';
5
+ import { a as JudgeDimension, S as Scenario, b as JudgeConfig } from './types-Ca_63YSD.js';
6
6
  import { TCloud } from '@tangle-network/tcloud';
7
- import { R as RawProviderSink } from './raw-provider-sink-C46HDghv.js';
7
+ import { R as RawProviderSink } from './store-9cAScOcb.js';
8
8
  import { D as DefaultVerdict } from './verdict-C9MlYujm.js';
9
9
 
10
10
  /**
@@ -1,7 +1,8 @@
1
- import { S as Scenario, g as Gate, G as GateResult, n as GateContext, C as CampaignResult, p as Mutator, f as SurfaceProposer, M as MutableSurface, h as GateDecision } from './types-C5gJrOVT.js';
2
- import { e as RedTeamCase, D as Direction, b as RunCampaignOptions } from './gepa-dne9JDPL.js';
3
- import { R as RunRecord } from './run-record-B7RTi_ix.js';
1
+ import { S as Scenario, g as Gate, G as GateResult, n as GateContext, C as CampaignResult, p as Mutator, f as SurfaceProposer, M as MutableSurface, o as GenerationCandidate, h as GateDecision } from './types-Ca_63YSD.js';
2
+ import { e as RedTeamCase, D as Direction, b as RunCampaignOptions } from './gepa-CQelRtuC.js';
3
+ import { R as RunRecord } from './run-record-CZmcpWPo.js';
4
4
  import { a as PairedBootstrapResult } from './statistics-oUbOJe-S.js';
5
+ import { P as PolicyEditCandidateRecord } from './policy-edit-Clb2v6Oa.js';
5
6
  import { HostedClient, TraceSpanEvent } from './hosted/index.js';
6
7
  import { C as CampaignStorage } from './storage-Dw_f7WMt.js';
7
8
 
@@ -356,7 +357,7 @@ declare function evolutionaryProposer<TFindings = unknown>(opts: EvolutionaryPro
356
357
  * Two artifacts, one source of truth:
357
358
  *
358
359
  * 1. `LoopProvenanceRecord` — a structured JSON record capturing every
359
- * candidate (surfaceHash + label + rationale), its measured composite,
360
+ * candidate (surfaceHash + label + rationale + structured cause), its measured composite,
360
361
  * the gate decision + reasons + delta, the held-out lift, the explicit
361
362
  * baseline→candidate diff, and BACKEND PROVENANCE (the
362
363
  * `assertRealBackend` verdict + worker call count + model). This is the
@@ -387,6 +388,18 @@ interface LoopProvenanceCandidate {
387
388
  /** Proposer rationale — the "because Z". When the proposer returned a bare
388
389
  * surface (blind mutator) this is absent. */
389
390
  rationale?: string;
391
+ /** Exact validated cause when the proposer emitted a structured record. */
392
+ candidateRecord?: PolicyEditCandidateRecord;
393
+ /** Exact complete incumbent this candidate mutated. */
394
+ parentSurfaceHash: string;
395
+ /** Search-split composite of the exact parent. */
396
+ parentComposite: number;
397
+ /** Search-split composite change relative to the exact parent. */
398
+ observedDeltaFromParent?: number;
399
+ /** Whether the candidate completed every designed cell and could be selected. */
400
+ eligibleForPromotion: boolean;
401
+ /** Designed-denominator receipt retained even for incomplete candidates. */
402
+ coverage: NonNullable<GenerationCandidate['coverage']>;
390
403
  /** Mean composite this candidate scored on the search split. */
391
404
  composite: number;
392
405
  /** Whether this candidate was promoted out of its generation. */
@@ -409,7 +422,7 @@ interface LoopProvenanceBackend {
409
422
  * the bare hosted event) + backend provenance.
410
423
  */
411
424
  interface LoopProvenanceRecord {
412
- schema: 'tangle.loop-provenance.v2';
425
+ schema: 'tangle.loop-provenance.v3';
413
426
  runId: string;
414
427
  runDir: string;
415
428
  timestamp: string;
@@ -421,8 +434,10 @@ interface LoopProvenanceRecord {
421
434
  winnerRationale?: string;
422
435
  /** The explicit baseline→winner unified diff the gate decided on. */
423
436
  diff: string;
424
- /** Every candidate across every generation, each carrying its rationale. */
437
+ /** Every candidate across every generation, with its rationale and structured cause. */
425
438
  candidates: LoopProvenanceCandidate[];
439
+ /** Baseline composite on the search split that generated the candidates. */
440
+ baselineSearchComposite: number;
426
441
  /** The gate verdict — decision + reasons + contributing gates + delta. */
427
442
  gate: {
428
443
  decision: GateDecision;
@@ -453,18 +468,15 @@ interface BuildLoopProvenanceArgs<TArtifact, TScenario extends Scenario> {
453
468
  winnerLabel?: string;
454
469
  winnerRationale?: string;
455
470
  diff: string;
471
+ /** Baseline composite on the search split, distinct from holdout scoring. */
472
+ baselineSearchComposite: number;
456
473
  /** Per-generation candidate records straight off the loop result. */
457
474
  generations: Array<{
458
475
  generationIndex: number;
459
- candidates: Array<{
460
- surfaceHash: string;
461
- composite: number;
462
- label?: string;
463
- rationale?: string;
464
- }>;
476
+ candidates: GenerationCandidate[];
465
477
  promoted: string[];
466
- /** Surfaces measured this generation, keyed positionally to candidates so
467
- * the content hash can be computed from the real surface text. */
478
+ /** Surfaces measured this generation, keyed by surface hash so the content
479
+ * hash can be computed and the loop identity rechecked from real bytes. */
468
480
  surfaces: Array<{
469
481
  surfaceHash: string;
470
482
  surface: MutableSurface;
@@ -1,6 +1,6 @@
1
1
  import { a as DatasetSplit, b as DatasetManifest, D as DatasetScenario } from './dataset-NENEzRgk.js';
2
- import { m as GateDecision } from './summary-report-BJ5aNwZ1.js';
3
- import { R as RunRecord, b as RunSplitTag } from './run-record-B7RTi_ix.js';
2
+ import { m as GateDecision } from './summary-report-DTNgQycC.js';
3
+ import { R as RunRecord, a as RunSplitTag } from './run-record-CZmcpWPo.js';
4
4
 
5
5
  /**
6
6
  * Release confidence gate.
@@ -1,9 +1,9 @@
1
- export { R as RubricOutcomePair, a as RubricPredictiveValidityInput, b as RubricPredictiveValidityReport, c as RubricRanking, r as rubricPredictiveValidity } from './rubric-predictive-validity-DYTLjGWu.js';
2
- export { B as BootstrapOptions, a as BootstrapResult, J as JudgeReplayGateArgs, R as ReleaseConfidenceAxis, b as ReleaseConfidenceAxisName, c as ReleaseConfidenceInput, d as ReleaseConfidenceIssue, e as ReleaseConfidenceMetrics, f as ReleaseConfidenceScorecard, g as ReleaseConfidenceStatus, h as ReleaseConfidenceThresholds, i as ReleaseTraceEvidence, j as RenderReleaseReportOptions, V as Verdict, k as assertReleaseConfidence, l as bootstrapCi, m as evaluateReleaseConfidence, n as judgeReplayGate, r as renderReleaseReport } from './release-report-CCtzajxP.js';
1
+ export { R as RubricOutcomePair, a as RubricPredictiveValidityInput, b as RubricPredictiveValidityReport, c as RubricRanking, r as rubricPredictiveValidity } from './rubric-predictive-validity-BIdf9h4R.js';
2
+ export { B as BootstrapOptions, a as BootstrapResult, J as JudgeReplayGateArgs, R as ReleaseConfidenceAxis, b as ReleaseConfidenceAxisName, c as ReleaseConfidenceInput, d as ReleaseConfidenceIssue, e as ReleaseConfidenceMetrics, f as ReleaseConfidenceScorecard, g as ReleaseConfidenceStatus, h as ReleaseConfidenceThresholds, i as ReleaseTraceEvidence, j as RenderReleaseReportOptions, V as Verdict, k as assertReleaseConfidence, l as bootstrapCi, m as evaluateReleaseConfidence, n as judgeReplayGate, r as renderReleaseReport } from './release-report-CamNDe90.js';
3
3
  export { I as InterimReleaseConfidence, a as InterimReleaseConfidenceInput, P as PairedEvalueOptions, b as PairedEvalueSequence, c as PairedEvalueStep, S as SequentialDecision, e as evaluateInterimReleaseConfidence, p as pairedEvalueSequence } from './sequential-5iSVfzl2.js';
4
4
  export { P as PairedBootstrapOptions, a as PairedBootstrapResult, b as benjaminiHochberg, p as pairedBootstrap, w as wilcoxonSignedRank } from './statistics-oUbOJe-S.js';
5
- export { G as GainDistributionBin, a as GainDistributionFigureSpec, b as GainDistributionOptions, P as ParetoFigureSpec, c as ParetoPoint, R as RESEARCH_REPORT_HARD_PAIR_FLOOR, d as ResearchReport, e as ResearchReportCandidate, f as ResearchReportDecision, g as ResearchReportMethodology, h as ResearchReportOptions, i as ResearchReportRecommendation, S as SummaryTable, j as SummaryTableOptions, k as SummaryTableRow, l as gainHistogram, p as paretoChart, r as researchReport, s as summaryTable } from './summary-report-BJ5aNwZ1.js';
6
- import './run-record-B7RTi_ix.js';
5
+ export { G as GainDistributionBin, a as GainDistributionFigureSpec, b as GainDistributionOptions, P as ParetoFigureSpec, c as ParetoPoint, R as RESEARCH_REPORT_HARD_PAIR_FLOOR, d as ResearchReport, e as ResearchReportCandidate, f as ResearchReportDecision, g as ResearchReportMethodology, h as ResearchReportOptions, i as ResearchReportRecommendation, S as SummaryTable, j as SummaryTableOptions, k as SummaryTableRow, l as gainHistogram, p as paretoChart, r as researchReport, s as summaryTable } from './summary-report-DTNgQycC.js';
6
+ import './run-record-CZmcpWPo.js';
7
7
  import '@tangle-network/agent-interface';
8
8
  import './errors-oeQrLqXC.js';
9
9
  import './schema-SGWcK9wa.js';
@@ -1,9 +1,9 @@
1
- import { b as RunSplitTag, d as RunTokenUsage, e as RunJudgeMetadata, J as JudgeScoresRecord, A as AgentProfileCell, f as AgentProfileCellInput, R as RunRecord } from './run-record-B7RTi_ix.js';
2
- import { L as LlmClientOptions, a as LlmRouteRequirements } from './llm-client-DyqEH4jH.js';
3
- import { h as ResearchReportOptions, d as ResearchReport, m as GateDecision } from './summary-report-BJ5aNwZ1.js';
1
+ import { a as RunSplitTag, c as RunTokenUsage, d as RunJudgeMetadata, J as JudgeScoresRecord, A as AgentProfileCell, e as AgentProfileCellInput, R as RunRecord } from './run-record-CZmcpWPo.js';
2
+ import { L as LlmClientOptions, a as LlmRouteRequirements } from './policy-edit-Clb2v6Oa.js';
3
+ import { h as ResearchReportOptions, d as ResearchReport, m as GateDecision } from './summary-report-DTNgQycC.js';
4
4
  import { T as TraceEmitter, R as RunCompleteHook } from './emitter-BRchAAAx.js';
5
- import { R as RunIntegrityExpectations, a as RunIntegrityReport } from './integrity-qemeBAyx.js';
6
- import { R as RawProviderSink } from './raw-provider-sink-C46HDghv.js';
5
+ import { R as RunIntegrityExpectations, a as RunIntegrityReport } from './integrity-C6PZ73iC.js';
6
+ import { R as RawProviderSink } from './store-9cAScOcb.js';
7
7
  import { F as FailureClass } from './schema-SGWcK9wa.js';
8
8
  import { T as TraceStore } from './store-BsVi7ncX.js';
9
9
 
package/dist/rl.d.ts CHANGED
@@ -1,24 +1,26 @@
1
- import { R as RunRecord, b as RunSplitTag } from './run-record-B7RTi_ix.js';
1
+ import { R as RunRecord, a as RunSplitTag } from './run-record-CZmcpWPo.js';
2
2
  export { A as AdversarialMutation } from './adversarial-B7loGVVX.js';
3
3
  import { S as Span } from './schema-SGWcK9wa.js';
4
4
  import { T as TraceStore } from './store-BsVi7ncX.js';
5
5
  export { O as OffPolicyEstimate, a as OffPolicyOptions, b as OffPolicyTrajectory, d as doublyRobust, i as inverseProbabilityWeighting, o as offPolicyEstimateAll, s as selfNormalizedImportanceWeighting } from './off-policy-DiwuKKg7.js';
6
6
  import { b as OutcomeStore } from './outcome-store-rnXLEqSn.js';
7
7
  export { D as DeploymentOutcome, F as FileSystemOutcomeStore, a as FileSystemOutcomeStoreOptions, I as InMemoryOutcomeStore } from './outcome-store-rnXLEqSn.js';
8
- import { b as RubricPredictiveValidityReport } from './rubric-predictive-validity-DYTLjGWu.js';
9
- import { R as Researcher, F as FailureMode, S as SteeringChange, E as ExperimentPlan, a as ExperimentResult, b as EvalCampaignResult, c as EvalCampaignOptions } from './researcher-Dq-EtpbE.js';
10
- export { r as runEvalCampaign } from './researcher-Dq-EtpbE.js';
8
+ import { b as RubricPredictiveValidityReport } from './rubric-predictive-validity-BIdf9h4R.js';
9
+ import { R as Researcher, F as FailureMode, S as SteeringChange, E as ExperimentPlan, a as ExperimentResult, b as EvalCampaignResult, c as EvalCampaignOptions } from './researcher-Dwbo_Fxx.js';
10
+ export { r as runEvalCampaign } from './researcher-Dwbo_Fxx.js';
11
11
  import { a as VerificationReport } from './multi-layer-verifier-BsqKuLyN.js';
12
12
  import { I as InterimReleaseConfidence } from './sequential-5iSVfzl2.js';
13
- import { C as CampaignResult } from './types-C5gJrOVT.js';
13
+ import { C as CampaignResult } from './types-Ca_63YSD.js';
14
14
  import '@tangle-network/agent-interface';
15
15
  import './errors-oeQrLqXC.js';
16
- import './llm-client-DyqEH4jH.js';
17
- import './raw-provider-sink-C46HDghv.js';
18
- import './summary-report-BJ5aNwZ1.js';
16
+ import './policy-edit-Clb2v6Oa.js';
17
+ import './store-9cAScOcb.js';
18
+ import './types-C7DGg5ex.js';
19
+ import '@tangle-network/tcloud';
20
+ import './summary-report-DTNgQycC.js';
19
21
  import './failure-cluster-C48PiReX.js';
20
22
  import './emitter-BRchAAAx.js';
21
- import './integrity-qemeBAyx.js';
23
+ import './integrity-C6PZ73iC.js';
22
24
  import './verdict-C9MlYujm.js';
23
25
 
24
26
  /**
package/dist/rl.js CHANGED
@@ -10,7 +10,7 @@ import {
10
10
  } from "./chunk-3RF76KTD.js";
11
11
  import {
12
12
  runEvalCampaign
13
- } from "./chunk-QG5F6463.js";
13
+ } from "./chunk-ONM6PEAE.js";
14
14
  import {
15
15
  detectRewardHacking,
16
16
  extractVerifiableReward,
@@ -38,7 +38,7 @@ import "./chunk-TVVP3ZZQ.js";
38
38
  import "./chunk-5UF54T55.js";
39
39
  import "./chunk-XJYR7XFV.js";
40
40
  import "./chunk-VSMTAMNK.js";
41
- import "./chunk-FUCQVFMU.js";
41
+ import "./chunk-GY4SYVPJ.js";
42
42
  import "./chunk-PC4UYEBM.js";
43
43
  import {
44
44
  ValidationError
@@ -1,4 +1,4 @@
1
- import { R as RunRecord } from './run-record-B7RTi_ix.js';
1
+ import { R as RunRecord } from './run-record-CZmcpWPo.js';
2
2
  import { b as OutcomeStore } from './outcome-store-rnXLEqSn.js';
3
3
 
4
4
  /**
@@ -357,4 +357,4 @@ declare function roundTripRunRecord(record: RunRecord): RunRecord;
357
357
  */
358
358
  declare function modelHasSnapshot(model: string): boolean;
359
359
 
360
- export { type AgentProfileCell as A, requireAgentProfileCell as B, resolveRunCostProvenance as C, roundTripRunRecord as D, toAgentProfileJson as E, validateAgentProfileCell as F, validateRunRecord as G, verifyAgentProfileCell as H, type JudgeScoresRecord as J, type RunRecord as R, type AgentProfileJson as a, type RunSplitTag as b, type RunCostProvenance as c, type RunTokenUsage as d, type RunJudgeMetadata as e, type AgentProfileCellInput as f, AGENT_PROFILE_KINDS as g, type AgentInterfaceProfileLike as h, type AgentProfileCellSchemaVersion as i, AgentProfileCellValidationError as j, type AgentProfileDimensionValue as k, type AgentProfileHarness as l, type AgentProfileKind as m, type AgentProfileSource as n, type AgentProfileSourceInput as o, type RunOutcome as p, RunRecordValidationError as q, agentProfileCellHashMaterial as r, agentProfileCellKey as s, assertRunAgentProfileCell as t, buildAgentInterfaceProfileCell as u, buildAgentProfileCell as v, groupRunsByAgentProfileCell as w, isRunRecord as x, modelHasSnapshot as y, parseRunRecordSafe as z };
360
+ export { type AgentProfileCell as A, requireAgentProfileCell as B, resolveRunCostProvenance as C, roundTripRunRecord as D, toAgentProfileJson as E, validateAgentProfileCell as F, validateRunRecord as G, verifyAgentProfileCell as H, type JudgeScoresRecord as J, type RunRecord as R, type RunSplitTag as a, type RunCostProvenance as b, type RunTokenUsage as c, type RunJudgeMetadata as d, type AgentProfileCellInput as e, type AgentProfileJson as f, AGENT_PROFILE_KINDS as g, type AgentInterfaceProfileLike as h, type AgentProfileCellSchemaVersion as i, AgentProfileCellValidationError as j, type AgentProfileDimensionValue as k, type AgentProfileHarness as l, type AgentProfileKind as m, type AgentProfileSource as n, type AgentProfileSourceInput as o, type RunOutcome as p, RunRecordValidationError as q, agentProfileCellHashMaterial as r, agentProfileCellKey as s, assertRunAgentProfileCell as t, buildAgentInterfaceProfileCell as u, buildAgentProfileCell as v, groupRunsByAgentProfileCell as w, isRunRecord as x, modelHasSnapshot as y, parseRunRecordSafe as z };
@@ -1,4 +1,4 @@
1
- import { b as RunSplitTag } from './run-record-B7RTi_ix.js';
1
+ import { a as RunSplitTag } from './run-record-CZmcpWPo.js';
2
2
 
3
3
  interface RuntimeTrajectoryHookEvent {
4
4
  id: string;
@@ -1,10 +1,10 @@
1
1
  import { AxAIService } from '@ax-llm/ax';
2
2
  import { z } from 'zod';
3
- import { A as AnalystFinding, T as TraceAnalystKindSpec, a as Analyst, b as AnalystContext } from './kind-factory-DcNg13sZ.js';
4
- import { a as TraceAnalystSpan } from './store-C1YxJDEK.js';
3
+ import { d as AnalystFinding, A as Analyst, b as AnalystContext, L as LlmClientOptions } from './policy-edit-Clb2v6Oa.js';
4
+ import { T as TraceAnalystKindSpec } from './kind-factory-DWOvXjR_.js';
5
+ import { a as TraceAnalystSpan } from './store-9cAScOcb.js';
5
6
  import { R as Run, S as Span, e as TraceEvent, A as Artifact, B as BudgetLedgerEntry } from './schema-SGWcK9wa.js';
6
7
  import { T as TraceStore } from './store-BsVi7ncX.js';
7
- import { L as LlmClientOptions } from './llm-client-DyqEH4jH.js';
8
8
  import { S as Severity } from './multi-layer-verifier-BsqKuLyN.js';
9
9
 
10
10
  interface CreateAnalystAiConfig {
@@ -1,3 +1,134 @@
1
+ /**
2
+ * RawProviderSink — first-class persistence for the actual HTTP-level
3
+ * request/response bodies of every LLM provider call.
4
+ *
5
+ * Why this is a separate sink from the structured `LlmSpan`:
6
+ *
7
+ * - `LlmSpan` records the *intent* — model name, messages, output text,
8
+ * usage. It's what dashboards read; it's NOT enough for forensics.
9
+ * - When a downstream consumer reports "the verifier used the wrong route"
10
+ * or "tokens look right but reasoning was missing," the only way to
11
+ * answer is the raw HTTP body. Span fields can lie (a proxy can echo
12
+ * a different `model` value than what actually answered); the raw
13
+ * response is ground truth.
14
+ *
15
+ * Default behaviour: opt-in. Pass `rawSink` to `LlmClientOptions` (or the
16
+ * matrix runner / BuilderSession sets it up automatically) and every
17
+ * request, response, and error is recorded — including retries, with the
18
+ * attempt index attached so a flaky call's full event chain is recoverable.
19
+ *
20
+ * Redaction is enforced at sink time. The default redactor strips
21
+ * `Authorization`, `X-Api-Key`, `X-Auth-Token`, `Cookie` headers and any
22
+ * payload field whose key matches `apiKey | api_key | bearer | password |
23
+ * secret | token` (case-insensitive). Override via the sink constructor or
24
+ * the per-call `redactor`. The `redactedFields` array on the persisted
25
+ * event lets a reviewer see what was stripped without exposing the values.
26
+ */
27
+ type RawProviderDirection = 'request' | 'response' | 'error';
28
+ interface RawProviderEvent {
29
+ /** Stable id. Generated by the sink if omitted. */
30
+ eventId: string;
31
+ /** Trace context populated by `LlmClient` when the call is wrapped in a span. */
32
+ runId?: string;
33
+ spanId?: string;
34
+ /**
35
+ * Logical provider name. Free-form so callers can use whatever id matches
36
+ * their topology (`'openai'`, `'anthropic'`, `'tangle-router'`, …). When
37
+ * omitted, derived from `baseUrl` in `LlmClientOptions`.
38
+ */
39
+ provider: string;
40
+ model: string;
41
+ /** Endpoint path, e.g. `'/v1/chat/completions'`. */
42
+ endpoint: string;
43
+ /** Base URL used for the call (already-normalised — no trailing slash). */
44
+ baseUrl: string;
45
+ /** 0-indexed retry attempt. The first attempt is 0; a retried call gets 1, 2, … */
46
+ attemptIndex: number;
47
+ direction: RawProviderDirection;
48
+ /** Unix ms. */
49
+ timestamp: number;
50
+ /** Wall-clock duration of the call leg. Set on `response` and `error` events; null on `request`. */
51
+ durationMs?: number;
52
+ statusCode?: number;
53
+ requestHeaders?: Record<string, string>;
54
+ requestBody?: unknown;
55
+ responseHeaders?: Record<string, string>;
56
+ responseBody?: unknown;
57
+ /** Set on `direction: 'error'` events. */
58
+ errorMessage?: string;
59
+ /** Field paths the redactor stripped from this event ('header:Authorization', 'body.apiKey', …). */
60
+ redactedFields: string[];
61
+ }
62
+ interface RawProviderSinkFilter {
63
+ runId?: string;
64
+ spanId?: string;
65
+ direction?: RawProviderDirection;
66
+ attemptIndex?: number;
67
+ }
68
+ interface RawProviderSink {
69
+ record(event: RawProviderEvent): Promise<void>;
70
+ /** Optional listing — implementations that durably persist (file, db) should support this. */
71
+ list?(filter?: RawProviderSinkFilter): Promise<RawProviderEvent[]>;
72
+ /** Optional teardown for backed implementations. */
73
+ close?(): Promise<void>;
74
+ }
75
+ type ProviderRedactor = (event: RawProviderEvent) => RawProviderEvent;
76
+ /**
77
+ * Default redactor — strips well-known auth headers and any body field whose
78
+ * key matches the credential pattern. Records every redacted path on
79
+ * `event.redactedFields` so a downstream reviewer can see what was removed.
80
+ */
81
+ declare function defaultProviderRedactor(event: RawProviderEvent): RawProviderEvent;
82
+ interface InMemoryRawProviderSinkOptions {
83
+ redactor?: ProviderRedactor;
84
+ }
85
+ declare class InMemoryRawProviderSink implements RawProviderSink {
86
+ private events;
87
+ private redactor;
88
+ constructor(opts?: InMemoryRawProviderSinkOptions);
89
+ record(event: RawProviderEvent): Promise<void>;
90
+ list(filter?: RawProviderSinkFilter): Promise<RawProviderEvent[]>;
91
+ size(): number;
92
+ }
93
+ declare class NoopRawProviderSink implements RawProviderSink {
94
+ record(): Promise<void>;
95
+ /**
96
+ * Returns an empty array. Implemented so `assertRunCaptured` does not
97
+ * trip the `no_raw_sink` issue when a caller explicitly opts out of
98
+ * capture by passing this sink — opt-out is a deliberate choice, not a
99
+ * misconfiguration.
100
+ */
101
+ list(): Promise<RawProviderEvent[]>;
102
+ }
103
+ interface FileSystemRawProviderSinkOptions {
104
+ /** Directory the NDJSON file is written into. Created if missing. */
105
+ dir: string;
106
+ /** File name; default `'raw-provider-events.ndjson'`. */
107
+ fileName?: string;
108
+ /** Bytes after which the writer rolls over to a new file (default 32 MiB). */
109
+ rollAtBytes?: number;
110
+ redactor?: ProviderRedactor;
111
+ }
112
+ declare class FileSystemRawProviderSink implements RawProviderSink {
113
+ private dir;
114
+ private fileName;
115
+ private rollAtBytes;
116
+ private redactor;
117
+ private bytesWritten;
118
+ private rollIndex;
119
+ private initPromise;
120
+ constructor(opts: FileSystemRawProviderSinkOptions);
121
+ private ensureInit;
122
+ private currentPath;
123
+ record(event: RawProviderEvent): Promise<void>;
124
+ list(filter?: RawProviderSinkFilter): Promise<RawProviderEvent[]>;
125
+ }
126
+ /**
127
+ * Best-effort provider id from a base URL. Falls back to the URL host when
128
+ * none of the well-known patterns match.
129
+ */
130
+ declare function providerFromBaseUrl(baseUrl: string): string;
131
+
1
132
  /**
2
133
  * Shared types for the trace-analyst module.
3
134
  *
@@ -245,4 +376,4 @@ interface TraceAnalysisStore {
245
376
  }): Promise<SearchSpanResult>;
246
377
  }
247
378
 
248
- export { DEFAULT_TRACE_ANALYST_BUDGETS as D, type ErrorCluster as E, type QueryTracesPage as Q, type SearchSpanResult as S, type TraceAnalysisStore as T, type ViewSpansResult as V, type TraceAnalystSpan as a, type DatasetOverview as b, type SearchTraceResult as c, type SpanMatchRecord as d, TRACE_ANALYST_TRUNCATION_MARKER_PREFIX as e, type TraceAnalystByteBudgets as f, type TraceAnalystFilters as g, type TraceAnalystSpanKind as h, type TraceAnalystSpanStatus as i, type TraceAnalystTraceSummary as j, type ViewTraceOversized as k, type ViewTraceResult as l };
379
+ export { DEFAULT_TRACE_ANALYST_BUDGETS as D, type ErrorCluster as E, FileSystemRawProviderSink as F, InMemoryRawProviderSink as I, NoopRawProviderSink as N, type ProviderRedactor as P, type QueryTracesPage as Q, type RawProviderSink as R, type SearchSpanResult as S, type TraceAnalysisStore as T, type ViewSpansResult as V, type TraceAnalystSpan as a, type DatasetOverview as b, type FileSystemRawProviderSinkOptions as c, type InMemoryRawProviderSinkOptions as d, type RawProviderDirection as e, type RawProviderEvent as f, type RawProviderSinkFilter as g, type SearchTraceResult as h, type SpanMatchRecord as i, TRACE_ANALYST_TRUNCATION_MARKER_PREFIX as j, type TraceAnalystByteBudgets as k, type TraceAnalystFilters as l, type TraceAnalystSpanKind as m, type TraceAnalystSpanStatus as n, type TraceAnalystTraceSummary as o, type ViewTraceOversized as p, type ViewTraceResult as q, defaultProviderRedactor as r, providerFromBaseUrl as s };
@@ -1,4 +1,4 @@
1
- import { R as RunRecord } from './run-record-B7RTi_ix.js';
1
+ import { R as RunRecord } from './run-record-CZmcpWPo.js';
2
2
  import { F as FailureClusterReport } from './failure-cluster-C48PiReX.js';
3
3
 
4
4
  /**
package/dist/traces.d.ts CHANGED
@@ -1,19 +1,17 @@
1
1
  import { N as NotFoundError, R as ReplayError } from './errors-oeQrLqXC.js';
2
- import { P as ProviderRedactor, R as RawProviderSink, d as RawProviderEvent } from './raw-provider-sink-C46HDghv.js';
3
- export { F as FileSystemRawProviderSink, a as FileSystemRawProviderSinkOptions, I as InMemoryRawProviderSink, b as InMemoryRawProviderSinkOptions, N as NoopRawProviderSink, c as RawProviderDirection, e as RawProviderSinkFilter, f as defaultProviderRedactor, p as providerFromBaseUrl } from './raw-provider-sink-C46HDghv.js';
2
+ import { m as TraceAnalystSpanKind, n as TraceAnalystSpanStatus, T as TraceAnalysisStore, l as TraceAnalystFilters, b as DatasetOverview, Q as QueryTracesPage, q as ViewTraceResult, V as ViewSpansResult, h as SearchTraceResult, S as SearchSpanResult, P as ProviderRedactor, R as RawProviderSink, f as RawProviderEvent } from './store-9cAScOcb.js';
3
+ export { D as DEFAULT_TRACE_ANALYST_BUDGETS, E as ErrorCluster, F as FileSystemRawProviderSink, c as FileSystemRawProviderSinkOptions, I as InMemoryRawProviderSink, d as InMemoryRawProviderSinkOptions, N as NoopRawProviderSink, e as RawProviderDirection, g as RawProviderSinkFilter, i as SpanMatchRecord, j as TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, k as TraceAnalystByteBudgets, a as TraceAnalystSpan, o as TraceAnalystTraceSummary, p as ViewTraceOversized, r as defaultProviderRedactor, s as providerFromBaseUrl } from './store-9cAScOcb.js';
4
4
  import { a as RunCompleteHookContext, R as RunCompleteHook } from './emitter-BRchAAAx.js';
5
5
  export { S as SpanHandle, T as TraceEmitter, b as TraceEmitterOptions, l as llmSpanFromProvider } from './emitter-BRchAAAx.js';
6
- export { b as RunIntegrityError, R as RunIntegrityExpectations, c as RunIntegrityIssue, d as RunIntegrityIssueCode, a as RunIntegrityReport, e as assertRunCaptured, t as throwIfRunIncomplete } from './integrity-qemeBAyx.js';
6
+ export { b as RunIntegrityError, R as RunIntegrityExpectations, c as RunIntegrityIssue, d as RunIntegrityIssueCode, a as RunIntegrityReport, e as assertRunCaptured, t as throwIfRunIncomplete } from './integrity-C6PZ73iC.js';
7
7
  import { T as TraceStore } from './store-BsVi7ncX.js';
8
8
  export { E as EventFilter, F as FileSystemTraceStore, a as FileSystemTraceStoreOptions, I as InMemoryTraceStore, R as RunFilter, S as SpanFilter } from './store-BsVi7ncX.js';
9
9
  export { a as aggregateLlm, b as argHash, g as groupBy, j as judgeSpans, l as llmSpans, r as runFailureClass, c as runsForScenario, t as toolSpans } from './query-Ck190MOd.js';
10
10
  import { R as Run } from './schema-SGWcK9wa.js';
11
11
  export { A as Artifact, B as BudgetLedgerEntry, h as BudgetSpec, E as EventKind, i as FAILURE_CLASSES, F as FailureClass, G as GenericSpan, J as JudgeSpan, L as LlmSpan, M as Message, c as RetrievalSpan, g as RunLayer, a as RunOutcome, f as RunStatus, d as SandboxSpan, S as Span, j as SpanBase, b as SpanKind, k as SpanStatus, l as TRACE_SCHEMA_VERSION, T as ToolSpan, e as TraceEvent, m as isJudgeSpan, n as isLlmSpan, o as isRetrievalSpan, p as isSandboxSpan, q as isToolSpan } from './schema-SGWcK9wa.js';
12
- import { A as AnalyzeTracesOptions, b as AnalyzeTracesResult } from './analyst-C8HHvfJp.js';
13
- export { a as AnalyzeTracesInput, c as AnalyzeTracesTurnSnapshot, d as analyzeTraces } from './analyst-C8HHvfJp.js';
14
- import { h as TraceAnalystSpanKind, i as TraceAnalystSpanStatus, T as TraceAnalysisStore, g as TraceAnalystFilters, b as DatasetOverview, Q as QueryTracesPage, l as ViewTraceResult, V as ViewSpansResult, c as SearchTraceResult, S as SearchSpanResult } from './store-C1YxJDEK.js';
15
- export { D as DEFAULT_TRACE_ANALYST_BUDGETS, E as ErrorCluster, d as SpanMatchRecord, e as TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, f as TraceAnalystByteBudgets, a as TraceAnalystSpan, j as TraceAnalystTraceSummary, k as ViewTraceOversized } from './store-C1YxJDEK.js';
16
- import { b as RunSplitTag, d as RunTokenUsage, R as RunRecord } from './run-record-B7RTi_ix.js';
12
+ import { A as AnalyzeTracesOptions, b as AnalyzeTracesResult } from './analyst-CFBc14Wc.js';
13
+ export { a as AnalyzeTracesInput, c as AnalyzeTracesTurnSnapshot, d as analyzeTraces } from './analyst-CFBc14Wc.js';
14
+ import { a as RunSplitTag, c as RunTokenUsage, R as RunRecord } from './run-record-CZmcpWPo.js';
17
15
  import { AxFunction } from '@ax-llm/ax';
18
16
  import '@tangle-network/agent-interface';
19
17
 
@@ -1,4 +1,5 @@
1
- import { d as RunTokenUsage } from './run-record-B7RTi_ix.js';
1
+ import { P as PolicyEditCandidateRecord } from './policy-edit-Clb2v6Oa.js';
2
+ import { c as RunTokenUsage } from './run-record-CZmcpWPo.js';
2
3
 
3
4
  /**
4
5
  * Pass A substrate types — `runCampaign` is the one primitive every
@@ -168,6 +169,9 @@ interface ProposedCandidate {
168
169
  * primitive it used. Survives to `GenerationCandidate.rationale` and the
169
170
  * emitted provenance record. */
170
171
  rationale: string;
172
+ /** Structured, JSON-safe cause for this exact candidate when the proposer
173
+ * can provide one. Policy edits retain the full validated edit here. */
174
+ candidateRecord?: PolicyEditCandidateRecord;
171
175
  }
172
176
  /** Type guard: a proposal carrying its rationale vs a bare
173
177
  * surface. The loop branches on this to populate `GenerationCandidate`. */
@@ -194,6 +198,28 @@ interface ParetoParent {
194
198
  label?: string;
195
199
  rationale?: string;
196
200
  }
201
+ /** Exact measured state for the surface an optimizer is learning from.
202
+ * Unlike a model-authored expected gain, every value here comes from a
203
+ * completed campaign over the designed denominator. */
204
+ interface ScoredSurfaceOutcome {
205
+ /** Optimization/search evidence only. Held-out results must never flow back
206
+ * into a proposer through this type. */
207
+ split: 'search';
208
+ /** Generation that actually measured this surface (`-1` for the baseline). */
209
+ generation: number;
210
+ surfaceHash: string;
211
+ composite: number;
212
+ dimensions: Record<string, number>;
213
+ scenarios: Array<{
214
+ scenarioId: string;
215
+ composite: number;
216
+ notes?: string;
217
+ }>;
218
+ coverage: {
219
+ expectedCells: number;
220
+ scorableCells: number;
221
+ };
222
+ }
197
223
  /** Stateless surface mutation — given findings + current
198
224
  * surface, return N candidate surfaces. Pure transform, no generation
199
225
  * awareness. Reflective-mutation and `AxGEPA` mutators conform. Wrapped by
@@ -221,6 +247,12 @@ interface ProposeContext<TFindings = unknown> {
221
247
  populationSize: number;
222
248
  generation: number;
223
249
  signal: AbortSignal;
250
+ /** Measured baseline for this optimization run. `runOptimization` always
251
+ * supplies it; optional for standalone proposer callers. */
252
+ baselineOutcome?: ScoredSurfaceOutcome;
253
+ /** Measured result for `currentSurface`, the complete global incumbent every
254
+ * new candidate mutates. `runOptimization` always supplies it. */
255
+ incumbentOutcome?: ScoredSurfaceOutcome;
224
256
  /** Optional analysis report produced before proposal. Opaque to the substrate:
225
257
  * the proposer that consumes it owns the shape. */
226
258
  report?: unknown;
@@ -517,6 +549,29 @@ interface GenerationCandidate {
517
549
  surfaceHash: string;
518
550
  composite: number;
519
551
  ci95: [number, number];
552
+ /** Exact surface this candidate mutated. */
553
+ parentSurfaceHash?: string;
554
+ /** Measured search-split composite of the exact parent surface. */
555
+ parentComposite?: number;
556
+ /** Candidate composite minus its parent's composite. Present only when the
557
+ * candidate completed the designed denominator. */
558
+ observedDeltaFromParent?: number;
559
+ /** Whether this candidate had a scorable result for every designed campaign
560
+ * cell and was therefore eligible for ranking, promotion, and Pareto
561
+ * selection. Older externally-authored records may omit this field; loop
562
+ * records always populate it. */
563
+ eligibleForPromotion?: boolean;
564
+ /** Exact denominator receipt for selection eligibility. Scores stay
565
+ * descriptive: an incomplete candidate is retained with its observed score
566
+ * and errors instead of receiving an invented penalty. */
567
+ coverage?: {
568
+ expectedCells: number;
569
+ scorableCells: number;
570
+ unscorableCells: Array<{
571
+ cellId: string;
572
+ reason: string;
573
+ }>;
574
+ };
520
575
  /** Mean score per judge dimension across all cells (scenarios × reps ×
521
576
  * judges that reported the dimension). */
522
577
  dimensions: Record<string, number>;
@@ -538,6 +593,8 @@ interface GenerationCandidate {
538
593
  * "because rationale Z" the audit requires to survive to the result.
539
594
  * Present when the proposer returned a `ProposedCandidate`. */
540
595
  rationale?: string;
596
+ /** Exact structured cause threaded from the proposer, when available. */
597
+ candidateRecord?: PolicyEditCandidateRecord;
541
598
  }
542
599
  interface CampaignAggregates {
543
600
  byJudge: Record<string, JudgeAggregate>;
@@ -572,4 +629,4 @@ interface CampaignResult<TArtifact = unknown, TScenario extends Scenario = Scena
572
629
  scenarios: Array<Pick<TScenario, 'id' | 'kind'>>;
573
630
  }
574
631
 
575
- export { type JudgeAggregate as A, type ScenarioAggregate as B, type CampaignResult as C, type DispatchContext as D, isProposedCandidate as E, labelTrustRank as F, type GateResult as G, type JudgeScore as J, type LabeledScenarioStore as L, type MutableSurface as M, type OptimizationProposer as O, type ParetoParent as P, type RedactionStatus as R, type Scenario as S, type TraceSpan as T, type JudgeDimension as a, type JudgeConfig as b, type DispatchFn as c, type CampaignTraceWriter as d, type GenerationRecord as e, type SurfaceProposer as f, type Gate as g, type GateDecision as h, type CampaignAggregates as i, type CampaignArtifactWriter as j, type CampaignCellResult as k, type CampaignCostMeter as l, type CodeSurface as m, type GateContext as n, type GenerationCandidate as o, type Mutator as p, type OptimizerConfig as q, type SessionScript as r, type LabeledScenarioWrite as s, type LabeledScenarioSampleArgs as t, type LabeledScenarioRecord as u, type LabelTrust as v, type ProposedCandidate as w, type ProposeContext as x, type LabeledScenarioSource as y, type CampaignTokenUsage as z };
632
+ export { type JudgeAggregate as A, type ScenarioAggregate as B, type CampaignResult as C, type DispatchContext as D, type ScoredSurfaceOutcome as E, isProposedCandidate as F, type GateResult as G, labelTrustRank as H, type JudgeScore as J, type LabeledScenarioStore as L, type MutableSurface as M, type OptimizationProposer as O, type ParetoParent as P, type RedactionStatus as R, type Scenario as S, type TraceSpan as T, type JudgeDimension as a, type JudgeConfig as b, type DispatchFn as c, type CampaignTraceWriter as d, type GenerationRecord as e, type SurfaceProposer as f, type Gate as g, type GateDecision as h, type CampaignAggregates as i, type CampaignArtifactWriter as j, type CampaignCellResult as k, type CampaignCostMeter as l, type CodeSurface as m, type GateContext as n, type GenerationCandidate as o, type Mutator as p, type OptimizerConfig as q, type SessionScript as r, type LabeledScenarioWrite as s, type LabeledScenarioSampleArgs as t, type LabeledScenarioRecord as u, type LabelTrust as v, type ProposedCandidate as w, type ProposeContext as x, type LabeledScenarioSource as y, type CampaignTokenUsage as z };
@@ -34,8 +34,8 @@ import {
34
34
  runRpcOnce,
35
35
  startServer,
36
36
  startServerAsync
37
- } from "../chunk-DRPIZQIT.js";
38
- import "../chunk-FUCQVFMU.js";
37
+ } from "../chunk-4D5RVB3W.js";
38
+ import "../chunk-GY4SYVPJ.js";
39
39
  import "../chunk-PC4UYEBM.js";
40
40
  import "../chunk-ONWEPEDO.js";
41
41
  import "../chunk-PZ5AY32C.js";
@@ -201,8 +201,7 @@ consume the **same dataset** the flywheel builds.
201
201
  agent-runtime.
202
202
  - **runCampaign** — a measurement: a surface scored over N scenarios × M reps.
203
203
  agent-eval. (A "campaign" = a coordinated batch of measurements.)
204
- - **runOptimization** — the improvement loop body: proposer suggests surfaces,
205
- each measured by a campaign, top-K promoted per generation. agent-eval.
204
+ - **runOptimization** — the improvement loop body: proposer suggests surfaces, each is measured, and only a candidate that beats the global incumbent is promoted. agent-eval.
206
205
  - **runImprovementLoop** — `runOptimization` + holdout re-score + release gate
207
206
  + optional PR. agent-eval.
208
207
  - **runAnalystLoop** — reflective autoresearch: findings + knowledge updates +
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@tangle-network/agent-eval",
3
- "version": "0.115.2",
3
+ "version": "0.116.0",
4
4
  "description": "Evaluate and improve AI agents from runs, traces, judges, and feedback. Compare candidates, cluster failures, measure lift, and gate releases.",
5
5
  "homepage": "https://github.com/tangle-network/agent-eval#readme",
6
6
  "repository": {