@tangle-network/agent-eval 0.109.1 → 0.110.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (177) hide show
  1. package/CHANGELOG.md +9 -0
  2. package/dist/analyst/index.d.ts +10 -12
  3. package/dist/analyst/index.js +8 -11
  4. package/dist/analyst/index.js.map +1 -1
  5. package/dist/{analyze-runs-DJYpep3L.d.ts → analyze-runs-Dmz6LA9e.d.ts} +4 -4
  6. package/dist/{baseline-Bbid3WoO.d.ts → baseline-DsNteOgR.d.ts} +32 -2
  7. package/dist/belief-state/index.d.ts +6 -6
  8. package/dist/benchmarks/index.d.ts +4 -4
  9. package/dist/benchmarks/index.js +7 -8
  10. package/dist/builder-eval/index.d.ts +4 -4
  11. package/dist/builder-eval/index.js +1 -2
  12. package/dist/builder-eval/index.js.map +1 -1
  13. package/dist/{calibration-BPmzuVPk.d.ts → calibration-Dz8TQV4y.d.ts} +2 -2
  14. package/dist/campaign/index.d.ts +62 -20
  15. package/dist/campaign/index.js +9 -8
  16. package/dist/{chunk-LOJ2QVCE.js → chunk-2IY4ILP4.js} +2 -2
  17. package/dist/{chunk-LIEJUH2I.js → chunk-6PL5MGDL.js} +9 -9
  18. package/dist/{chunk-2OGPXHOB.js → chunk-7NX6ZSBG.js} +36 -7
  19. package/dist/chunk-7NX6ZSBG.js.map +1 -0
  20. package/dist/{chunk-R6D7NEYJ.js → chunk-GBI5J5DB.js} +81 -11
  21. package/dist/chunk-GBI5J5DB.js.map +1 -0
  22. package/dist/{chunk-YEHAEDUD.js → chunk-IMWDSFUM.js} +604 -2
  23. package/dist/chunk-IMWDSFUM.js.map +1 -0
  24. package/dist/{chunk-OVPVM4JC.js → chunk-J4AKLZEV.js} +15 -4
  25. package/dist/{chunk-OVPVM4JC.js.map → chunk-J4AKLZEV.js.map} +1 -1
  26. package/dist/{chunk-JZXGWLK5.js → chunk-MHNQWM4I.js} +62 -6
  27. package/dist/chunk-MHNQWM4I.js.map +1 -0
  28. package/dist/{chunk-QRVS7MX4.js → chunk-OW47B5WA.js} +3 -5
  29. package/dist/{chunk-QRVS7MX4.js.map → chunk-OW47B5WA.js.map} +1 -1
  30. package/dist/{chunk-DBDRR6GF.js → chunk-PLOMR3HP.js} +48 -2
  31. package/dist/chunk-PLOMR3HP.js.map +1 -0
  32. package/dist/{chunk-GDZAWO2I.js → chunk-QFGTU7MT.js} +2 -2
  33. package/dist/{chunk-V7HNA47Z.js → chunk-RSVSSZKF.js} +5 -5
  34. package/dist/{chunk-5PK3626Q.js → chunk-XRGOKCMO.js} +88 -17
  35. package/dist/chunk-XRGOKCMO.js.map +1 -0
  36. package/dist/{code-agent-session-rnJKlqmT.d.ts → code-agent-session-yitf9I-F.d.ts} +1 -1
  37. package/dist/contract/index.d.ts +26 -28
  38. package/dist/contract/index.js +11 -13
  39. package/dist/contract/index.js.map +1 -1
  40. package/dist/{control-B8UthSBL.d.ts → control-U8LBKUES.d.ts} +5 -6
  41. package/dist/control.d.ts +8 -9
  42. package/dist/control.js +6 -8
  43. package/dist/{dataset-DS7ytHZU.d.ts → dataset-NENEzRgk.d.ts} +1 -1
  44. package/dist/{default-registry-BswHCXnU.d.ts → default-registry-Bcf1uKVI.d.ts} +1 -2
  45. package/dist/{emitter-C2rqGH_l.d.ts → emitter-BRchAAAx.d.ts} +2 -2
  46. package/dist/{failure-cluster-DH9Flgcf.d.ts → failure-cluster-C48PiReX.d.ts} +2 -2
  47. package/dist/feedback-trajectory-pDcz1lQ1.d.ts +348 -0
  48. package/dist/{gepa-B3x5Ulcv.d.ts → gepa-T8T215nw.d.ts} +149 -6
  49. package/dist/hosted/index.d.ts +7 -7
  50. package/dist/{index-pPtfoIJO.d.ts → index-Dc3VLGhp.d.ts} +2 -2
  51. package/dist/index.d.ts +645 -61
  52. package/dist/index.js +1282 -190
  53. package/dist/index.js.map +1 -1
  54. package/dist/{insight-report-B4xrdwEK.d.ts → insight-report-D4cXFsLt.d.ts} +1 -1
  55. package/dist/{integrity-DqGZg3st.d.ts → integrity-qemeBAyx.d.ts} +1 -1
  56. package/dist/{types-D1ytG0Yg.d.ts → kind-factory-20hcaYpf.d.ts} +169 -2
  57. package/dist/meta-eval/index.d.ts +5 -5
  58. package/dist/meta-eval/index.js +1 -2
  59. package/dist/meta-eval/index.js.map +1 -1
  60. package/dist/{multi-layer-verifier-CI4jdX-q.d.ts → multi-layer-verifier-BsqKuLyN.d.ts} +1 -1
  61. package/dist/multishot/index.d.ts +3 -3
  62. package/dist/openapi.json +1 -1
  63. package/dist/pipelines/index.d.ts +6 -7
  64. package/dist/pipelines/index.js +3 -6
  65. package/dist/pipelines/index.js.map +1 -1
  66. package/dist/{policy-edit-DQUXYMDm.d.ts → policy-edit-D2bBDZDf.d.ts} +2 -2
  67. package/dist/{pre-registration-BUhVPzE7.d.ts → pre-registration-BepVVa6P.d.ts} +3 -3
  68. package/dist/{provenance-DdDhf6cg.d.ts → provenance-CyxkvEi9.d.ts} +3 -5
  69. package/dist/{query-0aTmbmQe.d.ts → query-Ck190MOd.d.ts} +2 -2
  70. package/dist/{release-report-DeJpsBiA.d.ts → release-report-oBfOz8ku.d.ts} +3 -3
  71. package/dist/reporting.d.ts +8 -8
  72. package/dist/{researcher-Wc7dx6GM.d.ts → researcher-CaH0CwFC.d.ts} +6 -6
  73. package/dist/rl.d.ts +568 -15
  74. package/dist/rl.js +4 -4
  75. package/dist/{rubric-predictive-validity-DPnyG-CE.d.ts → rubric-predictive-validity-C-fMteAW.d.ts} +1 -1
  76. package/dist/{run-record-I-Z3JNvO.d.ts → run-record-DksGsfgv.d.ts} +1 -1
  77. package/dist/{runtime-trajectory-iW9IhV3e.d.ts → runtime-trajectory-h5i0SZUj.d.ts} +1 -1
  78. package/dist/{schema-m0gsnbt3.d.ts → schema-SGWcK9wa.d.ts} +1 -1
  79. package/dist/{semantic-concept-judge-BmNZPB_j.d.ts → semantic-concept-judge-D7z6JCLZ.d.ts} +57 -4
  80. package/dist/{store-BcFXE6LG.d.ts → store-BsVi7ncX.d.ts} +1 -1
  81. package/dist/storyboard/index.d.ts +1 -1
  82. package/dist/{summary-report-QMZVe3P-.d.ts → summary-report-Bz-0-t8v.d.ts} +2 -2
  83. package/dist/{test-graded-scenario-DeODGLra.d.ts → test-graded-scenario-mzYBKspu.d.ts} +3 -3
  84. package/dist/traces.d.ts +54 -11
  85. package/dist/traces.js +25 -27
  86. package/dist/{types-BdIv5dvA.d.ts → types-v--ctu-b.d.ts} +2 -2
  87. package/dist/wire/index.d.ts +5 -6
  88. package/docs/improvement-glossary.md +14 -13
  89. package/package.json +1 -71
  90. package/dist/adapters/http.d.ts +0 -142
  91. package/dist/adapters/http.js +0 -203
  92. package/dist/adapters/http.js.map +0 -1
  93. package/dist/adapters/langchain.d.ts +0 -95
  94. package/dist/adapters/langchain.js +0 -34
  95. package/dist/adapters/langchain.js.map +0 -1
  96. package/dist/adapters/otel.d.ts +0 -112
  97. package/dist/adapters/otel.js +0 -110
  98. package/dist/adapters/otel.js.map +0 -1
  99. package/dist/chunk-2OGPXHOB.js.map +0 -1
  100. package/dist/chunk-45EEMHTC.js +0 -35
  101. package/dist/chunk-45EEMHTC.js.map +0 -1
  102. package/dist/chunk-5BKGXME7.js +0 -65
  103. package/dist/chunk-5BKGXME7.js.map +0 -1
  104. package/dist/chunk-5PK3626Q.js.map +0 -1
  105. package/dist/chunk-6SK5VFYK.js +0 -100
  106. package/dist/chunk-6SK5VFYK.js.map +0 -1
  107. package/dist/chunk-DBDRR6GF.js.map +0 -1
  108. package/dist/chunk-DJWX3GVS.js +0 -81
  109. package/dist/chunk-DJWX3GVS.js.map +0 -1
  110. package/dist/chunk-FOUG2VVS.js +0 -855
  111. package/dist/chunk-FOUG2VVS.js.map +0 -1
  112. package/dist/chunk-JZXGWLK5.js.map +0 -1
  113. package/dist/chunk-K7QEIHHJ.js +0 -613
  114. package/dist/chunk-K7QEIHHJ.js.map +0 -1
  115. package/dist/chunk-KKHDIONI.js +0 -414
  116. package/dist/chunk-KKHDIONI.js.map +0 -1
  117. package/dist/chunk-KMPRBJK4.js +0 -74
  118. package/dist/chunk-KMPRBJK4.js.map +0 -1
  119. package/dist/chunk-Q2JRAWRI.js +0 -196
  120. package/dist/chunk-Q2JRAWRI.js.map +0 -1
  121. package/dist/chunk-R6D7NEYJ.js.map +0 -1
  122. package/dist/chunk-RZTMDUO7.js +0 -49
  123. package/dist/chunk-RZTMDUO7.js.map +0 -1
  124. package/dist/chunk-STGVSCDH.js +0 -202
  125. package/dist/chunk-STGVSCDH.js.map +0 -1
  126. package/dist/chunk-YEHAEDUD.js.map +0 -1
  127. package/dist/control-runtime-Acf9CGhw.d.ts +0 -182
  128. package/dist/corpus-eBVwhCp1.d.ts +0 -560
  129. package/dist/counterfactual-DlOz8PBx.d.ts +0 -85
  130. package/dist/diagnose.d.ts +0 -252
  131. package/dist/diagnose.js +0 -382
  132. package/dist/diagnose.js.map +0 -1
  133. package/dist/feedback-trajectory-C9KCo8ag.d.ts +0 -169
  134. package/dist/governance/index.d.ts +0 -135
  135. package/dist/governance/index.js +0 -18
  136. package/dist/governance/index.js.map +0 -1
  137. package/dist/groundedness/index.d.ts +0 -112
  138. package/dist/groundedness/index.js +0 -77
  139. package/dist/groundedness/index.js.map +0 -1
  140. package/dist/harness-optimizer-mOl9XX_O.d.ts +0 -106
  141. package/dist/kind-factory-DvIGo_cP.d.ts +0 -171
  142. package/dist/knowledge/index.d.ts +0 -103
  143. package/dist/knowledge/index.js +0 -18
  144. package/dist/knowledge/index.js.map +0 -1
  145. package/dist/pareto-E-pembql.d.ts +0 -81
  146. package/dist/perf/index.d.ts +0 -123
  147. package/dist/perf/index.js +0 -18
  148. package/dist/perf/index.js.map +0 -1
  149. package/dist/prm/index.d.ts +0 -104
  150. package/dist/prm/index.js +0 -265
  151. package/dist/prm/index.js.map +0 -1
  152. package/dist/product-benchmark/index.d.ts +0 -247
  153. package/dist/product-benchmark/index.js +0 -37
  154. package/dist/product-benchmark/index.js.map +0 -1
  155. package/dist/red-team-KmmiqBlY.d.ts +0 -63
  156. package/dist/redact-B40YG2M_.d.ts +0 -45
  157. package/dist/rubric-Cc6UHvUb.d.ts +0 -73
  158. package/dist/run-critic-CmMf05uV.d.ts +0 -56
  159. package/dist/sink-fetch-B1Yg4Til.d.ts +0 -101
  160. package/dist/telemetry/file.d.ts +0 -19
  161. package/dist/telemetry/file.js +0 -45
  162. package/dist/telemetry/file.js.map +0 -1
  163. package/dist/telemetry/index.d.ts +0 -38
  164. package/dist/telemetry/index.js +0 -130
  165. package/dist/telemetry/index.js.map +0 -1
  166. package/dist/testing-C21CHsq2.d.ts +0 -20
  167. package/dist/testing.d.ts +0 -1
  168. package/dist/testing.js +0 -8
  169. package/dist/testing.js.map +0 -1
  170. package/dist/trajectory-2TkpSEVh.d.ts +0 -33
  171. package/dist/workflow/index.d.ts +0 -496
  172. package/dist/workflow/index.js +0 -2178
  173. package/dist/workflow/index.js.map +0 -1
  174. /package/dist/{chunk-LOJ2QVCE.js.map → chunk-2IY4ILP4.js.map} +0 -0
  175. /package/dist/{chunk-LIEJUH2I.js.map → chunk-6PL5MGDL.js.map} +0 -0
  176. /package/dist/{chunk-GDZAWO2I.js.map → chunk-QFGTU7MT.js.map} +0 -0
  177. /package/dist/{chunk-V7HNA47Z.js.map → chunk-RSVSSZKF.js.map} +0 -0
@@ -196,4 +196,4 @@ declare function isRetrievalSpan(s: Span): s is RetrievalSpan;
196
196
  declare function isJudgeSpan(s: Span): s is JudgeSpan;
197
197
  declare function isSandboxSpan(s: Span): s is SandboxSpan;
198
198
 
199
- export { type Artifact as A, type BudgetLedgerEntry as B, type EventKind as E, type FailureClass as F, type GenericSpan as G, type JudgeSpan as J, type LlmSpan as L, type Message as M, type Run as R, type Span as S, type ToolSpan as T, type TraceEvent as a, type RunOutcome as b, type SpanKind as c, type RetrievalSpan as d, type SandboxSpan as e, type RunStatus as f, type RunLayer as g, type BudgetSpec as h, FAILURE_CLASSES as i, type SpanBase as j, type SpanStatus as k, TRACE_SCHEMA_VERSION as l, isJudgeSpan as m, isLlmSpan as n, isRetrievalSpan as o, isSandboxSpan as p, isToolSpan as q };
199
+ export { type Artifact as A, type BudgetLedgerEntry as B, type EventKind as E, type FailureClass as F, type GenericSpan as G, type JudgeSpan as J, type LlmSpan as L, type Message as M, type Run as R, type Span as S, type ToolSpan as T, type RunOutcome as a, type SpanKind as b, type RetrievalSpan as c, type SandboxSpan as d, type TraceEvent as e, type RunStatus as f, type RunLayer as g, type BudgetSpec as h, FAILURE_CLASSES as i, type SpanBase as j, type SpanStatus as k, TRACE_SCHEMA_VERSION as l, isJudgeSpan as m, isLlmSpan as n, isRetrievalSpan as o, isSandboxSpan as p, isToolSpan as q };
@@ -1,10 +1,11 @@
1
1
  import { AxAIService } from '@ax-llm/ax';
2
2
  import { z } from 'zod';
3
- import { A as AnalystFinding, a as Analyst, b as AnalystContext } from './types-D1ytG0Yg.js';
4
- import { T as TraceAnalystKindSpec } from './kind-factory-DvIGo_cP.js';
3
+ import { A as AnalystFinding, T as TraceAnalystKindSpec, a as Analyst, b as AnalystContext } from './kind-factory-20hcaYpf.js';
5
4
  import { a as TraceAnalystSpan } from './store-C1YxJDEK.js';
5
+ import { R as Run, S as Span, e as TraceEvent, A as Artifact, B as BudgetLedgerEntry } from './schema-SGWcK9wa.js';
6
+ import { T as TraceStore } from './store-BsVi7ncX.js';
6
7
  import { L as LlmClientOptions } from './llm-client-DyqEH4jH.js';
7
- import { S as Severity } from './multi-layer-verifier-CI4jdX-q.js';
8
+ import { S as Severity } from './multi-layer-verifier-BsqKuLyN.js';
8
9
 
9
10
  interface CreateAnalystAiConfig {
10
11
  /** OpenAI-compatible API key forwarded as `Authorization: Bearer`.
@@ -487,6 +488,58 @@ interface BehavioralMetrics {
487
488
  */
488
489
  declare function computeTraceMetrics(spans: readonly TraceAnalystSpan[]): BehavioralMetrics;
489
490
 
491
+ interface RunScore {
492
+ success: number;
493
+ goalProgress: number;
494
+ repoGroundedness: number;
495
+ driftPenalty: number;
496
+ toolUseQuality: number;
497
+ patchQuality: number;
498
+ testReality: number;
499
+ finalGate: number;
500
+ reviewerBlockers: number;
501
+ costUsd: number;
502
+ wallSeconds: number;
503
+ notes?: string[];
504
+ }
505
+ interface RunScoreWeights {
506
+ success: number;
507
+ goalProgress: number;
508
+ repoGroundedness: number;
509
+ driftPenalty: number;
510
+ toolUseQuality: number;
511
+ patchQuality: number;
512
+ testReality: number;
513
+ finalGate: number;
514
+ reviewerBlockers: number;
515
+ costUsd: number;
516
+ wallSeconds: number;
517
+ }
518
+ declare const DEFAULT_RUN_SCORE_WEIGHTS: RunScoreWeights;
519
+ declare function aggregateRunScore(score: RunScore, weights?: Partial<RunScoreWeights>): number;
520
+ declare function clamp01(value: number): number;
521
+
522
+ interface RunTrace {
523
+ run: Run;
524
+ spans: Span[];
525
+ events: TraceEvent[];
526
+ artifacts: Artifact[];
527
+ budget: BudgetLedgerEntry[];
528
+ }
529
+ interface RunCriticOptions {
530
+ weights?: Partial<RunScoreWeights>;
531
+ driftPatterns?: RegExp[];
532
+ }
533
+ declare class RunCritic {
534
+ private readonly weights?;
535
+ private readonly driftPatterns;
536
+ constructor(options?: RunCriticOptions);
537
+ score(store: TraceStore, runId: string): Promise<RunScore>;
538
+ scoreTrace(trace: RunTrace): RunScore;
539
+ rank(score: RunScore): number;
540
+ private isDrift;
541
+ }
542
+
490
543
  /**
491
544
  * Semantic concept judge — "does the built artifact actually implement
492
545
  * the features the user asked for?"
@@ -621,4 +674,4 @@ declare function runSemanticConceptJudge(input: SemanticConceptJudgeInput, optio
621
674
  */
622
675
  declare function createSemanticConceptJudge(options?: SemanticConceptJudgeOptions): (input: SemanticConceptJudgeInput) => Promise<SemanticConceptJudgeResult>;
623
676
 
624
- export { type ConceptWeightStrategy as A, type BehavioralMetrics as B, type CreateAnalystAiConfig as C, DEFAULT_TRACE_ANALYST_KINDS as D, DEFAULT_COMPLEXITY_WEIGHTS as E, FAILURE_MODE_KIND_SPEC as F, SEMANTIC_CONCEPT_JUDGE_VERSION as G, type SemanticConceptJudgeResult as H, IMPROVEMENT_KIND_SPEC as I, type SuboptimalCode as J, KIND_EXPECTED_SUBJECTS as K, type SuboptimalSignal as L, computeTraceMetrics as M, createSemanticConceptJudge as N, runSemanticConceptJudge as O, type PersistedFinding as P, type SemanticConceptJudgeOptions as S, type SemanticConceptJudgeInput as a, type DiffPolicy as b, FINDING_SUBJECT_GRAMMAR_PROMPT as c, FINDING_SUBJECT_KINDS as d, type FindingSubject as e, type FindingSubjectKind as f, FindingSubjectStringSchema as g, type FindingsDiff as h, FindingsStore as i, KNOWLEDGE_GAP_KIND_SPEC as j, KNOWLEDGE_POISONING_KIND_SPEC as k, SKILL_USAGE_ANALYST as l, SkillUsageAnalyst as m, type SkillUsageRecord as n, type SkillUsageReport as o, type SkillUsageScanConfig as p, buildSkillUsageReport as q, createAnalystAi as r, defaultIsMaterial as s, diffFindings as t, emitSkillUsageFindings as u, parseFindingSubject as v, renderFindingSubject as w, type ConceptComplexity as x, type ConceptFinding as y, type ConceptSpec as z };
677
+ export { type ConceptComplexity as A, type BehavioralMetrics as B, type CreateAnalystAiConfig as C, DEFAULT_TRACE_ANALYST_KINDS as D, type ConceptFinding as E, FAILURE_MODE_KIND_SPEC as F, type ConceptSpec as G, type ConceptWeightStrategy as H, IMPROVEMENT_KIND_SPEC as I, DEFAULT_COMPLEXITY_WEIGHTS as J, KIND_EXPECTED_SUBJECTS as K, DEFAULT_RUN_SCORE_WEIGHTS as L, type RunCriticOptions as M, SEMANTIC_CONCEPT_JUDGE_VERSION as N, type SemanticConceptJudgeResult as O, type PersistedFinding as P, type SuboptimalCode as Q, RunCritic as R, type SemanticConceptJudgeOptions as S, type SuboptimalSignal as T, aggregateRunScore as U, clamp01 as V, computeTraceMetrics as W, createSemanticConceptJudge as X, runSemanticConceptJudge as Y, type RunTrace as a, type SemanticConceptJudgeInput as b, type DiffPolicy as c, FINDING_SUBJECT_GRAMMAR_PROMPT as d, FINDING_SUBJECT_KINDS as e, type FindingSubject as f, type FindingSubjectKind as g, FindingSubjectStringSchema as h, type FindingsDiff as i, FindingsStore as j, KNOWLEDGE_GAP_KIND_SPEC as k, KNOWLEDGE_POISONING_KIND_SPEC as l, SKILL_USAGE_ANALYST as m, SkillUsageAnalyst as n, type SkillUsageRecord as o, type SkillUsageReport as p, type SkillUsageScanConfig as q, buildSkillUsageReport as r, createAnalystAi as s, defaultIsMaterial as t, diffFindings as u, emitSkillUsageFindings as v, parseFindingSubject as w, renderFindingSubject as x, type RunScore as y, type RunScoreWeights as z };
@@ -1,4 +1,4 @@
1
- import { R as Run, S as Span, a as TraceEvent, A as Artifact, B as BudgetLedgerEntry, f as RunStatus, g as RunLayer, c as SpanKind, E as EventKind } from './schema-m0gsnbt3.js';
1
+ import { R as Run, S as Span, e as TraceEvent, A as Artifact, B as BudgetLedgerEntry, f as RunStatus, g as RunLayer, b as SpanKind, E as EventKind } from './schema-SGWcK9wa.js';
2
2
 
3
3
  interface RunFilter {
4
4
  scenarioId?: string;
@@ -1,4 +1,4 @@
1
- import { S as Span } from '../schema-m0gsnbt3.js';
1
+ import { S as Span } from '../schema-SGWcK9wa.js';
2
2
 
3
3
  /**
4
4
  * Code-edit extraction + the "watch the agent write code" animation.
@@ -1,5 +1,5 @@
1
- import { R as RunRecord } from './run-record-I-Z3JNvO.js';
2
- import { F as FailureClusterReport } from './failure-cluster-DH9Flgcf.js';
1
+ import { R as RunRecord } from './run-record-DksGsfgv.js';
2
+ import { F as FailureClusterReport } from './failure-cluster-C48PiReX.js';
3
3
 
4
4
  /**
5
5
  * HeldOutGate — first-class held-out paired-delta promotion gate.
@@ -1,6 +1,6 @@
1
- import { T as TraceEmitter } from './emitter-C2rqGH_l.js';
2
- import { R as Run, F as FailureClass } from './schema-m0gsnbt3.js';
3
- import { T as TraceStore } from './store-BcFXE6LG.js';
1
+ import { T as TraceEmitter } from './emitter-BRchAAAx.js';
2
+ import { R as Run, F as FailureClass } from './schema-SGWcK9wa.js';
3
+ import { T as TraceStore } from './store-BsVi7ncX.js';
4
4
 
5
5
  /**
6
6
  * SandboxHarness — executes a scenario in an isolated environment and
package/dist/traces.d.ts CHANGED
@@ -1,20 +1,19 @@
1
1
  import { N as NotFoundError, R as ReplayError } from './errors-oeQrLqXC.js';
2
2
  import { P as ProviderRedactor, R as RawProviderSink, d as RawProviderEvent } from './raw-provider-sink-C46HDghv.js';
3
3
  export { F as FileSystemRawProviderSink, a as FileSystemRawProviderSinkOptions, I as InMemoryRawProviderSink, b as InMemoryRawProviderSinkOptions, N as NoopRawProviderSink, c as RawProviderDirection, e as RawProviderSinkFilter, f as defaultProviderRedactor, p as providerFromBaseUrl } from './raw-provider-sink-C46HDghv.js';
4
- import { a as RunCompleteHookContext, R as RunCompleteHook } from './emitter-C2rqGH_l.js';
5
- export { S as SpanHandle, T as TraceEmitter, b as TraceEmitterOptions, l as llmSpanFromProvider } from './emitter-C2rqGH_l.js';
6
- export { b as RunIntegrityError, R as RunIntegrityExpectations, c as RunIntegrityIssue, d as RunIntegrityIssueCode, a as RunIntegrityReport, e as assertRunCaptured, t as throwIfRunIncomplete } from './integrity-DqGZg3st.js';
7
- import { T as TraceStore } from './store-BcFXE6LG.js';
8
- export { E as EventFilter, F as FileSystemTraceStore, a as FileSystemTraceStoreOptions, I as InMemoryTraceStore, R as RunFilter, S as SpanFilter } from './store-BcFXE6LG.js';
9
- export { a as aggregateLlm, b as argHash, g as groupBy, j as judgeSpans, l as llmSpans, r as runFailureClass, c as runsForScenario, t as toolSpans } from './query-0aTmbmQe.js';
10
- export { D as DEFAULT_REDACTION_RULES, b as REDACTION_VERSION, a as RedactionReport, R as RedactionRule, r as redactString, c as redactValue } from './redact-B40YG2M_.js';
11
- import { R as Run } from './schema-m0gsnbt3.js';
12
- export { A as Artifact, B as BudgetLedgerEntry, h as BudgetSpec, E as EventKind, i as FAILURE_CLASSES, F as FailureClass, G as GenericSpan, J as JudgeSpan, L as LlmSpan, M as Message, d as RetrievalSpan, g as RunLayer, b as RunOutcome, f as RunStatus, e as SandboxSpan, S as Span, j as SpanBase, c as SpanKind, k as SpanStatus, l as TRACE_SCHEMA_VERSION, T as ToolSpan, a as TraceEvent, m as isJudgeSpan, n as isLlmSpan, o as isRetrievalSpan, p as isSandboxSpan, q as isToolSpan } from './schema-m0gsnbt3.js';
4
+ import { a as RunCompleteHookContext, R as RunCompleteHook } from './emitter-BRchAAAx.js';
5
+ export { S as SpanHandle, T as TraceEmitter, b as TraceEmitterOptions, l as llmSpanFromProvider } from './emitter-BRchAAAx.js';
6
+ export { b as RunIntegrityError, R as RunIntegrityExpectations, c as RunIntegrityIssue, d as RunIntegrityIssueCode, a as RunIntegrityReport, e as assertRunCaptured, t as throwIfRunIncomplete } from './integrity-qemeBAyx.js';
7
+ import { T as TraceStore } from './store-BsVi7ncX.js';
8
+ export { E as EventFilter, F as FileSystemTraceStore, a as FileSystemTraceStoreOptions, I as InMemoryTraceStore, R as RunFilter, S as SpanFilter } from './store-BsVi7ncX.js';
9
+ export { a as aggregateLlm, b as argHash, g as groupBy, j as judgeSpans, l as llmSpans, r as runFailureClass, c as runsForScenario, t as toolSpans } from './query-Ck190MOd.js';
10
+ import { R as Run } from './schema-SGWcK9wa.js';
11
+ export { A as Artifact, B as BudgetLedgerEntry, h as BudgetSpec, E as EventKind, i as FAILURE_CLASSES, F as FailureClass, G as GenericSpan, J as JudgeSpan, L as LlmSpan, M as Message, c as RetrievalSpan, g as RunLayer, a as RunOutcome, f as RunStatus, d as SandboxSpan, S as Span, j as SpanBase, b as SpanKind, k as SpanStatus, l as TRACE_SCHEMA_VERSION, T as ToolSpan, e as TraceEvent, m as isJudgeSpan, n as isLlmSpan, o as isRetrievalSpan, p as isSandboxSpan, q as isToolSpan } from './schema-SGWcK9wa.js';
13
12
  import { A as AnalyzeTracesOptions, b as AnalyzeTracesResult } from './analyst-C8HHvfJp.js';
14
13
  export { a as AnalyzeTracesInput, c as AnalyzeTracesTurnSnapshot, d as analyzeTraces } from './analyst-C8HHvfJp.js';
15
14
  import { h as TraceAnalystSpanKind, i as TraceAnalystSpanStatus, T as TraceAnalysisStore, g as TraceAnalystFilters, b as DatasetOverview, Q as QueryTracesPage, l as ViewTraceResult, V as ViewSpansResult, c as SearchTraceResult, S as SearchSpanResult } from './store-C1YxJDEK.js';
16
15
  export { D as DEFAULT_TRACE_ANALYST_BUDGETS, E as ErrorCluster, d as SpanMatchRecord, e as TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, f as TraceAnalystByteBudgets, a as TraceAnalystSpan, j as TraceAnalystTraceSummary, k as ViewTraceOversized } from './store-C1YxJDEK.js';
17
- import { b as RunSplitTag, c as RunTokenUsage, R as RunRecord } from './run-record-I-Z3JNvO.js';
16
+ import { b as RunSplitTag, c as RunTokenUsage, R as RunRecord } from './run-record-DksGsfgv.js';
18
17
  import { AxFunction } from '@ax-llm/ax';
19
18
  import '@tangle-network/agent-interface';
20
19
 
@@ -782,6 +781,50 @@ declare function otelRunCompleteHook(exporter: OtelExporter): RunCompleteHook;
782
781
  */
783
782
  declare function createOtelTracingStore(inner: TraceStore, exporter: OtelExporter, traceId: string): TraceStore;
784
783
 
784
+ /**
785
+ * Redaction — remove PII / secrets from trace payloads before persist.
786
+ *
787
+ * Pre-persistence rules mean raw traces in storage are already scrubbed.
788
+ * Unredacted variants (for debugging / post-mortems) live in a separate
789
+ * storage layer with stricter access controls; this module only covers
790
+ * the default scrub-then-persist path.
791
+ *
792
+ * Rules compose: pass an array of `RedactionRule`, each is applied in
793
+ * order. Strings that match get replaced with a tagged sentinel so the
794
+ * eval framework can count how many redactions happened per run
795
+ * (surfaced via `redaction_applied` events).
796
+ */
797
+ interface RedactionRule {
798
+ id: string;
799
+ pattern: RegExp;
800
+ /** Replacement — e.g. '[PII:email]'. Defaults to `[redacted:{id}]`. */
801
+ replacement?: string;
802
+ }
803
+ interface RedactionReport {
804
+ redactionCount: number;
805
+ byRule: Record<string, number>;
806
+ }
807
+ /** OWASP / common-sense defaults — extend per-domain. */
808
+ declare const DEFAULT_REDACTION_RULES: RedactionRule[];
809
+ declare const REDACTION_VERSION = "1.0.0";
810
+ /**
811
+ * Redact a single string. Returns the new string and a per-rule count of
812
+ * how many substitutions fired.
813
+ */
814
+ declare function redactString(input: string, rules?: RedactionRule[]): {
815
+ output: string;
816
+ report: RedactionReport;
817
+ };
818
+ /**
819
+ * Walk a JSON-ish value applying `redactString` to every string leaf.
820
+ * Arrays and plain objects are recursed; other types pass through
821
+ * untouched. Circular references throw — traces should be tree-shaped.
822
+ */
823
+ declare function redactValue(value: unknown, rules?: RedactionRule[], report?: RedactionReport): {
824
+ value: unknown;
825
+ report: RedactionReport;
826
+ };
827
+
785
828
  /**
786
829
  * Convert agent-eval's internal trace shape (`FileSystemTraceStore` → `Run`,
787
830
  * `Span`, `TraceEvent`) into the OTLP-flat JSONL the trace analyst
@@ -973,4 +1016,4 @@ declare function iterateRawCalls(sink: RawProviderSink, filter?: {
973
1016
  spanId?: string;
974
1017
  }): AsyncGenerator<ReplayCacheEntry>;
975
1018
 
976
- export { AnalyzeTracesOptions, AnalyzeTracesResult, type CaptureFetchContext, type CaptureFetchOptions, DatasetOverview, type ExportableSpan, type ExtractedUsage, type FlattenOtlpOptions, LLM_CACHED_TOKENS, LLM_CACHED_TOKEN_ATTR_KEYS, LLM_COST_ATTR_KEYS, LLM_COST_USD, LLM_INPUT_TOKENS, LLM_INPUT_TOKEN_ATTR_KEYS, LLM_MODEL_ATTR_KEYS, LLM_MODEL_NAME, LLM_OUTPUT_TOKENS, LLM_OUTPUT_TOKEN_ATTR_KEYS, OPENINFERENCE_SPAN_KIND, OTEL_AGENT_EVAL_SCOPE, type OtelExportConfig, type OtelExporter, type OtlpExport, OtlpFileTraceStore, type OtlpFileTraceStoreOptions, type OtlpFlatLine, type OtlpResourceSpans, type OtlpSpan, type OtlpToRunRecordsOptions, type OtlpTraceRunRecord, type ProjectedOtlpSpan, ProviderRedactor, QueryTracesPage, RawProviderEvent, RawProviderSink, ReplayCache, type ReplayCacheEntry, ReplayCacheMissError, type ReplayCacheStats, type ReplayFetchOptions, Run, RunCompleteHook, RunCompleteHookContext, SPAN_KIND_ATTR_KEYS, SearchSpanResult, SearchTraceResult, SpanNotFoundError, TOOL_NAME, TOOL_NAME_ATTR_KEYS, TRACE_ANALYST_ACTOR_DESCRIPTION, TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION, TRACE_ANALYST_SUBAGENT_DESCRIPTION, type TraceAggregate, TraceAnalysisStore, TraceAnalystFilters, type TraceAnalystHookOptions, TraceAnalystSpanKind, TraceAnalystSpanStatus, TraceFileMissingError, type TraceInsightContext, type TraceInsightFinding, type TraceInsightPanelRole, type TraceInsightPromptInput, type TraceInsightQualityGate, type TraceInsightQuestion, type TraceInsightReadiness, type TraceInsightSuite, type TraceInsightTask, TraceNotFoundError, TraceStore, type TraceStoreSource, type TraceStoreToOtlpOptions, type TracesToOtlpResult, ViewSpansResult, ViewTraceResult, asNumber, asString, buildTraceAnalystTools, buildTraceInsightContext, buildTraceInsightPrompt, captureFetchToRawSink, convertTraceStoresToOtlp, createOtelExporter, createOtelTracingStore, createReplayFetch, defaultTraceInsightPanel, describeTraceInsightScope, domainEvidencePattern, exportRunAsOtlp, extractOtlpAttributes, extractUsage, extractUsageFromResponse, extractUsageFromSse, firstNumberAttr, firstStringAttr, flattenOtlpExportToNdjson, inferDomainKeywords, inferOtlpKind, iterateRawCalls, otelRunCompleteHook, otlpToRunRecords, otlpToTraceRunRecords, planTraceInsightQuestions, projectOtlpFlatLine, readOtlpStatus, scoreTraceInsightReadiness, stringField, tokenizeDomainWords, traceAnalystFunctionGroup, traceAnalystOnRunComplete, traceSpanKindToOpenInferenceKind };
1019
+ export { AnalyzeTracesOptions, AnalyzeTracesResult, type CaptureFetchContext, type CaptureFetchOptions, DEFAULT_REDACTION_RULES, DatasetOverview, type ExportableSpan, type ExtractedUsage, type FlattenOtlpOptions, LLM_CACHED_TOKENS, LLM_CACHED_TOKEN_ATTR_KEYS, LLM_COST_ATTR_KEYS, LLM_COST_USD, LLM_INPUT_TOKENS, LLM_INPUT_TOKEN_ATTR_KEYS, LLM_MODEL_ATTR_KEYS, LLM_MODEL_NAME, LLM_OUTPUT_TOKENS, LLM_OUTPUT_TOKEN_ATTR_KEYS, OPENINFERENCE_SPAN_KIND, OTEL_AGENT_EVAL_SCOPE, type OtelExportConfig, type OtelExporter, type OtlpExport, OtlpFileTraceStore, type OtlpFileTraceStoreOptions, type OtlpFlatLine, type OtlpResourceSpans, type OtlpSpan, type OtlpToRunRecordsOptions, type OtlpTraceRunRecord, type ProjectedOtlpSpan, ProviderRedactor, QueryTracesPage, REDACTION_VERSION, RawProviderEvent, RawProviderSink, type RedactionReport, type RedactionRule, ReplayCache, type ReplayCacheEntry, ReplayCacheMissError, type ReplayCacheStats, type ReplayFetchOptions, Run, RunCompleteHook, RunCompleteHookContext, SPAN_KIND_ATTR_KEYS, SearchSpanResult, SearchTraceResult, SpanNotFoundError, TOOL_NAME, TOOL_NAME_ATTR_KEYS, TRACE_ANALYST_ACTOR_DESCRIPTION, TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION, TRACE_ANALYST_SUBAGENT_DESCRIPTION, type TraceAggregate, TraceAnalysisStore, TraceAnalystFilters, type TraceAnalystHookOptions, TraceAnalystSpanKind, TraceAnalystSpanStatus, TraceFileMissingError, type TraceInsightContext, type TraceInsightFinding, type TraceInsightPanelRole, type TraceInsightPromptInput, type TraceInsightQualityGate, type TraceInsightQuestion, type TraceInsightReadiness, type TraceInsightSuite, type TraceInsightTask, TraceNotFoundError, TraceStore, type TraceStoreSource, type TraceStoreToOtlpOptions, type TracesToOtlpResult, ViewSpansResult, ViewTraceResult, asNumber, asString, buildTraceAnalystTools, buildTraceInsightContext, buildTraceInsightPrompt, captureFetchToRawSink, convertTraceStoresToOtlp, createOtelExporter, createOtelTracingStore, createReplayFetch, defaultTraceInsightPanel, describeTraceInsightScope, domainEvidencePattern, exportRunAsOtlp, extractOtlpAttributes, extractUsage, extractUsageFromResponse, extractUsageFromSse, firstNumberAttr, firstStringAttr, flattenOtlpExportToNdjson, inferDomainKeywords, inferOtlpKind, iterateRawCalls, otelRunCompleteHook, otlpToRunRecords, otlpToTraceRunRecords, planTraceInsightQuestions, projectOtlpFlatLine, readOtlpStatus, redactString, redactValue, scoreTraceInsightReadiness, stringField, tokenizeDomainWords, traceAnalystFunctionGroup, traceAnalystOnRunComplete, traceSpanKindToOpenInferenceKind };
package/dist/traces.js CHANGED
@@ -28,7 +28,24 @@ import {
28
28
  scoreTraceInsightReadiness,
29
29
  tokenizeDomainWords,
30
30
  traceAnalystOnRunComplete
31
- } from "./chunk-V7HNA47Z.js";
31
+ } from "./chunk-RSVSSZKF.js";
32
+ import {
33
+ FAILURE_CLASSES,
34
+ TRACE_SCHEMA_VERSION,
35
+ aggregateLlm,
36
+ argHash,
37
+ groupBy,
38
+ isJudgeSpan,
39
+ isLlmSpan,
40
+ isRetrievalSpan,
41
+ isSandboxSpan,
42
+ isToolSpan,
43
+ judgeSpans,
44
+ llmSpans,
45
+ runFailureClass,
46
+ runsForScenario,
47
+ toolSpans
48
+ } from "./chunk-MHNQWM4I.js";
32
49
  import {
33
50
  TRACE_ANALYST_ACTOR_DESCRIPTION,
34
51
  TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION,
@@ -41,25 +58,6 @@ import {
41
58
  redactString,
42
59
  redactValue
43
60
  } from "./chunk-GGE4NNQT.js";
44
- import {
45
- aggregateLlm,
46
- argHash,
47
- groupBy,
48
- judgeSpans,
49
- llmSpans,
50
- runFailureClass,
51
- runsForScenario,
52
- toolSpans
53
- } from "./chunk-JZXGWLK5.js";
54
- import {
55
- FAILURE_CLASSES,
56
- TRACE_SCHEMA_VERSION,
57
- isJudgeSpan,
58
- isLlmSpan,
59
- isRetrievalSpan,
60
- isSandboxSpan,
61
- isToolSpan
62
- } from "./chunk-5BKGXME7.js";
63
61
  import {
64
62
  DEFAULT_TRACE_ANALYST_BUDGETS,
65
63
  LLM_CACHED_TOKENS,
@@ -99,6 +97,13 @@ import {
99
97
  assertRunCaptured,
100
98
  throwIfRunIncomplete
101
99
  } from "./chunk-TT4KNT67.js";
100
+ import {
101
+ TraceEmitter,
102
+ llmSpanFromProvider
103
+ } from "./chunk-TVVP3ZZQ.js";
104
+ import "./chunk-VK6HBGAE.js";
105
+ import "./chunk-XJYR7XFV.js";
106
+ import "./chunk-VSMTAMNK.js";
102
107
  import {
103
108
  FileSystemRawProviderSink,
104
109
  InMemoryRawProviderSink,
@@ -106,13 +111,6 @@ import {
106
111
  defaultProviderRedactor,
107
112
  providerFromBaseUrl
108
113
  } from "./chunk-PC4UYEBM.js";
109
- import "./chunk-VK6HBGAE.js";
110
- import {
111
- TraceEmitter,
112
- llmSpanFromProvider
113
- } from "./chunk-TVVP3ZZQ.js";
114
- import "./chunk-XJYR7XFV.js";
115
- import "./chunk-VSMTAMNK.js";
116
114
  import "./chunk-ONWEPEDO.js";
117
115
  import "./chunk-PZ5AY32C.js";
118
116
  export {
@@ -1,4 +1,4 @@
1
- import { c as RunTokenUsage } from './run-record-I-Z3JNvO.js';
1
+ import { c as RunTokenUsage } from './run-record-DksGsfgv.js';
2
2
 
3
3
  /**
4
4
  * Pass A substrate types — `runCampaign` is the one primitive every
@@ -557,4 +557,4 @@ interface CampaignResult<TArtifact = unknown, TScenario extends Scenario = Scena
557
557
  scenarios: Array<Pick<TScenario, 'id' | 'kind'>>;
558
558
  }
559
559
 
560
- export { type JudgeAggregate as A, type ScenarioAggregate as B, type CampaignResult as C, type DispatchFn as D, isProposedCandidate as E, labelTrustRank as F, type GateResult as G, type JudgeScore as J, type LabeledScenarioStore as L, type MutableSurface as M, type OptimizationProposer as O, type ParetoParent as P, type RedactionStatus as R, type Scenario as S, type TraceSpan as T, type JudgeConfig as a, type DispatchContext as b, type CampaignTraceWriter as c, type GenerationRecord as d, type SurfaceProposer as e, type Gate as f, type JudgeDimension as g, type GateDecision as h, type CampaignAggregates as i, type CampaignArtifactWriter as j, type CampaignCellResult as k, type CampaignCostMeter as l, type CodeSurface as m, type GateContext as n, type GenerationCandidate as o, type Mutator as p, type OptimizerConfig as q, type SessionScript as r, type LabeledScenarioWrite as s, type LabeledScenarioSampleArgs as t, type LabeledScenarioRecord as u, type LabelTrust as v, type ProposedCandidate as w, type ProposeContext as x, type LabeledScenarioSource as y, type CampaignTokenUsage as z };
560
+ export { type JudgeAggregate as A, type ScenarioAggregate as B, type CampaignResult as C, type DispatchContext as D, isProposedCandidate as E, labelTrustRank as F, type GateResult as G, type JudgeScore as J, type LabeledScenarioStore as L, type MutableSurface as M, type OptimizationProposer as O, type ParetoParent as P, type RedactionStatus as R, type Scenario as S, type TraceSpan as T, type JudgeDimension as a, type JudgeConfig as b, type DispatchFn as c, type CampaignTraceWriter as d, type GenerationRecord as e, type SurfaceProposer as f, type Gate as g, type GateDecision as h, type CampaignAggregates as i, type CampaignArtifactWriter as j, type CampaignCellResult as k, type CampaignCostMeter as l, type CodeSurface as m, type GateContext as n, type GenerationCandidate as o, type Mutator as p, type OptimizerConfig as q, type SessionScript as r, type LabeledScenarioWrite as s, type LabeledScenarioSampleArgs as t, type LabeledScenarioRecord as u, type LabelTrust as v, type ProposedCandidate as w, type ProposeContext as x, type LabeledScenarioSource as y, type CampaignTokenUsage as z };
@@ -1,14 +1,13 @@
1
- import { F as FeedbackTrajectoryStore } from '../feedback-trajectory-C9KCo8ag.js';
2
- import { T as TraceStore } from '../store-BcFXE6LG.js';
1
+ import { F as FeedbackTrajectoryStore } from '../feedback-trajectory-pDcz1lQ1.js';
2
+ import { T as TraceStore } from '../store-BsVi7ncX.js';
3
3
  import { z } from 'zod';
4
4
  import { OpenAPIObject } from 'openapi3-ts/oas31';
5
5
  import * as hono_types from 'hono/types';
6
6
  import { ServerType } from '@hono/node-server';
7
7
  import { Hono } from 'hono';
8
- import '../control-runtime-Acf9CGhw.js';
9
- import '../emitter-C2rqGH_l.js';
10
- import '../schema-m0gsnbt3.js';
11
- import '../dataset-DS7ytHZU.js';
8
+ import '../emitter-BRchAAAx.js';
9
+ import '../schema-SGWcK9wa.js';
10
+ import '../dataset-NENEzRgk.js';
12
11
  import '../errors-oeQrLqXC.js';
13
12
 
14
13
  declare const RubricDimensionSchema: z.ZodObject<{
@@ -61,25 +61,26 @@ Start at the top row and only move down when the row's *"reach for it when"* mat
61
61
  | `gepaProposer` | You want the strong default: reflective full-surface prompt rewrites, grounded in findings, keeping a Pareto frontier of complementary winners. | prompt string | **Yes — `improve({ surface: 'prompt' })` default.** Proven live. |
62
62
  | `skillOptProposer` | You are editing a structured `SKILL.md`/runbook and want small anchored add/delete/replace patches that preserve earlier rules. | skill/prompt string | **Yes — `improve({ surface: 'skills' })` default.** Not yet proven live. |
63
63
  | `parameterSweepProposer` | The likely fix is a config knob, not words — `retrieval.k`, `temperature`, `max_tokens`. You give it candidate patches; it applies them to a JSON surface. | JSON config string | Yes, but you supply the candidate list. |
64
- | `fapoProposer` | You want *evidence to decide when to escalate*: try prompt edits first, move to parameters, then to structural code — one scoped change per cycle, only escalating when the cheaper level is exhausted. **This is the composite/orchestration proposer** (there is no separate `compositeProposer`). | whatever its level proposers return | Exported; you wire the level proposers. |
64
+ | `fapoProposer` | You want *evidence to decide when to escalate*: try prompt edits first, move to parameters, then to structural code — one scoped change per cycle, only escalating when the cheaper level is exhausted. | whatever its level proposers return | Exported; you wire the level proposers. |
65
+ | `compositeProposer` | You want several proposers to share one candidate-generation budget in the same round. It allocates the population by declared weights, preserves member provenance, deduplicates surfaces, and isolates a member failure unless every member fails. | whatever its member proposers return | Exported; you wire the member proposers. |
65
66
  | `aceProposer` | You are accumulating hard-won lessons into a playbook and must **never** summarize an old lesson away (append-only, provenance-tagged). | playbook string | Exported. |
66
67
  | `memoryCurationProposer` | Same as ACE but you want a compact, deduped, re-ranked memory instead of append-only growth. | memory string | Exported. |
67
68
  | `evolutionaryProposer` | You want blind population search (mutate → measure → select) with no reflection over findings — a cheap control or a baseline to beat. | any string | Exported. |
68
69
  | `traceAnalystProposer` | Bench-only: race our trace-analysis evidence engine head-to-head inside `compareProposers`. | prompt string | Bench-only. |
69
70
  | `haloProposer` | Bench-only: race the external `halo-engine` analysis against ours. | prompt string | Bench-only, external. |
70
71
 
71
- Default path: `gepaProposer` for prompts; add `parameterSweepProposer` when a config knob is the suspect; wrap both in `fapoProposer` when you want the loop to decide *when* to escalate from words to knobs to code.
72
+ Default path: `gepaProposer` for prompts; add `parameterSweepProposer` when a config knob is the suspect; wrap levels in `fapoProposer` when the loop should decide *when* to escalate, or use `compositeProposer` when multiple proposer families must split one fixed population budget.
72
73
 
73
- ## Composing proposers — the only three ways, with runnable code
74
+ ## Composing proposers — four distinct shapes
74
75
 
75
- There is deliberately **no** `chainProposer`/`compositeProposer` primitive.
76
- Composition happens in exactly three shapes:
76
+ Choose the shape that matches the experiment:
77
77
 
78
- 1. **Escalate** — `fapoProposer` wraps a prompt + parameter + structural proposer into one, spending the cheapest level first.
79
- 2. **Race** — `compareProposers` runs N proposers on one holdout and returns a per-proposer lift CI + pairwise "who won".
80
- 3. **Plug in** — hand any composed proposer to the gated loop via `runImprovementLoop({ proposer })`, or the one-liner `improve({ surface, generator })` in `@tangle-network/agent-runtime`.
78
+ 1. **Portfolio** — `compositeProposer` splits one generation's population across member proposers by fixed weights and returns one provenance-labelled pool.
79
+ 2. **Escalate** — `fapoProposer` wraps prompt + parameter + structural levels into one proposer and spends on the cheapest level until evidence says to escalate.
80
+ 3. **Race** — `compareProposers` gives proposers separate loops, then re-scores their winners on one holdout and returns per-proposer lift intervals plus pairwise results.
81
+ 4. **Plug in** — hand any proposer to `runImprovementLoop({ proposer })`, or use `improve({ surface, generator })` in `@tangle-network/agent-runtime`.
81
82
 
82
- ### 1 + 3 — compose by escalation, then run the gated loop
83
+ ### 2 + 4 — compose by escalation, then run the improvement loop
83
84
 
84
85
  ```ts
85
86
  import {
@@ -93,8 +94,8 @@ import {
93
94
  const llm = { baseUrl: process.env.TANGLE_BASE_URL, apiKey: process.env.TANGLE_API_KEY }
94
95
  const model = 'deepseek-v4-flash'
95
96
 
96
- // Compose: prompt edits first (GEPA), escalate to a config knob only when
97
- // prompt-level search plateaus. `fapoProposer` IS the composite proposer.
97
+ // Compose by escalation: prompt edits first (GEPA), then a config knob only
98
+ // when prompt-level search plateaus. This is distinct from a peer portfolio.
98
99
  const proposer = fapoProposer({
99
100
  scope: { allowedLevels: ['prompt', 'parameter'] }, // no structural/code tier here
100
101
  promptProposer: gepaProposer({ llm, model, target: 'agent system prompt' }),
@@ -147,7 +148,7 @@ const out = await improve(profile, findings, {
147
148
  if (out.shipped) deploy(out.profile) // out.lift is the held-out winner − baseline
148
149
  ```
149
150
 
150
- ### 2 — race proposers head-to-head for a lift CI
151
+ ### 3 — race proposers head-to-head for a lift CI
151
152
 
152
153
  ```ts
153
154
  import {
@@ -196,7 +197,7 @@ Every entrant is re-scored on the **same** holdout with the **same** judges, so
196
197
  - Do not put eval logic inside a proposer — scoring lives in `dispatch` + `judges`, proposing lives in the proposer.
197
198
  - Do not let a proposer read held-out judge scores — `ProposeContext` makes that a compile error on purpose; a proposer that games the acceptance axis is an oracle, not an optimizer.
198
199
  - Do not read `lift` without `result.power`/MDE — a "+4" on a valset too small to detect +4 is noise wearing a number.
199
- - Do not look for a `compositeProposer` `fapoProposer` is the composite; `compareProposers` is the race; `runImprovementLoop`/`improve` is the plug.
200
+ - Do not confuse `compositeProposer` with `fapoProposer`: the former allocates one fixed population across peers, while the latter escalates through ordered levels from cheaper to more structural changes.
200
201
 
201
202
  ### neutralizationGate — the placebo / content-causality control
202
203
 
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@tangle-network/agent-eval",
3
- "version": "0.109.1",
3
+ "version": "0.110.1",
4
4
  "description": "Evaluate and improve AI agents from runs, traces, judges, and feedback. Compare candidates, cluster failures, measure lift, and gate releases.",
5
5
  "homepage": "https://github.com/tangle-network/agent-eval#readme",
6
6
  "repository": {
@@ -39,11 +39,6 @@
39
39
  "import": "./dist/rl.js",
40
40
  "default": "./dist/rl.js"
41
41
  },
42
- "./diagnose": {
43
- "types": "./dist/diagnose.d.ts",
44
- "import": "./dist/diagnose.js",
45
- "default": "./dist/diagnose.js"
46
- },
47
42
  "./fuzz": {
48
43
  "types": "./dist/fuzz.d.ts",
49
44
  "import": "./dist/fuzz.js",
@@ -54,16 +49,6 @@
54
49
  "import": "./dist/traces.js",
55
50
  "default": "./dist/traces.js"
56
51
  },
57
- "./telemetry": {
58
- "types": "./dist/telemetry/index.d.ts",
59
- "import": "./dist/telemetry/index.js",
60
- "default": "./dist/telemetry/index.js"
61
- },
62
- "./telemetry/file": {
63
- "types": "./dist/telemetry/file.d.ts",
64
- "import": "./dist/telemetry/file.js",
65
- "default": "./dist/telemetry/file.js"
66
- },
67
52
  "./wire": {
68
53
  "types": "./dist/wire/index.d.ts",
69
54
  "import": "./dist/wire/index.js",
@@ -84,41 +69,16 @@
84
69
  "import": "./dist/meta-eval/index.js",
85
70
  "default": "./dist/meta-eval/index.js"
86
71
  },
87
- "./prm": {
88
- "types": "./dist/prm/index.d.ts",
89
- "import": "./dist/prm/index.js",
90
- "default": "./dist/prm/index.js"
91
- },
92
72
  "./builder-eval": {
93
73
  "types": "./dist/builder-eval/index.d.ts",
94
74
  "import": "./dist/builder-eval/index.js",
95
75
  "default": "./dist/builder-eval/index.js"
96
76
  },
97
- "./governance": {
98
- "types": "./dist/governance/index.d.ts",
99
- "import": "./dist/governance/index.js",
100
- "default": "./dist/governance/index.js"
101
- },
102
- "./knowledge": {
103
- "types": "./dist/knowledge/index.d.ts",
104
- "import": "./dist/knowledge/index.js",
105
- "default": "./dist/knowledge/index.js"
106
- },
107
77
  "./matrix": {
108
78
  "types": "./dist/matrix/index.d.ts",
109
79
  "import": "./dist/matrix/index.js",
110
80
  "default": "./dist/matrix/index.js"
111
81
  },
112
- "./perf": {
113
- "types": "./dist/perf/index.d.ts",
114
- "import": "./dist/perf/index.js",
115
- "default": "./dist/perf/index.js"
116
- },
117
- "./product-benchmark": {
118
- "types": "./dist/product-benchmark/index.d.ts",
119
- "import": "./dist/product-benchmark/index.js",
120
- "default": "./dist/product-benchmark/index.js"
121
- },
122
82
  "./multishot": {
123
83
  "types": "./dist/multishot/index.d.ts",
124
84
  "import": "./dist/multishot/index.js",
@@ -139,51 +99,21 @@
139
99
  "import": "./dist/authenticity/index.js",
140
100
  "default": "./dist/authenticity/index.js"
141
101
  },
142
- "./groundedness": {
143
- "types": "./dist/groundedness/index.d.ts",
144
- "import": "./dist/groundedness/index.js",
145
- "default": "./dist/groundedness/index.js"
146
- },
147
102
  "./belief-state": {
148
103
  "types": "./dist/belief-state/index.d.ts",
149
104
  "import": "./dist/belief-state/index.js",
150
105
  "default": "./dist/belief-state/index.js"
151
106
  },
152
- "./workflow": {
153
- "types": "./dist/workflow/index.d.ts",
154
- "import": "./dist/workflow/index.js",
155
- "default": "./dist/workflow/index.js"
156
- },
157
107
  "./contract": {
158
108
  "types": "./dist/contract/index.d.ts",
159
109
  "import": "./dist/contract/index.js",
160
110
  "default": "./dist/contract/index.js"
161
111
  },
162
- "./adapters/langchain": {
163
- "types": "./dist/adapters/langchain.d.ts",
164
- "import": "./dist/adapters/langchain.js",
165
- "default": "./dist/adapters/langchain.js"
166
- },
167
- "./adapters/http": {
168
- "types": "./dist/adapters/http.d.ts",
169
- "import": "./dist/adapters/http.js",
170
- "default": "./dist/adapters/http.js"
171
- },
172
- "./adapters/otel": {
173
- "types": "./dist/adapters/otel.d.ts",
174
- "import": "./dist/adapters/otel.js",
175
- "default": "./dist/adapters/otel.js"
176
- },
177
112
  "./hosted": {
178
113
  "types": "./dist/hosted/index.d.ts",
179
114
  "import": "./dist/hosted/index.js",
180
115
  "default": "./dist/hosted/index.js"
181
116
  },
182
- "./testing": {
183
- "types": "./dist/testing.d.ts",
184
- "import": "./dist/testing.js",
185
- "default": "./dist/testing.js"
186
- },
187
117
  "./openapi.json": {
188
118
  "default": "./dist/openapi.json"
189
119
  }