@tangle-network/agent-eval 0.94.0 → 0.95.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (140) hide show
  1. package/CHANGELOG.md +32 -0
  2. package/README.md +44 -30
  3. package/dist/adapters/http.d.ts +8 -7
  4. package/dist/adapters/http.js.map +1 -1
  5. package/dist/adapters/langchain.d.ts +3 -2
  6. package/dist/adapters/otel.d.ts +5 -4
  7. package/dist/analyst/index.d.ts +11 -31
  8. package/dist/analyst/index.js +5 -65
  9. package/dist/analyst/index.js.map +1 -1
  10. package/dist/{analyze-runs-B6Ljo_dI.d.ts → analyze-runs-DtT6F_6T.d.ts} +3 -3
  11. package/dist/belief-state/index.d.ts +4 -3
  12. package/dist/benchmarks/index.d.ts +3 -2
  13. package/dist/campaign/index.d.ts +727 -616
  14. package/dist/campaign/index.js +1863 -1316
  15. package/dist/campaign/index.js.map +1 -1
  16. package/dist/{chunk-2K6UUZ7P.js → chunk-2T4EZACH.js} +1 -1
  17. package/dist/chunk-2T4EZACH.js.map +1 -0
  18. package/dist/{chunk-CTBHKLEU.js → chunk-77T4STFI.js} +59 -86
  19. package/dist/chunk-77T4STFI.js.map +1 -0
  20. package/dist/{chunk-EGPMSBEZ.js → chunk-7QTQKIDD.js} +178 -177
  21. package/dist/chunk-7QTQKIDD.js.map +1 -0
  22. package/dist/{chunk-MIFZUPEK.js → chunk-AQ5WQAIV.js} +21 -6
  23. package/dist/chunk-AQ5WQAIV.js.map +1 -0
  24. package/dist/chunk-DJWX3GVS.js +81 -0
  25. package/dist/chunk-DJWX3GVS.js.map +1 -0
  26. package/dist/{chunk-S6OZEZQK.js → chunk-HMA63UEO.js} +37 -9
  27. package/dist/{chunk-S6OZEZQK.js.map → chunk-HMA63UEO.js.map} +1 -1
  28. package/dist/{chunk-TBDR6PAI.js → chunk-IZCEK2HR.js} +2 -2
  29. package/dist/{chunk-SD2YFWQQ.js → chunk-KKWJD5E6.js} +20 -20
  30. package/dist/chunk-KKWJD5E6.js.map +1 -0
  31. package/dist/{chunk-KWRRMR3J.js → chunk-LO6IOIJ2.js} +10 -10
  32. package/dist/chunk-LO6IOIJ2.js.map +1 -0
  33. package/dist/{chunk-E4GH6USR.js → chunk-NZEQVRH5.js} +2 -2
  34. package/dist/chunk-NZEQVRH5.js.map +1 -0
  35. package/dist/{chunk-MPQWFX6Y.js → chunk-PSWWQXHF.js} +13 -88
  36. package/dist/chunk-PSWWQXHF.js.map +1 -0
  37. package/dist/{chunk-Q5LIB7BC.js → chunk-S4SYLDFX.js} +2 -2
  38. package/dist/chunk-S4SYLDFX.js.map +1 -0
  39. package/dist/{chunk-KW53MSA5.js → chunk-X74V6ESX.js} +2 -2
  40. package/dist/{chunk-QMUEXQJS.js → chunk-YBIGNSCZ.js} +81 -4
  41. package/dist/chunk-YBIGNSCZ.js.map +1 -0
  42. package/dist/{chunk-2KNZHH3P.js → chunk-Z6L6YSU6.js} +2 -2
  43. package/dist/{code-agent-session-BO8nCnv3.d.ts → code-agent-session-CPHRCb4-.d.ts} +1 -1
  44. package/dist/contract/index.d.ts +91 -43
  45. package/dist/contract/index.js +127 -17
  46. package/dist/contract/index.js.map +1 -1
  47. package/dist/{control-D6qwHXIR.d.ts → control-Doncu-B_.d.ts} +2 -2
  48. package/dist/control.d.ts +3 -2
  49. package/dist/control.js +2 -2
  50. package/dist/{corpus-B8A4BDR3.d.ts → corpus-D4YW9UoJ.d.ts} +1 -1
  51. package/dist/{default-registry-6dhErQbs.d.ts → default-registry-GyE8X5SP.d.ts} +3 -3
  52. package/dist/diagnose.d.ts +4 -3
  53. package/dist/diagnose.js +1 -1
  54. package/dist/{run-improvement-loop-DBahB8Ax.d.ts → gepa-C1NCIZ9o.d.ts} +117 -130
  55. package/dist/hosted/index.d.ts +5 -4
  56. package/dist/{index-Bx3gZ8xl.d.ts → index-_Y4oNOOb.d.ts} +1 -1
  57. package/dist/index.d.ts +76 -81
  58. package/dist/index.js +66 -31
  59. package/dist/index.js.map +1 -1
  60. package/dist/{insight-report-DWl3z9tl.d.ts → insight-report-BnRjTibG.d.ts} +1 -1
  61. package/dist/{kind-factory-0BhLSI27.d.ts → kind-factory-X3eDYbKn.d.ts} +2 -3
  62. package/dist/matrix/index.d.ts +1 -1
  63. package/dist/meta-eval/index.d.ts +3 -2
  64. package/dist/multishot/index.d.ts +4 -4
  65. package/dist/multishot/index.js.map +1 -1
  66. package/dist/openapi.json +1 -1
  67. package/dist/{pre-registration-mAnCugl9.d.ts → pre-registration-nfUdc9EQ.d.ts} +2 -42
  68. package/dist/{provenance-P-bCL2Fo.d.ts → provenance-CncDq9qE.d.ts} +26 -41
  69. package/dist/{release-report-BEbWmVYj.d.ts → release-report-pidWUMZ2.d.ts} +2 -2
  70. package/dist/reporting.d.ts +5 -4
  71. package/dist/{researcher-B0C2_fVO.d.ts → researcher-Jr8ME1dZ.d.ts} +2 -2
  72. package/dist/rl.d.ts +516 -515
  73. package/dist/rl.js +612 -612
  74. package/dist/rl.js.map +1 -1
  75. package/dist/{rubric-predictive-validity-Cy_W-hWZ.d.ts → rubric-predictive-validity-C2hDKM8Z.d.ts} +1 -1
  76. package/dist/{run-campaign-7WNXMDSN.js → run-campaign-WXY7KI67.js} +2 -2
  77. package/dist/{run-record-e7vj1uZQ.d.ts → run-record-CP2ObebC.d.ts} +14 -18
  78. package/dist/{runtime-trajectory-BDgfGZSr.d.ts → runtime-trajectory-BOUUjI0y.d.ts} +1 -1
  79. package/dist/{semantic-concept-judge-B9MgmBnM.d.ts → semantic-concept-judge-DSBB2Cfp.d.ts} +2 -2
  80. package/dist/{summary-report-BDOFevaT.d.ts → summary-report-CInXwsza.d.ts} +1 -1
  81. package/dist/testing-C21CHsq2.d.ts +20 -0
  82. package/dist/testing.d.ts +1 -0
  83. package/dist/testing.js +8 -0
  84. package/dist/testing.js.map +1 -0
  85. package/dist/traces.d.ts +26 -10
  86. package/dist/traces.js +41 -11
  87. package/dist/{types-Ce17tDlG.d.ts → types-B5x54y6n.d.ts} +1 -1
  88. package/dist/{types-mn5Aqk7x.d.ts → types-BUxNaJ8c.d.ts} +2 -4
  89. package/dist/{types-BU-7W85F.d.ts → types-DQRY8ZT-.d.ts} +60 -58
  90. package/dist/workflow/index.d.ts +5 -4
  91. package/dist/workflow/index.js +1 -1
  92. package/docs/campaign-proposers.md +170 -0
  93. package/docs/concepts.md +8 -4
  94. package/docs/customer-journeys.md +15 -13
  95. package/docs/design/loop-taxonomy.md +34 -66
  96. package/docs/distributed-driver.md +14 -14
  97. package/docs/feature-guide.md +1 -1
  98. package/docs/hosted-ingest-spec.md +2 -3
  99. package/docs/multi-shot-optimization.md +8 -8
  100. package/docs/product-eval-adoption.md +1 -1
  101. package/docs/self-improvement-map.md +33 -29
  102. package/package.json +8 -14
  103. package/dist/chunk-2K6UUZ7P.js.map +0 -1
  104. package/dist/chunk-CTBHKLEU.js.map +0 -1
  105. package/dist/chunk-E4GH6USR.js.map +0 -1
  106. package/dist/chunk-EGPMSBEZ.js.map +0 -1
  107. package/dist/chunk-KWRRMR3J.js.map +0 -1
  108. package/dist/chunk-MIFZUPEK.js.map +0 -1
  109. package/dist/chunk-MPQWFX6Y.js.map +0 -1
  110. package/dist/chunk-Q5LIB7BC.js.map +0 -1
  111. package/dist/chunk-QMUEXQJS.js.map +0 -1
  112. package/dist/chunk-SD2YFWQQ.js.map +0 -1
  113. package/docs/design/external-agent-wedge.md +0 -89
  114. package/docs/design/phase-d-rfc.md +0 -125
  115. package/docs/design/phase4-consumer-migration.md +0 -70
  116. package/docs/design/primitives-integration-spec.md +0 -393
  117. package/docs/design/product-self-improvement-loop.md +0 -146
  118. package/docs/design/self-improvement-engine.md +0 -140
  119. package/docs/design/self-improvement-protocol.md +0 -223
  120. package/docs/design/self-improvement-roadmap.md +0 -106
  121. package/docs/design/substrate-gaps.md +0 -118
  122. package/docs/phase-b-pairing-kit.md +0 -188
  123. package/docs/phase-b-runbook.md +0 -176
  124. package/docs/pilot/README.md +0 -62
  125. package/docs/pilot/customer-checklist.md +0 -90
  126. package/docs/pilot/integration-foreign-stack.md +0 -296
  127. package/docs/pilot/integration-tangle-stack.md +0 -248
  128. package/docs/pilot/one-pager.md +0 -161
  129. package/docs/pilot/sample-insight-report.json +0 -172
  130. package/docs/quickstart-external.md +0 -229
  131. package/docs/research/belief-state-agent-eval-roadmap.md +0 -593
  132. package/docs/research/research-roadmap.md +0 -205
  133. package/docs/specs/driver-honest-spec.md +0 -251
  134. package/docs/specs/hermes-self-improvement-audit.md +0 -93
  135. package/docs/specs/profile-versioning.md +0 -291
  136. package/docs/three-package-architecture.md +0 -168
  137. /package/dist/{chunk-TBDR6PAI.js.map → chunk-IZCEK2HR.js.map} +0 -0
  138. /package/dist/{chunk-KW53MSA5.js.map → chunk-X74V6ESX.js.map} +0 -0
  139. /package/dist/{chunk-2KNZHH3P.js.map → chunk-Z6L6YSU6.js.map} +0 -0
  140. /package/dist/{run-campaign-7WNXMDSN.js.map → run-campaign-WXY7KI67.js.map} +0 -0
@@ -3,7 +3,7 @@ import { C as ControlEvalResult, a as ControlRunResult, h as ControlRuntimeConfi
3
3
  import { T as TraceEmitter } from './emitter-C2rqGH_l.js';
4
4
  import { F as FailureClass } from './schema-m0gsnbt3.js';
5
5
  import { T as TraceStore } from './store-BcFXE6LG.js';
6
- import { a as RunSplitTag, b as RunTokenUsage, R as RunRecord } from './run-record-e7vj1uZQ.js';
6
+ import { a as RunSplitTag, b as RunTokenUsage, R as RunRecord } from './run-record-CP2ObebC.js';
7
7
 
8
8
  interface ActionExecutionPolicy {
9
9
  allowedTypes?: string[];
@@ -52,7 +52,7 @@ interface ControlRunToRunRecordOptions extends RunEvidenceMetadata {
52
52
  * release gates, optimizer tables, and research reports.
53
53
  *
54
54
  * The control loop owns live execution evidence. The caller still supplies the
55
- * experimental cell metadata because prompt/config hashes, split assignment,
55
+ * experiment-cell metadata because prompt/config hashes, split assignment,
56
56
  * model snapshot, and commit SHA are product/harness concerns.
57
57
  */
58
58
  declare function controlRunToRunRecord<TState, TAction, TActionResult, TEval extends ControlEvalResult = ControlEvalResult>(run: ControlRunResult<TState, TAction, TActionResult, TEval>, options: ControlRunToRunRecordOptions): RunRecord;
package/dist/control.d.ts CHANGED
@@ -1,4 +1,4 @@
1
- export { A as ActionExecutionPolicy, a as ActionPolicyDecision, C as ControlRunToRunRecordOptions, e as ProposeReviewConfig, f as ProposeReviewControlAction, g as ProposeReviewControlConfig, h as ProposeReviewControlResult, i as ProposeReviewControlState, j as ProposeReviewReport, p as RunEvidenceMetadata, s as controlRunToRunRecord, u as evaluateActionPolicy, x as runProposeReview, y as runProposeReviewAsControlLoop, z as scoreFromEvals } from './control-D6qwHXIR.js';
1
+ export { A as ActionExecutionPolicy, a as ActionPolicyDecision, C as ControlRunToRunRecordOptions, e as ProposeReviewConfig, f as ProposeReviewControlAction, g as ProposeReviewControlConfig, h as ProposeReviewControlResult, i as ProposeReviewControlState, j as ProposeReviewReport, p as RunEvidenceMetadata, s as controlRunToRunRecord, u as evaluateActionPolicy, x as runProposeReview, y as runProposeReviewAsControlLoop, z as scoreFromEvals } from './control-Doncu-B_.js';
2
2
  export { c as ControlActionFailureMode, d as ControlActionOutcome, e as ControlBudget, f as ControlContext, g as ControlDecision, C as ControlEvalResult, a as ControlRunResult, h as ControlRuntimeConfig, i as ControlRuntimeError, j as ControlSeverity, b as ControlStep, k as ControlStopPolicies, S as StopDecision, l as allCriticalPassed, o as objectiveEval, r as runAgentControlLoop, s as stopOnNoProgress, m as stopOnRepeatedAction, n as subjectiveEval } from './control-runtime-Acf9CGhw.js';
3
3
  import './feedback-trajectory-BxY0cKfs.js';
4
4
  import './dataset-BbGkaN2I.js';
@@ -6,4 +6,5 @@ import './errors-CzMUYo7b.js';
6
6
  import './emitter-C2rqGH_l.js';
7
7
  import './schema-m0gsnbt3.js';
8
8
  import './store-BcFXE6LG.js';
9
- import './run-record-e7vj1uZQ.js';
9
+ import './run-record-CP2ObebC.js';
10
+ import '@tangle-network/agent-interface';
package/dist/control.js CHANGED
@@ -4,7 +4,7 @@ import {
4
4
  runProposeReview,
5
5
  runProposeReviewAsControlLoop,
6
6
  scoreFromEvals
7
- } from "./chunk-E4GH6USR.js";
7
+ } from "./chunk-NZEQVRH5.js";
8
8
  import {
9
9
  allCriticalPassed,
10
10
  objectiveEval,
@@ -13,7 +13,7 @@ import {
13
13
  stopOnRepeatedAction,
14
14
  subjectiveEval
15
15
  } from "./chunk-YEHAEDUD.js";
16
- import "./chunk-KWRRMR3J.js";
16
+ import "./chunk-LO6IOIJ2.js";
17
17
  import "./chunk-TVVP3ZZQ.js";
18
18
  import "./chunk-VSMTAMNK.js";
19
19
  import "./chunk-3BFEG2F6.js";
@@ -1,4 +1,4 @@
1
- import { R as RunRecord, a as RunSplitTag } from './run-record-e7vj1uZQ.js';
1
+ import { R as RunRecord, a as RunSplitTag } from './run-record-CP2ObebC.js';
2
2
  import { S as Span } from './schema-m0gsnbt3.js';
3
3
  import { T as TraceStore } from './store-BcFXE6LG.js';
4
4
 
@@ -1,6 +1,6 @@
1
1
  import { AxAIService } from '@ax-llm/ax';
2
- import { T as TraceAnalystKindSpec } from './kind-factory-0BhLSI27.js';
3
- import { A as Analyst, a as AnalystContext, b as AnalystRunSummary, c as AnalystFinding, d as AnalystRunResult, C as ChatClient, e as AnalystRunInputs, f as AnalystRunEvent } from './types-Ce17tDlG.js';
2
+ import { T as TraceAnalystKindSpec } from './kind-factory-X3eDYbKn.js';
3
+ import { A as Analyst, a as AnalystContext, b as AnalystRunSummary, c as AnalystFinding, d as AnalystRunResult, C as ChatClient, e as AnalystRunInputs, f as AnalystRunEvent } from './types-B5x54y6n.js';
4
4
 
5
5
  /**
6
6
  * AnalystRegistry — orchestrate N analysts against one run.
@@ -152,4 +152,4 @@ interface DefaultAnalystRegistryOptions {
152
152
  }
153
153
  declare function buildDefaultAnalystRegistry(opts?: DefaultAnalystRegistryOptions): AnalystRegistry;
154
154
 
155
- export { AnalystRegistry as A, type BudgetPolicy as B, type DefaultAnalystRegistryOptions as D, type RegistryRunOpts as R, type AnalystHooks as a, buildDefaultAnalystRegistry as b, type AnalystRegistryOptions as c };
155
+ export { AnalystRegistry as A, type BudgetPolicy as B, type DefaultAnalystRegistryOptions as D, type RegistryRunOpts as R, type AnalystHooks as a, type AnalystRegistryOptions as b, buildDefaultAnalystRegistry as c };
@@ -3,9 +3,9 @@ export { c as CounterfactualContext, d as CounterfactualResult } from './counter
3
3
  import { S as Span } from './schema-m0gsnbt3.js';
4
4
  import { T as TraceStore } from './store-BcFXE6LG.js';
5
5
  import { a as TrajectoryStep, T as Trajectory } from './trajectory-2TkpSEVh.js';
6
- import { h as AnalystSeverity, c as AnalystFinding } from './types-Ce17tDlG.js';
7
- import { C as CorpusRecord } from './corpus-B8A4BDR3.js';
8
- import { R as RunRecord } from './run-record-e7vj1uZQ.js';
6
+ import { h as AnalystSeverity, c as AnalystFinding } from './types-B5x54y6n.js';
7
+ import { C as CorpusRecord } from './corpus-D4YW9UoJ.js';
8
+ import { R as RunRecord } from './run-record-CP2ObebC.js';
9
9
  import './emitter-C2rqGH_l.js';
10
10
  import './store-C1YxJDEK.js';
11
11
  import './types-C7DGg5ex.js';
@@ -13,6 +13,7 @@ import '@tangle-network/tcloud';
13
13
  import './llm-client-Bj7g0rqu.js';
14
14
  import './errors-CzMUYo7b.js';
15
15
  import './raw-provider-sink-C46HDghv.js';
16
+ import '@tangle-network/agent-interface';
16
17
 
17
18
  /**
18
19
  * Causal sweep — WHY did this run fail?
package/dist/diagnose.js CHANGED
@@ -13,7 +13,7 @@ import {
13
13
  } from "./chunk-56GC6VYO.js";
14
14
  import {
15
15
  validateRunRecord
16
- } from "./chunk-KWRRMR3J.js";
16
+ } from "./chunk-LO6IOIJ2.js";
17
17
  import "./chunk-TVVP3ZZQ.js";
18
18
  import "./chunk-VSMTAMNK.js";
19
19
  import {
@@ -1,96 +1,7 @@
1
+ import { S as Scenario, C as CampaignResult, l as GateResult, D as DispatchFn, a as JudgeConfig, L as LabeledScenarioStore, i as CampaignTraceWriter, n as GenerationRecord, M as MutableSurface, P as ParetoParent, c as SurfaceProposer, G as Gate } from './types-DQRY8ZT-.js';
1
2
  import { L as LlmClientOptions } from './llm-client-Bj7g0rqu.js';
2
- import { I as ImprovementDriver, S as Scenario, g as CampaignResult, k as GateResult, D as DispatchFn, a as JudgeConfig, L as LabeledScenarioStore, h as CampaignTraceWriter, m as GenerationRecord, M as MutableSurface, P as ParetoParent, G as Gate } from './types-BU-7W85F.js';
3
3
 
4
4
  /**
5
- * @experimental
6
- *
7
- * `gepaDriver` — a reflective `ImprovementDriver` for prompt-tier surfaces.
8
- * Each generation it reflects on the prior best candidate's per-scenario
9
- * scores + weakest dimensions, asks an LLM to propose targeted rewrites of
10
- * the current surface, and returns them as the next population.
11
- *
12
- * Maps onto the GEPA paper (Agrawal et al., arXiv:2507.19457):
13
- * - *Reflection*: each generation reflects on the best parent's weakest
14
- * dimensions + per-scenario top/bottom scores to propose targeted rewrites.
15
- * - *Pareto frontier*: `runOptimization` maintains the non-dominated set of
16
- * surfaces across generations (per-scenario objective vectors) and supplies
17
- * it as `ctx.paretoParents`. A surface uniquely best on one hard scenario
18
- * survives even when its mean composite is lower.
19
- * - *Combine complementary lessons*: when the frontier has >1 member, the
20
- * first population slot is a merge of those parents' strengths (one LLM
21
- * call citing each parent's winning scenarios). Toggle via `combineParents`.
22
- * Dominance is computed by the package-canonical `paretoFrontier` (`pareto.ts`).
23
- *
24
- * Optional `constraints` move structured-doc guards into the driver
25
- * (preserve H2 section headings, cap sentence-level edits) — useful when
26
- * the surface IS a structured procedure like a SKILL.md / runbook /
27
- * judge rubric. When `constraints` is omitted, behavior is unchanged.
28
- *
29
- * The driver is surface-agnostic — any string surface in any consumer opts
30
- * in by selecting it. Reuses the generic reflection primitive
31
- * (`buildReflectionPrompt` / `parseReflectionResponse`) and the router client.
32
- *
33
- * Earns its keep where there is real per-instance signal (which the
34
- * dimensional + per-scenario evidence + the `LabeledScenarioStore` flywheel
35
- * now provide). For thin-signal surfaces it degrades to plain reflection.
36
- * On generation 0 (no history) it reflects on the current surface against
37
- * the mutation primitives alone.
38
- */
39
-
40
- interface GepaDriverConstraints {
41
- /** H2 section headings that MUST appear unchanged in every candidate.
42
- * When set, the driver auto-detects current H2s if this is empty AND
43
- * rejects any candidate that drops or renames a preserved heading.
44
- * Use when the surface is a structured doc (SKILL.md, runbook,
45
- * sectioned system prompt, judge rubric). */
46
- preserveSections?: string[];
47
- /** Maximum sentence-level edits per candidate vs the parent surface.
48
- * Rejection threshold = maxSentenceEdits × 2 (counts adds + removes).
49
- * Inspired by SkillOpt's edit-budget as a "textual learning rate."
50
- * Cap prevents an LLM rewrite from overwriting useful prior rules. */
51
- maxSentenceEdits?: number;
52
- }
53
- interface GepaDriverOptions {
54
- /** Router transport (apiKey/baseUrl). */
55
- llm: LlmClientOptions;
56
- /** Model that performs the reflection. */
57
- model: string;
58
- /** What is being optimized — appears in the reflection prompt for orientation. */
59
- target: string;
60
- /** Surface-specific mutation levers offered to the model. */
61
- mutationPrimitives?: string[];
62
- /** Top/bottom scenarios surfaced as evidence each generation. Default 3. */
63
- evidenceK?: number;
64
- /** Reflection sampling temperature. Default 0.7. */
65
- temperature?: number;
66
- /** Reflection max tokens. Default 6000. */
67
- maxTokens?: number;
68
- /** Structured-doc constraints. Candidates violating any are rejected
69
- * post-parse and dropped from the returned population. */
70
- constraints?: GepaDriverConstraints;
71
- /** GEPA combine-complementary-lessons: when the loop supplies a Pareto
72
- * frontier of >1 non-dominated parents (`ctx.paretoParents`), spend one
73
- * slot of the population on a merge of their strengths. Default `true` —
74
- * this is the GEPA-faithful behavior; the merge only fires once the
75
- * frontier has more than one member (generation ≥ 1). Set `false` for
76
- * pure single-parent reflection. */
77
- combineParents?: boolean;
78
- /** Cap on how many frontier parents feed one combine prompt (highest
79
- * composite first), to bound prompt size. Default 4. */
80
- combineMaxParents?: number;
81
- }
82
- declare function gepaDriver(opts: GepaDriverOptions): ImprovementDriver;
83
- /** Extract H2 headings (`## Foo`) from a markdown surface. Exported for
84
- * consumers building custom mutators that share the same invariant. */
85
- declare function extractH2Sections(text: string): string[];
86
- /** Sentence-level edit distance — count distinct add/remove ops between
87
- * two surfaces via a normalised line-by-line set diff. Treats trivial
88
- * whitespace as identical. Exported for tests + consumer-side validators. */
89
- declare function countSentenceEdits(baseline: string, candidate: string): number;
90
-
91
- /**
92
- * @experimental
93
- *
94
5
  * `openAutoPr` — thin shell-out helper for the `runImprovementLoop` preset's
95
6
  * `autoOnPromote: 'pr'` mode. Substitutes for the per-product PR-opening
96
7
  * code consumers duplicated 4 times. The PR body includes the campaign's
@@ -137,8 +48,6 @@ interface OpenAutoPrResult {
137
48
  declare function openAutoPr<TArtifact, TScenario extends Scenario>(options: OpenAutoPrOptions<TArtifact, TScenario>): OpenAutoPrResult;
138
49
 
139
50
  /**
140
- * @experimental
141
- *
142
51
  * `CampaignStorage` — the filesystem seam `runCampaign` writes through
143
52
  * (run/cell dirs, the resumability cache, per-cell artifacts, trace spans).
144
53
  *
@@ -177,8 +86,6 @@ declare function fsCampaignStorage(): CampaignStorage;
177
86
  declare function inMemoryCampaignStorage(): CampaignStorage;
178
87
 
179
88
  /**
180
- * @experimental
181
- *
182
89
  * `runCampaign` — Pass A substrate primitive. ONE function that orchestrates
183
90
  * scenarios → dispatch → artifacts → judges → aggregates, with full
184
91
  * reproducibility (seed + manifest hash), cell-level resumability, bootstrap
@@ -268,51 +175,47 @@ interface RunCampaignOptions<TScenario extends Scenario, TArtifact> {
268
175
  declare function runCampaign<TScenario extends Scenario, TArtifact>(opts: RunCampaignOptions<TScenario, TArtifact>): Promise<CampaignResult<TArtifact, TScenario>>;
269
176
 
270
177
  /**
271
- * @experimental
272
- *
273
178
  * `runOptimization` — the improvement loop body. Runs N generations: the
274
- * `ImprovementDriver` proposes K candidate surfaces per generation, each
179
+ * `SurfaceProposer` proposes K candidate surfaces per generation, each
275
180
  * candidate runs a campaign (the measurement), top-scoring promote to the
276
- * next generation. Driver-agnostic — the same loop runs an evolutionary
277
- * population mutator (`evolutionaryDriver`) or agent-runtime's
278
- * `improvementDriver` (reflective / agentic generators); they differ only in
279
- * how `propose()` picks candidates.
181
+ * next generation. Proposer-agnostic — the same loop runs an evolutionary
182
+ * population mutator (`evolutionaryProposer`) or any reflective / agentic
183
+ * proposer; they differ only in how `propose()` picks candidates.
280
184
  *
281
185
  * This is `runLoop`'s shape (plan → measure → decide) specialized to surface
282
- * improvement: `driver.propose` = plan, `runCampaign` = the measurement (which
283
- * runs the worker behind `dispatch`), the mean-composite ranking = the
284
- * validator, `driver.decide` = the stop check.
186
+ * improvement: `proposer.propose` = plan, `runCampaign` = the measurement
187
+ * (which runs the worker behind `dispatch`), the mean-composite ranking = the
188
+ * validator, `proposer.decide` = the stop check.
285
189
  *
286
190
  * The gated-promotion shell (`runImprovementLoop`) wraps this with a holdout
287
191
  * re-score + release gate + optional PR.
288
192
  */
289
193
 
290
- interface RunOptimizationOptions<TScenario extends Scenario, TArtifact> extends Omit<RunCampaignOptions<TScenario, TArtifact>, 'dispatch'> {
194
+ interface RunOptimizationBaseOptions<TScenario extends Scenario, TArtifact> extends Omit<RunCampaignOptions<TScenario, TArtifact>, 'dispatch'> {
291
195
  /** Initial mutable surface (typically system prompt or addendum). */
292
196
  baselineSurface: MutableSurface;
293
197
  /** Dispatcher that takes the CURRENT surface + scenario → artifact. */
294
198
  dispatchWithSurface: (surface: MutableSurface, scenario: TScenario, ctx: Parameters<RunCampaignOptions<TScenario, TArtifact>['dispatch']>[1]) => Promise<TArtifact>;
295
- /** The improvement strategy. Wrap a population `Mutator` via
296
- * `evolutionaryDriver({ mutator })`, or pass agent-runtime's
297
- * `improvementDriver` (reflective / agentic generators). */
298
- driver: ImprovementDriver;
199
+ /** The candidate-generation strategy. Wrap a population `Mutator` via
200
+ * `evolutionaryProposer({ mutator })`, or pass any reflective / agentic
201
+ * proposer that implements `SurfaceProposer`. */
202
+ proposer: SurfaceProposer;
299
203
  populationSize: number;
300
204
  maxGenerations: number;
301
205
  /** How many top-scoring candidates carry to the next generation. Default 2. */
302
206
  promoteTopK?: number;
303
- /** DEPTH knob forwarded to the driver's `propose()` — max iterations the
207
+ /** DEPTH knob forwarded to the proposer's `propose()` — max iterations the
304
208
  * agentic generator may take per candidate. */
305
209
  maxImprovementShots?: number;
306
- /** Phase-2 research report forwarded to `propose()` (analyst findings +
307
- * diff). Opaque here; the driver types it. */
210
+ /** Optional analysis report forwarded to `propose()`. Opaque here; the
211
+ * proposer types it. */
308
212
  report?: unknown;
309
213
  /** Structured findings forwarded to `propose()` as `ctx.findings`. A
310
- * findings producer (trace-analyst registry, HALO) emits these from the
311
- * generation's traces; findings-grounded drivers (`improvementDriver`,
312
- * `memoryCurationDriver`, `traceAnalystDriver`) consume them. Opaque here;
313
- * the driver types its `TFindings`. Empty when no producer is wired. */
214
+ * findings producer emits these from the
215
+ * generation's traces; findings-grounded proposers consume them. Opaque here;
216
+ * the proposer types its `TFindings`. Empty when no producer is wired. */
314
217
  findings?: unknown[];
315
- /** Per-generation findings producer — the EYES→HANDS loop closure. After each
218
+ /** Per-generation findings producer. After each
316
219
  * generation's candidates are scored, this is called with that generation's
317
220
  * results; whatever it returns REPLACES `ctx.findings` for the NEXT
318
221
  * generation's `propose()`, so the diagnosis is refreshed each round instead
@@ -331,6 +234,7 @@ interface RunOptimizationOptions<TScenario extends Scenario, TArtifact> extends
331
234
  history: GenerationRecord[];
332
235
  }) => Promise<unknown[]>;
333
236
  }
237
+ type RunOptimizationOptions<TScenario extends Scenario, TArtifact> = RunOptimizationBaseOptions<TScenario, TArtifact>;
334
238
  interface RunOptimizationResult<TArtifact, TScenario extends Scenario> {
335
239
  generations: Array<{
336
240
  record: GenerationRecord;
@@ -342,11 +246,11 @@ interface RunOptimizationResult<TArtifact, TScenario extends Scenario> {
342
246
  }>;
343
247
  winnerSurface: MutableSurface;
344
248
  winnerSurfaceHash: string;
345
- /** Driver label for the promoted surface. Present when the winning
346
- * candidate came from a `ProposedCandidate` (a reflective driver);
249
+ /** Proposer label for the promoted surface. Present when the winning
250
+ * candidate came from a `ProposedCandidate` (a reflective proposer);
347
251
  * absent when the winner is the baseline or a bare-surface mutator. */
348
252
  winnerLabel?: string;
349
- /** Driver rationale for the promoted surface — the "because Z" that
253
+ /** Proposer rationale for the promoted surface — the "because Z" that
350
254
  * motivated the winning change. Survives to `SelfImproveResult` and the
351
255
  * emitted provenance record. Absent when the winner is the baseline. */
352
256
  winnerRationale?: string;
@@ -362,15 +266,13 @@ declare function runOptimization<TScenario extends Scenario, TArtifact>(opts: Ru
362
266
  declare function surfaceHash(surface: MutableSurface): string;
363
267
 
364
268
  /**
365
- * @experimental
366
- *
367
269
  * `runImprovementLoop` — the gated-promotion shell around the improvement
368
- * loop body (`runOptimization`). Drives candidate surfaces via the
369
- * `ImprovementDriver`, re-scores the winner against the baseline on a
270
+ * loop body (`runOptimization`). Proposes candidate surfaces via the
271
+ * `SurfaceProposer`, re-scores the winner against the baseline on a
370
272
  * holdout set, runs the release gate, and optionally opens a PR.
371
273
  *
372
274
  * Role vocabulary (see docs/design/loop-taxonomy.md):
373
- * - DRIVER = the `ImprovementDriver` (evolutionary GEPA mutator OR
275
+ * - PROPOSER = the `SurfaceProposer` (evolutionary GEPA mutator OR
374
276
  * reflective analyst). Proposes candidate SURFACES — the
375
277
  * worker's system prompt / tool config — NOT conversation
376
278
  * turns.
@@ -380,16 +282,16 @@ declare function surfaceHash(surface: MutableSurface): string;
380
282
  * topology-opaque `dispatch` seam — never referenced here.
381
283
  *
382
284
  * Distinct from `runLoop` in `@tangle-network/agent-runtime`, which is the
383
- * INNER conversation loop (driver↔workers in a sandbox). `runImprovementLoop`
285
+ * INNER conversation loop (execution driver workers in a sandbox). `runImprovementLoop`
384
286
  * is the OUTER loop: it improves the surface that those workers run.
385
287
  *
386
288
  * Hard-refuses unsafe configurations:
387
- * - `tracing: 'off'` when a driver is wired (improvement is unattributable)
289
+ * - `tracing: 'off'` when a proposer is wired (improvement is unattributable)
388
290
  * - `autoOnPromote: 'config'` — DEFERRED to Pass B; v0.40 only ships
389
291
  * `'pr'` and `'none'`.
390
292
  */
391
293
 
392
- interface RunImprovementLoopOptions<TScenario extends Scenario, TArtifact> extends RunOptimizationOptions<TScenario, TArtifact> {
294
+ type RunImprovementLoopOptions<TScenario extends Scenario, TArtifact> = RunOptimizationOptions<TScenario, TArtifact> & {
393
295
  /** Holdout scenarios kept OUT of the training optimization pool — used
394
296
  * ONLY to score baseline vs winner for the gate. */
395
297
  holdoutScenarios: TScenario[];
@@ -409,7 +311,7 @@ interface RunImprovementLoopOptions<TScenario extends Scenario, TArtifact> exten
409
311
  /** Optional render override — substrate writes a diff-shaped surface; pass
410
312
  * a function to format the promoted surface differently. */
411
313
  renderPromotedDiff?: (winnerSurface: MutableSurface, baselineSurface: MutableSurface) => string;
412
- }
314
+ };
413
315
  interface RunImprovementLoopResult<TArtifact, TScenario extends Scenario> extends RunOptimizationResult<TArtifact, TScenario> {
414
316
  baselineOnHoldout: CampaignResult<TArtifact, TScenario>;
415
317
  winnerOnHoldout: CampaignResult<TArtifact, TScenario>;
@@ -424,4 +326,89 @@ interface RunImprovementLoopResult<TArtifact, TScenario extends Scenario> extend
424
326
  declare function runImprovementLoop<TScenario extends Scenario, TArtifact>(opts: RunImprovementLoopOptions<TScenario, TArtifact>): Promise<RunImprovementLoopResult<TArtifact, TScenario>>;
425
327
  declare function defaultRenderDiff(winnerSurface: MutableSurface, baselineSurface: MutableSurface): string;
426
328
 
427
- export { type CampaignStorage as C, type GepaDriverOptions as G, type OpenAutoPrOptions as O, type RunOptimizationOptions as R, type RunImprovementLoopResult as a, type RunCampaignOptions as b, type RunImprovementLoopOptions as c, runImprovementLoop as d, type GepaDriverConstraints as e, fsCampaignStorage as f, gepaDriver as g, type OpenAutoPrResult as h, inMemoryCampaignStorage as i, type RunOptimizationResult as j, countSentenceEdits as k, defaultRenderDiff as l, extractH2Sections as m, runOptimization as n, openAutoPr as o, runCampaign as r, surfaceHash as s };
329
+ /**
330
+ * `gepaProposer` — a reflective `SurfaceProposer` for prompt-tier surfaces.
331
+ * Each generation it reflects on the prior best candidate's per-scenario
332
+ * scores + weakest dimensions, asks an LLM to propose targeted rewrites of
333
+ * the current surface, and returns them as the next population.
334
+ *
335
+ * Maps onto the GEPA paper (Agrawal et al., arXiv:2507.19457):
336
+ * - *Reflection*: each generation reflects on the best parent's weakest
337
+ * dimensions + per-scenario top/bottom scores to propose targeted rewrites.
338
+ * - *Pareto frontier*: `runOptimization` maintains the non-dominated set of
339
+ * surfaces across generations (per-scenario objective vectors) and supplies
340
+ * it as `ctx.paretoParents`. A surface uniquely best on one hard scenario
341
+ * survives even when its mean composite is lower.
342
+ * - *Combine complementary lessons*: when the frontier has >1 member, the
343
+ * first population slot is a merge of those parents' strengths (one LLM
344
+ * call citing each parent's winning scenarios). Toggle via `combineParents`.
345
+ * Dominance is computed by the package-canonical `paretoFrontier` (`pareto.ts`).
346
+ *
347
+ * Optional `constraints` move structured-doc guards into the proposer
348
+ * (preserve H2 section headings, cap sentence-level edits) — useful when
349
+ * the surface IS a structured procedure like a SKILL.md / runbook /
350
+ * judge rubric. When `constraints` is omitted, behavior is unchanged.
351
+ *
352
+ * The proposer is surface-agnostic — any string surface in any consumer opts
353
+ * in by selecting it. Reuses the generic reflection primitive
354
+ * (`buildReflectionPrompt` / `parseReflectionResponse`) and the router client.
355
+ *
356
+ * Earns its keep where there is real per-instance signal (which the
357
+ * dimensional + per-scenario evidence + the `LabeledScenarioStore` flywheel
358
+ * now provide). For thin-signal surfaces it degrades to plain reflection.
359
+ * On generation 0 (no history) it reflects on the current surface against
360
+ * the mutation primitives alone.
361
+ */
362
+
363
+ interface GepaProposerConstraints {
364
+ /** H2 section headings that MUST appear unchanged in every candidate.
365
+ * When set, the proposer auto-detects current H2s if this is empty AND
366
+ * rejects any candidate that drops or renames a preserved heading.
367
+ * Use when the surface is a structured doc (SKILL.md, runbook,
368
+ * sectioned system prompt, judge rubric). */
369
+ preserveSections?: string[];
370
+ /** Maximum sentence-level edits per candidate vs the parent surface.
371
+ * Rejection threshold = maxSentenceEdits × 2 (counts adds + removes).
372
+ * Inspired by SkillOpt's edit-budget as a "textual learning rate."
373
+ * Cap prevents an LLM rewrite from overwriting useful prior rules. */
374
+ maxSentenceEdits?: number;
375
+ }
376
+ interface GepaProposerOptions {
377
+ /** Router transport (apiKey/baseUrl). */
378
+ llm: LlmClientOptions;
379
+ /** Model that performs the reflection. */
380
+ model: string;
381
+ /** What is being optimized — appears in the reflection prompt for orientation. */
382
+ target: string;
383
+ /** Surface-specific mutation levers offered to the model. */
384
+ mutationPrimitives?: string[];
385
+ /** Top/bottom scenarios surfaced as evidence each generation. Default 3. */
386
+ evidenceK?: number;
387
+ /** Reflection sampling temperature. Default 0.7. */
388
+ temperature?: number;
389
+ /** Reflection max tokens. Default 6000. */
390
+ maxTokens?: number;
391
+ /** Structured-doc constraints. Candidates violating any are rejected
392
+ * post-parse and dropped from the returned population. */
393
+ constraints?: GepaProposerConstraints;
394
+ /** GEPA combine-complementary-lessons: when the loop supplies a Pareto
395
+ * frontier of >1 non-dominated parents (`ctx.paretoParents`), spend one
396
+ * slot of the population on a merge of their strengths. Default `true` —
397
+ * this is the GEPA-faithful behavior; the merge only fires once the
398
+ * frontier has more than one member (generation ≥ 1). Set `false` for
399
+ * pure single-parent reflection. */
400
+ combineParents?: boolean;
401
+ /** Cap on how many frontier parents feed one combine prompt (highest
402
+ * composite first), to bound prompt size. Default 4. */
403
+ combineMaxParents?: number;
404
+ }
405
+ declare function gepaProposer(opts: GepaProposerOptions): SurfaceProposer;
406
+ /** Extract H2 headings (`## Foo`) from a markdown surface. Exported for
407
+ * consumers building custom mutators that share the same invariant. */
408
+ declare function extractH2Sections(text: string): string[];
409
+ /** Sentence-level edit distance — count distinct add/remove ops between
410
+ * two surfaces via a normalised line-by-line set diff. Treats trivial
411
+ * whitespace as identical. Exported for tests + consumer-side validators. */
412
+ declare function countSentenceEdits(baseline: string, candidate: string): number;
413
+
414
+ export { type CampaignStorage as C, type GepaProposerOptions as G, type OpenAutoPrOptions as O, type RunOptimizationOptions as R, type RunImprovementLoopResult as a, type RunCampaignOptions as b, type RunImprovementLoopOptions as c, runImprovementLoop as d, type GepaProposerConstraints as e, fsCampaignStorage as f, gepaProposer as g, type OpenAutoPrResult as h, inMemoryCampaignStorage as i, type RunOptimizationResult as j, countSentenceEdits as k, defaultRenderDiff as l, extractH2Sections as m, runOptimization as n, openAutoPr as o, runCampaign as r, surfaceHash as s };
@@ -1,9 +1,10 @@
1
- import { M as MutableSurface, c as GateDecision } from '../types-BU-7W85F.js';
2
- import { I as InsightReport } from '../insight-report-DWl3z9tl.js';
3
- import '../run-record-e7vj1uZQ.js';
1
+ import { M as MutableSurface, d as GateDecision } from '../types-DQRY8ZT-.js';
2
+ import { I as InsightReport } from '../insight-report-BnRjTibG.js';
3
+ import '../run-record-CP2ObebC.js';
4
+ import '@tangle-network/agent-interface';
4
5
  import '../errors-CzMUYo7b.js';
5
6
  import '../schema-m0gsnbt3.js';
6
- import '../summary-report-BDOFevaT.js';
7
+ import '../summary-report-CInXwsza.js';
7
8
  import '../failure-cluster-DH9Flgcf.js';
8
9
  import '../store-BcFXE6LG.js';
9
10
  import '../judge-calibration-0p2QcWNE.js';
@@ -1,4 +1,4 @@
1
- import { a as RunSplitTag } from './run-record-e7vj1uZQ.js';
1
+ import { a as RunSplitTag } from './run-record-CP2ObebC.js';
2
2
 
3
3
  /**
4
4
  * Shared types for the reference benchmark wrappers under