@tangle-network/agent-eval 0.94.0 → 0.95.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (140) hide show
  1. package/CHANGELOG.md +32 -0
  2. package/README.md +44 -30
  3. package/dist/adapters/http.d.ts +8 -7
  4. package/dist/adapters/http.js.map +1 -1
  5. package/dist/adapters/langchain.d.ts +3 -2
  6. package/dist/adapters/otel.d.ts +5 -4
  7. package/dist/analyst/index.d.ts +11 -31
  8. package/dist/analyst/index.js +5 -65
  9. package/dist/analyst/index.js.map +1 -1
  10. package/dist/{analyze-runs-B6Ljo_dI.d.ts → analyze-runs-DtT6F_6T.d.ts} +3 -3
  11. package/dist/belief-state/index.d.ts +4 -3
  12. package/dist/benchmarks/index.d.ts +3 -2
  13. package/dist/campaign/index.d.ts +727 -616
  14. package/dist/campaign/index.js +1863 -1316
  15. package/dist/campaign/index.js.map +1 -1
  16. package/dist/{chunk-2K6UUZ7P.js → chunk-2T4EZACH.js} +1 -1
  17. package/dist/chunk-2T4EZACH.js.map +1 -0
  18. package/dist/{chunk-CTBHKLEU.js → chunk-77T4STFI.js} +59 -86
  19. package/dist/chunk-77T4STFI.js.map +1 -0
  20. package/dist/{chunk-EGPMSBEZ.js → chunk-7QTQKIDD.js} +178 -177
  21. package/dist/chunk-7QTQKIDD.js.map +1 -0
  22. package/dist/{chunk-MIFZUPEK.js → chunk-AQ5WQAIV.js} +21 -6
  23. package/dist/chunk-AQ5WQAIV.js.map +1 -0
  24. package/dist/chunk-DJWX3GVS.js +81 -0
  25. package/dist/chunk-DJWX3GVS.js.map +1 -0
  26. package/dist/{chunk-S6OZEZQK.js → chunk-HMA63UEO.js} +37 -9
  27. package/dist/{chunk-S6OZEZQK.js.map → chunk-HMA63UEO.js.map} +1 -1
  28. package/dist/{chunk-TBDR6PAI.js → chunk-IZCEK2HR.js} +2 -2
  29. package/dist/{chunk-SD2YFWQQ.js → chunk-KKWJD5E6.js} +20 -20
  30. package/dist/chunk-KKWJD5E6.js.map +1 -0
  31. package/dist/{chunk-KWRRMR3J.js → chunk-LO6IOIJ2.js} +10 -10
  32. package/dist/chunk-LO6IOIJ2.js.map +1 -0
  33. package/dist/{chunk-E4GH6USR.js → chunk-NZEQVRH5.js} +2 -2
  34. package/dist/chunk-NZEQVRH5.js.map +1 -0
  35. package/dist/{chunk-MPQWFX6Y.js → chunk-PSWWQXHF.js} +13 -88
  36. package/dist/chunk-PSWWQXHF.js.map +1 -0
  37. package/dist/{chunk-Q5LIB7BC.js → chunk-S4SYLDFX.js} +2 -2
  38. package/dist/chunk-S4SYLDFX.js.map +1 -0
  39. package/dist/{chunk-KW53MSA5.js → chunk-X74V6ESX.js} +2 -2
  40. package/dist/{chunk-QMUEXQJS.js → chunk-YBIGNSCZ.js} +81 -4
  41. package/dist/chunk-YBIGNSCZ.js.map +1 -0
  42. package/dist/{chunk-2KNZHH3P.js → chunk-Z6L6YSU6.js} +2 -2
  43. package/dist/{code-agent-session-BO8nCnv3.d.ts → code-agent-session-CPHRCb4-.d.ts} +1 -1
  44. package/dist/contract/index.d.ts +91 -43
  45. package/dist/contract/index.js +127 -17
  46. package/dist/contract/index.js.map +1 -1
  47. package/dist/{control-D6qwHXIR.d.ts → control-Doncu-B_.d.ts} +2 -2
  48. package/dist/control.d.ts +3 -2
  49. package/dist/control.js +2 -2
  50. package/dist/{corpus-B8A4BDR3.d.ts → corpus-D4YW9UoJ.d.ts} +1 -1
  51. package/dist/{default-registry-6dhErQbs.d.ts → default-registry-GyE8X5SP.d.ts} +3 -3
  52. package/dist/diagnose.d.ts +4 -3
  53. package/dist/diagnose.js +1 -1
  54. package/dist/{run-improvement-loop-DBahB8Ax.d.ts → gepa-C1NCIZ9o.d.ts} +117 -130
  55. package/dist/hosted/index.d.ts +5 -4
  56. package/dist/{index-Bx3gZ8xl.d.ts → index-_Y4oNOOb.d.ts} +1 -1
  57. package/dist/index.d.ts +76 -81
  58. package/dist/index.js +66 -31
  59. package/dist/index.js.map +1 -1
  60. package/dist/{insight-report-DWl3z9tl.d.ts → insight-report-BnRjTibG.d.ts} +1 -1
  61. package/dist/{kind-factory-0BhLSI27.d.ts → kind-factory-X3eDYbKn.d.ts} +2 -3
  62. package/dist/matrix/index.d.ts +1 -1
  63. package/dist/meta-eval/index.d.ts +3 -2
  64. package/dist/multishot/index.d.ts +4 -4
  65. package/dist/multishot/index.js.map +1 -1
  66. package/dist/openapi.json +1 -1
  67. package/dist/{pre-registration-mAnCugl9.d.ts → pre-registration-nfUdc9EQ.d.ts} +2 -42
  68. package/dist/{provenance-P-bCL2Fo.d.ts → provenance-CncDq9qE.d.ts} +26 -41
  69. package/dist/{release-report-BEbWmVYj.d.ts → release-report-pidWUMZ2.d.ts} +2 -2
  70. package/dist/reporting.d.ts +5 -4
  71. package/dist/{researcher-B0C2_fVO.d.ts → researcher-Jr8ME1dZ.d.ts} +2 -2
  72. package/dist/rl.d.ts +516 -515
  73. package/dist/rl.js +612 -612
  74. package/dist/rl.js.map +1 -1
  75. package/dist/{rubric-predictive-validity-Cy_W-hWZ.d.ts → rubric-predictive-validity-C2hDKM8Z.d.ts} +1 -1
  76. package/dist/{run-campaign-7WNXMDSN.js → run-campaign-WXY7KI67.js} +2 -2
  77. package/dist/{run-record-e7vj1uZQ.d.ts → run-record-CP2ObebC.d.ts} +14 -18
  78. package/dist/{runtime-trajectory-BDgfGZSr.d.ts → runtime-trajectory-BOUUjI0y.d.ts} +1 -1
  79. package/dist/{semantic-concept-judge-B9MgmBnM.d.ts → semantic-concept-judge-DSBB2Cfp.d.ts} +2 -2
  80. package/dist/{summary-report-BDOFevaT.d.ts → summary-report-CInXwsza.d.ts} +1 -1
  81. package/dist/testing-C21CHsq2.d.ts +20 -0
  82. package/dist/testing.d.ts +1 -0
  83. package/dist/testing.js +8 -0
  84. package/dist/testing.js.map +1 -0
  85. package/dist/traces.d.ts +26 -10
  86. package/dist/traces.js +41 -11
  87. package/dist/{types-Ce17tDlG.d.ts → types-B5x54y6n.d.ts} +1 -1
  88. package/dist/{types-mn5Aqk7x.d.ts → types-BUxNaJ8c.d.ts} +2 -4
  89. package/dist/{types-BU-7W85F.d.ts → types-DQRY8ZT-.d.ts} +60 -58
  90. package/dist/workflow/index.d.ts +5 -4
  91. package/dist/workflow/index.js +1 -1
  92. package/docs/campaign-proposers.md +170 -0
  93. package/docs/concepts.md +8 -4
  94. package/docs/customer-journeys.md +15 -13
  95. package/docs/design/loop-taxonomy.md +34 -66
  96. package/docs/distributed-driver.md +14 -14
  97. package/docs/feature-guide.md +1 -1
  98. package/docs/hosted-ingest-spec.md +2 -3
  99. package/docs/multi-shot-optimization.md +8 -8
  100. package/docs/product-eval-adoption.md +1 -1
  101. package/docs/self-improvement-map.md +33 -29
  102. package/package.json +8 -14
  103. package/dist/chunk-2K6UUZ7P.js.map +0 -1
  104. package/dist/chunk-CTBHKLEU.js.map +0 -1
  105. package/dist/chunk-E4GH6USR.js.map +0 -1
  106. package/dist/chunk-EGPMSBEZ.js.map +0 -1
  107. package/dist/chunk-KWRRMR3J.js.map +0 -1
  108. package/dist/chunk-MIFZUPEK.js.map +0 -1
  109. package/dist/chunk-MPQWFX6Y.js.map +0 -1
  110. package/dist/chunk-Q5LIB7BC.js.map +0 -1
  111. package/dist/chunk-QMUEXQJS.js.map +0 -1
  112. package/dist/chunk-SD2YFWQQ.js.map +0 -1
  113. package/docs/design/external-agent-wedge.md +0 -89
  114. package/docs/design/phase-d-rfc.md +0 -125
  115. package/docs/design/phase4-consumer-migration.md +0 -70
  116. package/docs/design/primitives-integration-spec.md +0 -393
  117. package/docs/design/product-self-improvement-loop.md +0 -146
  118. package/docs/design/self-improvement-engine.md +0 -140
  119. package/docs/design/self-improvement-protocol.md +0 -223
  120. package/docs/design/self-improvement-roadmap.md +0 -106
  121. package/docs/design/substrate-gaps.md +0 -118
  122. package/docs/phase-b-pairing-kit.md +0 -188
  123. package/docs/phase-b-runbook.md +0 -176
  124. package/docs/pilot/README.md +0 -62
  125. package/docs/pilot/customer-checklist.md +0 -90
  126. package/docs/pilot/integration-foreign-stack.md +0 -296
  127. package/docs/pilot/integration-tangle-stack.md +0 -248
  128. package/docs/pilot/one-pager.md +0 -161
  129. package/docs/pilot/sample-insight-report.json +0 -172
  130. package/docs/quickstart-external.md +0 -229
  131. package/docs/research/belief-state-agent-eval-roadmap.md +0 -593
  132. package/docs/research/research-roadmap.md +0 -205
  133. package/docs/specs/driver-honest-spec.md +0 -251
  134. package/docs/specs/hermes-self-improvement-audit.md +0 -93
  135. package/docs/specs/profile-versioning.md +0 -291
  136. package/docs/three-package-architecture.md +0 -168
  137. /package/dist/{chunk-TBDR6PAI.js.map → chunk-IZCEK2HR.js.map} +0 -0
  138. /package/dist/{chunk-KW53MSA5.js.map → chunk-X74V6ESX.js.map} +0 -0
  139. /package/dist/{chunk-2KNZHH3P.js.map → chunk-Z6L6YSU6.js.map} +0 -0
  140. /package/dist/{run-campaign-7WNXMDSN.js.map → run-campaign-WXY7KI67.js.map} +0 -0
@@ -1,7 +1,7 @@
1
1
  import {
2
2
  runCampaign,
3
3
  summarizeBackendIntegrity
4
- } from "./chunk-2K6UUZ7P.js";
4
+ } from "./chunk-2T4EZACH.js";
5
5
  import {
6
6
  callLlm
7
7
  } from "./chunk-CWNP4DV4.js";
@@ -868,173 +868,6 @@ function parseReflectionResponse(raw, maxProposals) {
868
868
  return out;
869
869
  }
870
870
 
871
- // src/campaign/drivers/gepa.ts
872
- var REFLECTION_SYSTEM = 'You are an expert prompt engineer performing GEPA-style reflective mutation. You are given a prompt surface, its top trials (preserve what works) and its bottom trials (the evidence to fix). For each proposal, reason in this order before writing the payload: (1) LOCALIZE \u2014 point to the exact span of the current surface responsible for a bottom-trial failure; (2) DIAGNOSE the root cause (a missing rule, an ambiguous instruction, an over-broad directive), not just the symptom; (3) propose the MINIMAL, GENERALIZABLE edit that fixes the whole failure class \u2014 state it as a rule the agent should follow, never a patch memorized to the shown trials (that is overfitting and will not transfer to the held-out set); (4) PRESERVE every instruction the top trials depend on \u2014 do not delete or weaken working guidance. Put this localize\u2192diagnose\u2192fix reasoning in each proposal\'s `rationale`. Output ONLY a JSON object of shape {"proposals":[{"label":string,"rationale":string,"payload":string}]} where each `payload` is the FULL improved surface text. No prose outside the JSON.';
873
- var COMBINE_SYSTEM = 'You are an expert prompt engineer performing a GEPA "combine complementary lessons" merge. You are given several non-dominated versions of one surface; each is uniquely best on different scenarios. Produce ONE new version that keeps what makes each version strong on its winning scenarios and resolves conflicts in favor of the more general rule. Output ONLY a JSON object of shape {"proposals":[{"label":string,"rationale":string,"payload":string}]} with exactly one proposal whose `payload` is the FULL merged surface text. No prose outside the JSON.';
874
- function gepaDriver(opts) {
875
- const evidenceK = opts.evidenceK ?? 3;
876
- const combineParents = opts.combineParents ?? true;
877
- const combineMaxParents = opts.combineMaxParents ?? 4;
878
- if (combineParents && combineMaxParents < 1) {
879
- throw new Error("gepaDriver: combineMaxParents must be >= 1 when combineParents is enabled");
880
- }
881
- return {
882
- kind: "gepa",
883
- async propose(ctx) {
884
- const parent = typeof ctx.currentSurface === "string" ? ctx.currentSurface : JSON.stringify(ctx.currentSurface);
885
- const constraints = opts.constraints;
886
- const preserveSections = constraints?.preserveSections !== void 0 ? constraints.preserveSections.length === 0 ? extractH2Sections(parent) : constraints.preserveSections : null;
887
- const maxEdits = constraints?.maxSentenceEdits;
888
- const out = [];
889
- const seen = /* @__PURE__ */ new Set();
890
- const accept = (payload, label, rationale) => {
891
- const text = typeof payload === "string" ? payload.trim() : "";
892
- if (!text || text === parent || seen.has(text)) return;
893
- if (preserveSections && !validatePreservedSections(text, preserveSections)) return;
894
- if (maxEdits !== void 0 && countSentenceEdits(parent, text) > maxEdits * 2) return;
895
- seen.add(text);
896
- out.push({ surface: text, label, rationale });
897
- };
898
- const stringParents = (combineParents ? ctx.paretoParents ?? [] : []).filter((p) => typeof p.surface === "string").sort((a, b) => b.composite - a.composite).slice(0, combineMaxParents);
899
- if (stringParents.length > 1) {
900
- const combinePrompt = buildCombinePrompt({
901
- target: opts.target,
902
- parents: stringParents,
903
- evidenceK
904
- });
905
- const combineResult = await callLlm(
906
- {
907
- model: opts.model,
908
- messages: [
909
- { role: "system", content: COMBINE_SYSTEM },
910
- { role: "user", content: combinePrompt }
911
- ],
912
- jsonMode: true,
913
- temperature: opts.temperature ?? 0.7,
914
- maxTokens: opts.maxTokens ?? 6e3
915
- },
916
- opts.llm
917
- );
918
- const merged = parseReflectionResponse(combineResult.content, 1)[0];
919
- if (merged) {
920
- accept(
921
- merged.payload,
922
- merged.label || "pareto-combine",
923
- merged.rationale || `combined ${stringParents.length} non-dominated parents (gen ${stringParents.map((p) => p.generation).join(",")})`
924
- );
925
- }
926
- }
927
- const reflectCount = Math.max(0, ctx.populationSize - out.length);
928
- if (reflectCount > 0) {
929
- const { top, bottom, target } = buildEvidence(ctx, evidenceK, opts.target);
930
- const userPrompt = buildReflectionPrompt({
931
- target,
932
- parentPayload: parent,
933
- topTrials: top,
934
- bottomTrials: bottom,
935
- childCount: reflectCount,
936
- mutationPrimitives: opts.mutationPrimitives
937
- });
938
- const analyst = renderAnalystEvidence(ctx.findings, ctx.report);
939
- const finalPrompt = analyst ? `${userPrompt}
940
-
941
- ${analyst}` : userPrompt;
942
- const result = await callLlm(
943
- {
944
- model: opts.model,
945
- messages: [
946
- { role: "system", content: REFLECTION_SYSTEM },
947
- { role: "user", content: finalPrompt }
948
- ],
949
- jsonMode: true,
950
- temperature: opts.temperature ?? 0.7,
951
- maxTokens: opts.maxTokens ?? 6e3
952
- },
953
- opts.llm
954
- );
955
- for (const proposal of parseReflectionResponse(result.content, reflectCount)) {
956
- accept(proposal.payload, proposal.label, proposal.rationale);
957
- }
958
- }
959
- return out.slice(0, ctx.populationSize);
960
- }
961
- };
962
- }
963
- function buildCombinePrompt(args) {
964
- const lines = [
965
- `You are merging ${args.parents.length} versions of: ${args.target}.`,
966
- "",
967
- "Each version is on the Pareto frontier \u2014 none dominates the others; each",
968
- "wins on different scenarios. Combine their complementary strengths into",
969
- "ONE version. Below, each version lists the scenarios it scores highest on.",
970
- ""
971
- ];
972
- args.parents.forEach((p, i) => {
973
- const tag = String.fromCharCode(65 + i);
974
- const best = Object.entries(p.objectives).sort((a, b) => b[1] - a[1]).slice(0, args.evidenceK).map(([id, score]) => `${id} (${score.toFixed(2)})`);
975
- lines.push(
976
- `### Version ${tag} (mean ${p.composite.toFixed(2)}; strongest on: ${best.join(", ") || "n/a"})`,
977
- "```",
978
- p.surface,
979
- "```",
980
- ""
981
- );
982
- });
983
- lines.push(
984
- "Return ONE merged version that would score well on the union of every",
985
- "version's winning scenarios. Keep each version's specific winning rule;",
986
- "where two rules conflict, prefer the more general one and note the choice",
987
- "in your rationale."
988
- );
989
- return lines.join("\n");
990
- }
991
- function extractH2Sections(text) {
992
- const out = [];
993
- for (const line of text.split("\n")) {
994
- const match = /^##\s+(.+?)\s*$/.exec(line);
995
- if (match) out.push(match[1]);
996
- }
997
- return out;
998
- }
999
- function countSentenceEdits(baseline, candidate) {
1000
- const norm = (s) => s.split(/(?<=[.!?])\s+|\n/g).map((p) => p.trim()).filter((p) => p.length > 0);
1001
- const a = new Set(norm(baseline));
1002
- const b = new Set(norm(candidate));
1003
- let edits = 0;
1004
- for (const s of a) if (!b.has(s)) edits++;
1005
- for (const s of b) if (!a.has(s)) edits++;
1006
- return edits;
1007
- }
1008
- function validatePreservedSections(candidate, required) {
1009
- if (required.length === 0) return true;
1010
- const have = new Set(extractH2Sections(candidate));
1011
- for (const section of required) {
1012
- if (!have.has(section)) return false;
1013
- }
1014
- return true;
1015
- }
1016
- function buildEvidence(ctx, evidenceK, baseTarget) {
1017
- const last = ctx.history.at(-1);
1018
- if (!last || last.candidates.length === 0) {
1019
- return { top: [], bottom: [], target: baseTarget };
1020
- }
1021
- const best = [...last.candidates].sort((a, b) => b.composite - a.composite)[0];
1022
- if (!best) return { top: [], bottom: [], target: baseTarget };
1023
- const byScore = [...best.scenarios].sort((a, b) => b.composite - a.composite);
1024
- const toTrace = (s) => ({
1025
- id: s.scenarioId,
1026
- score: s.composite,
1027
- // The judge's "why it scored low" — grounds the reflection on real failure
1028
- // patterns instead of blind rephrasing. Generalizable by the judge contract.
1029
- ...s.notes ? { failureNote: s.notes } : {}
1030
- });
1031
- const top = byScore.slice(0, evidenceK).map(toTrace);
1032
- const bottom = byScore.slice(-evidenceK).reverse().map(toTrace);
1033
- const weakest = Object.entries(best.dimensions).sort((a, b) => a[1] - b[1]).slice(0, 3).map(([dim, value]) => `${dim} (${value.toFixed(2)})`);
1034
- const target = weakest.length > 0 ? `${baseTarget} \u2014 weakest dimensions: ${weakest.join(", ")}` : baseTarget;
1035
- return { top, bottom, target };
1036
- }
1037
-
1038
871
  // src/campaign/gates/heldout-gate.ts
1039
872
  function heldOutGate(options) {
1040
873
  const deltaThreshold = options.deltaThreshold ?? 0.5;
@@ -1251,6 +1084,7 @@ function labelTrustRank(trust) {
1251
1084
  // src/campaign/presets/run-optimization.ts
1252
1085
  import { createHash } from "crypto";
1253
1086
  async function runOptimization(opts) {
1087
+ const { proposer } = opts;
1254
1088
  const promoteTopK = opts.promoteTopK ?? 2;
1255
1089
  if (typeof opts.runDir !== "string" || opts.runDir.trim().length === 0) {
1256
1090
  throw new Error("runOptimization: runDir is required and must be a non-empty string");
@@ -1273,9 +1107,9 @@ async function runOptimization(opts) {
1273
1107
  toParetoParent(opts.baselineSurface, winnerSurfaceHash, baselineCampaign, -1)
1274
1108
  ];
1275
1109
  for (let gen = 0; gen < opts.maxGenerations; gen++) {
1276
- if (opts.driver.decide?.({ history }).stop) break;
1110
+ if (proposer.decide?.({ history }).stop) break;
1277
1111
  const paretoParents = computeParetoFrontier(scored);
1278
- const proposed = await opts.driver.propose({
1112
+ const proposed = await proposer.propose({
1279
1113
  currentSurface: currentSurfaces[0] ?? opts.baselineSurface,
1280
1114
  history,
1281
1115
  findings: currentFindings,
@@ -1423,9 +1257,9 @@ async function runImprovementLoop(opts) {
1423
1257
  "runImprovementLoop: autoOnPromote='config' is deferred to Pass B (requires shadow deploy + rollback + ensemble judges). Use 'pr' or 'none' in v0.40."
1424
1258
  );
1425
1259
  }
1426
- if (opts.tracing === "off" && opts.driver) {
1260
+ if (opts.tracing === "off" && opts.proposer) {
1427
1261
  throw new Error(
1428
- "runImprovementLoop: tracing='off' is forbidden when a driver is wired. The improvement loop without traces is unattributable; candidate surfaces cannot be cited back to spans and the optimization dataset goes unfed."
1262
+ "runImprovementLoop: tracing='off' is forbidden when a proposer is wired. The improvement loop without traces is unattributable; candidate surfaces cannot be cited back to spans and the optimization dataset goes unfed."
1429
1263
  );
1430
1264
  }
1431
1265
  if (opts.autoOnPromote === "pr" && (!opts.ghOwner || !opts.ghRepo)) {
@@ -1443,7 +1277,7 @@ async function runImprovementLoop(opts) {
1443
1277
  const dispatchTimeoutMs = opts.dispatchTimeoutMs ?? DEFAULT_DISPATCH_TIMEOUT_MS;
1444
1278
  const optimization = await runOptimization({ ...opts, dispatchTimeoutMs });
1445
1279
  const winnerIsBaseline = optimization.winnerSurfaceHash === surfaceHash(opts.baselineSurface);
1446
- const { runCampaign: runCampaign2 } = await import("./run-campaign-7WNXMDSN.js");
1280
+ const { runCampaign: runCampaign2 } = await import("./run-campaign-WXY7KI67.js");
1447
1281
  const baselineOnHoldout = await runCampaign2({
1448
1282
  ...opts,
1449
1283
  dispatchTimeoutMs,
@@ -1538,6 +1372,173 @@ ${fmt(winnerSurface)}`;
1538
1372
  return lines.join("\n");
1539
1373
  }
1540
1374
 
1375
+ // src/campaign/proposers/gepa.ts
1376
+ var REFLECTION_SYSTEM = 'You are an expert prompt engineer performing GEPA-style reflective mutation. You are given a prompt surface, its top trials (preserve what works) and its bottom trials (the evidence to fix). For each proposal, reason in this order before writing the payload: (1) LOCALIZE \u2014 point to the exact span of the current surface responsible for a bottom-trial failure; (2) DIAGNOSE the root cause (a missing rule, an ambiguous instruction, an over-broad directive), not just the symptom; (3) propose the MINIMAL, GENERALIZABLE edit that fixes the whole failure class \u2014 state it as a rule the agent should follow, never a patch memorized to the shown trials (that is overfitting and will not transfer to the held-out set); (4) PRESERVE every instruction the top trials depend on \u2014 do not delete or weaken working guidance. Put this localize\u2192diagnose\u2192fix reasoning in each proposal\'s `rationale`. Output ONLY a JSON object of shape {"proposals":[{"label":string,"rationale":string,"payload":string}]} where each `payload` is the FULL improved surface text. No prose outside the JSON.';
1377
+ var COMBINE_SYSTEM = 'You are an expert prompt engineer performing a GEPA "combine complementary lessons" merge. You are given several non-dominated versions of one surface; each is uniquely best on different scenarios. Produce ONE new version that keeps what makes each version strong on its winning scenarios and resolves conflicts in favor of the more general rule. Output ONLY a JSON object of shape {"proposals":[{"label":string,"rationale":string,"payload":string}]} with exactly one proposal whose `payload` is the FULL merged surface text. No prose outside the JSON.';
1378
+ function gepaProposer(opts) {
1379
+ const evidenceK = opts.evidenceK ?? 3;
1380
+ const combineParents = opts.combineParents ?? true;
1381
+ const combineMaxParents = opts.combineMaxParents ?? 4;
1382
+ if (combineParents && combineMaxParents < 1) {
1383
+ throw new Error("gepaProposer: combineMaxParents must be >= 1 when combineParents is enabled");
1384
+ }
1385
+ return {
1386
+ kind: "gepa",
1387
+ async propose(ctx) {
1388
+ const parent = typeof ctx.currentSurface === "string" ? ctx.currentSurface : JSON.stringify(ctx.currentSurface);
1389
+ const constraints = opts.constraints;
1390
+ const preserveSections = constraints?.preserveSections !== void 0 ? constraints.preserveSections.length === 0 ? extractH2Sections(parent) : constraints.preserveSections : null;
1391
+ const maxEdits = constraints?.maxSentenceEdits;
1392
+ const out = [];
1393
+ const seen = /* @__PURE__ */ new Set();
1394
+ const accept = (payload, label, rationale) => {
1395
+ const text = typeof payload === "string" ? payload.trim() : "";
1396
+ if (!text || text === parent || seen.has(text)) return;
1397
+ if (preserveSections && !validatePreservedSections(text, preserveSections)) return;
1398
+ if (maxEdits !== void 0 && countSentenceEdits(parent, text) > maxEdits * 2) return;
1399
+ seen.add(text);
1400
+ out.push({ surface: text, label, rationale });
1401
+ };
1402
+ const stringParents = (combineParents ? ctx.paretoParents ?? [] : []).filter((p) => typeof p.surface === "string").sort((a, b) => b.composite - a.composite).slice(0, combineMaxParents);
1403
+ if (stringParents.length > 1) {
1404
+ const combinePrompt = buildCombinePrompt({
1405
+ target: opts.target,
1406
+ parents: stringParents,
1407
+ evidenceK
1408
+ });
1409
+ const combineResult = await callLlm(
1410
+ {
1411
+ model: opts.model,
1412
+ messages: [
1413
+ { role: "system", content: COMBINE_SYSTEM },
1414
+ { role: "user", content: combinePrompt }
1415
+ ],
1416
+ jsonMode: true,
1417
+ temperature: opts.temperature ?? 0.7,
1418
+ maxTokens: opts.maxTokens ?? 6e3
1419
+ },
1420
+ opts.llm
1421
+ );
1422
+ const merged = parseReflectionResponse(combineResult.content, 1)[0];
1423
+ if (merged) {
1424
+ accept(
1425
+ merged.payload,
1426
+ merged.label || "pareto-combine",
1427
+ merged.rationale || `combined ${stringParents.length} non-dominated parents (gen ${stringParents.map((p) => p.generation).join(",")})`
1428
+ );
1429
+ }
1430
+ }
1431
+ const reflectCount = Math.max(0, ctx.populationSize - out.length);
1432
+ if (reflectCount > 0) {
1433
+ const { top, bottom, target } = buildEvidence(ctx, evidenceK, opts.target);
1434
+ const userPrompt = buildReflectionPrompt({
1435
+ target,
1436
+ parentPayload: parent,
1437
+ topTrials: top,
1438
+ bottomTrials: bottom,
1439
+ childCount: reflectCount,
1440
+ mutationPrimitives: opts.mutationPrimitives
1441
+ });
1442
+ const analyst = renderAnalystEvidence(ctx.findings, ctx.report);
1443
+ const finalPrompt = analyst ? `${userPrompt}
1444
+
1445
+ ${analyst}` : userPrompt;
1446
+ const result = await callLlm(
1447
+ {
1448
+ model: opts.model,
1449
+ messages: [
1450
+ { role: "system", content: REFLECTION_SYSTEM },
1451
+ { role: "user", content: finalPrompt }
1452
+ ],
1453
+ jsonMode: true,
1454
+ temperature: opts.temperature ?? 0.7,
1455
+ maxTokens: opts.maxTokens ?? 6e3
1456
+ },
1457
+ opts.llm
1458
+ );
1459
+ for (const proposal of parseReflectionResponse(result.content, reflectCount)) {
1460
+ accept(proposal.payload, proposal.label, proposal.rationale);
1461
+ }
1462
+ }
1463
+ return out.slice(0, ctx.populationSize);
1464
+ }
1465
+ };
1466
+ }
1467
+ function buildCombinePrompt(args) {
1468
+ const lines = [
1469
+ `You are merging ${args.parents.length} versions of: ${args.target}.`,
1470
+ "",
1471
+ "Each version is on the Pareto frontier \u2014 none dominates the others; each",
1472
+ "wins on different scenarios. Combine their complementary strengths into",
1473
+ "ONE version. Below, each version lists the scenarios it scores highest on.",
1474
+ ""
1475
+ ];
1476
+ args.parents.forEach((p, i) => {
1477
+ const tag = String.fromCharCode(65 + i);
1478
+ const best = Object.entries(p.objectives).sort((a, b) => b[1] - a[1]).slice(0, args.evidenceK).map(([id, score]) => `${id} (${score.toFixed(2)})`);
1479
+ lines.push(
1480
+ `### Version ${tag} (mean ${p.composite.toFixed(2)}; strongest on: ${best.join(", ") || "n/a"})`,
1481
+ "```",
1482
+ p.surface,
1483
+ "```",
1484
+ ""
1485
+ );
1486
+ });
1487
+ lines.push(
1488
+ "Return ONE merged version that would score well on the union of every",
1489
+ "version's winning scenarios. Keep each version's specific winning rule;",
1490
+ "where two rules conflict, prefer the more general one and note the choice",
1491
+ "in your rationale."
1492
+ );
1493
+ return lines.join("\n");
1494
+ }
1495
+ function extractH2Sections(text) {
1496
+ const out = [];
1497
+ for (const line of text.split("\n")) {
1498
+ const match = /^##\s+(.+?)\s*$/.exec(line);
1499
+ if (match) out.push(match[1]);
1500
+ }
1501
+ return out;
1502
+ }
1503
+ function countSentenceEdits(baseline, candidate) {
1504
+ const norm = (s) => s.split(/(?<=[.!?])\s+|\n/g).map((p) => p.trim()).filter((p) => p.length > 0);
1505
+ const a = new Set(norm(baseline));
1506
+ const b = new Set(norm(candidate));
1507
+ let edits = 0;
1508
+ for (const s of a) if (!b.has(s)) edits++;
1509
+ for (const s of b) if (!a.has(s)) edits++;
1510
+ return edits;
1511
+ }
1512
+ function validatePreservedSections(candidate, required) {
1513
+ if (required.length === 0) return true;
1514
+ const have = new Set(extractH2Sections(candidate));
1515
+ for (const section of required) {
1516
+ if (!have.has(section)) return false;
1517
+ }
1518
+ return true;
1519
+ }
1520
+ function buildEvidence(ctx, evidenceK, baseTarget) {
1521
+ const last = ctx.history.at(-1);
1522
+ if (!last || last.candidates.length === 0) {
1523
+ return { top: [], bottom: [], target: baseTarget };
1524
+ }
1525
+ const best = [...last.candidates].sort((a, b) => b.composite - a.composite)[0];
1526
+ if (!best) return { top: [], bottom: [], target: baseTarget };
1527
+ const byScore = [...best.scenarios].sort((a, b) => b.composite - a.composite);
1528
+ const toTrace = (s) => ({
1529
+ id: s.scenarioId,
1530
+ score: s.composite,
1531
+ // The judge's "why it scored low" — grounds the reflection on real failure
1532
+ // patterns instead of blind rephrasing. Generalizable by the judge contract.
1533
+ ...s.notes ? { failureNote: s.notes } : {}
1534
+ });
1535
+ const top = byScore.slice(0, evidenceK).map(toTrace);
1536
+ const bottom = byScore.slice(-evidenceK).reverse().map(toTrace);
1537
+ const weakest = Object.entries(best.dimensions).sort((a, b) => a[1] - b[1]).slice(0, 3).map(([dim, value]) => `${dim} (${value.toFixed(2)})`);
1538
+ const target = weakest.length > 0 ? `${baseTarget} \u2014 weakest dimensions: ${weakest.join(", ")}` : baseTarget;
1539
+ return { top, bottom, target };
1540
+ }
1541
+
1541
1542
  // src/campaign/provenance.ts
1542
1543
  import { createHash as createHash2 } from "crypto";
1543
1544
  import { join as join2 } from "path";
@@ -1824,9 +1825,6 @@ export {
1824
1825
  buildReflectionPrompt,
1825
1826
  renderAnalystEvidence,
1826
1827
  parseReflectionResponse,
1827
- gepaDriver,
1828
- extractH2Sections,
1829
- countSentenceEdits,
1830
1828
  heldOutGate,
1831
1829
  openAutoPr,
1832
1830
  campaignMeanComposite,
@@ -1837,6 +1835,9 @@ export {
1837
1835
  surfaceHash,
1838
1836
  runImprovementLoop,
1839
1837
  defaultRenderDiff,
1838
+ gepaProposer,
1839
+ extractH2Sections,
1840
+ countSentenceEdits,
1840
1841
  surfaceContentHash,
1841
1842
  buildLoopProvenanceRecord,
1842
1843
  loopProvenanceSpans,
@@ -1844,4 +1845,4 @@ export {
1844
1845
  provenanceSpansPath,
1845
1846
  emitLoopProvenance
1846
1847
  };
1847
- //# sourceMappingURL=chunk-EGPMSBEZ.js.map
1848
+ //# sourceMappingURL=chunk-7QTQKIDD.js.map