@tangle-network/agent-eval 0.94.0 → 0.95.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +32 -0
- package/README.md +44 -30
- package/dist/adapters/http.d.ts +8 -7
- package/dist/adapters/http.js.map +1 -1
- package/dist/adapters/langchain.d.ts +3 -2
- package/dist/adapters/otel.d.ts +5 -4
- package/dist/analyst/index.d.ts +11 -31
- package/dist/analyst/index.js +5 -65
- package/dist/analyst/index.js.map +1 -1
- package/dist/{analyze-runs-B6Ljo_dI.d.ts → analyze-runs-DtT6F_6T.d.ts} +3 -3
- package/dist/belief-state/index.d.ts +4 -3
- package/dist/benchmarks/index.d.ts +3 -2
- package/dist/campaign/index.d.ts +727 -616
- package/dist/campaign/index.js +1863 -1316
- package/dist/campaign/index.js.map +1 -1
- package/dist/{chunk-2K6UUZ7P.js → chunk-2T4EZACH.js} +1 -1
- package/dist/chunk-2T4EZACH.js.map +1 -0
- package/dist/{chunk-CTBHKLEU.js → chunk-77T4STFI.js} +59 -86
- package/dist/chunk-77T4STFI.js.map +1 -0
- package/dist/{chunk-EGPMSBEZ.js → chunk-7QTQKIDD.js} +178 -177
- package/dist/chunk-7QTQKIDD.js.map +1 -0
- package/dist/{chunk-MIFZUPEK.js → chunk-AQ5WQAIV.js} +21 -6
- package/dist/chunk-AQ5WQAIV.js.map +1 -0
- package/dist/chunk-DJWX3GVS.js +81 -0
- package/dist/chunk-DJWX3GVS.js.map +1 -0
- package/dist/{chunk-S6OZEZQK.js → chunk-HMA63UEO.js} +37 -9
- package/dist/{chunk-S6OZEZQK.js.map → chunk-HMA63UEO.js.map} +1 -1
- package/dist/{chunk-TBDR6PAI.js → chunk-IZCEK2HR.js} +2 -2
- package/dist/{chunk-SD2YFWQQ.js → chunk-KKWJD5E6.js} +20 -20
- package/dist/chunk-KKWJD5E6.js.map +1 -0
- package/dist/{chunk-KWRRMR3J.js → chunk-LO6IOIJ2.js} +10 -10
- package/dist/chunk-LO6IOIJ2.js.map +1 -0
- package/dist/{chunk-E4GH6USR.js → chunk-NZEQVRH5.js} +2 -2
- package/dist/chunk-NZEQVRH5.js.map +1 -0
- package/dist/{chunk-MPQWFX6Y.js → chunk-PSWWQXHF.js} +13 -88
- package/dist/chunk-PSWWQXHF.js.map +1 -0
- package/dist/{chunk-Q5LIB7BC.js → chunk-S4SYLDFX.js} +2 -2
- package/dist/chunk-S4SYLDFX.js.map +1 -0
- package/dist/{chunk-KW53MSA5.js → chunk-X74V6ESX.js} +2 -2
- package/dist/{chunk-QMUEXQJS.js → chunk-YBIGNSCZ.js} +81 -4
- package/dist/chunk-YBIGNSCZ.js.map +1 -0
- package/dist/{chunk-2KNZHH3P.js → chunk-Z6L6YSU6.js} +2 -2
- package/dist/{code-agent-session-BO8nCnv3.d.ts → code-agent-session-CPHRCb4-.d.ts} +1 -1
- package/dist/contract/index.d.ts +91 -43
- package/dist/contract/index.js +127 -17
- package/dist/contract/index.js.map +1 -1
- package/dist/{control-D6qwHXIR.d.ts → control-Doncu-B_.d.ts} +2 -2
- package/dist/control.d.ts +3 -2
- package/dist/control.js +2 -2
- package/dist/{corpus-B8A4BDR3.d.ts → corpus-D4YW9UoJ.d.ts} +1 -1
- package/dist/{default-registry-6dhErQbs.d.ts → default-registry-GyE8X5SP.d.ts} +3 -3
- package/dist/diagnose.d.ts +4 -3
- package/dist/diagnose.js +1 -1
- package/dist/{run-improvement-loop-DBahB8Ax.d.ts → gepa-C1NCIZ9o.d.ts} +117 -130
- package/dist/hosted/index.d.ts +5 -4
- package/dist/{index-Bx3gZ8xl.d.ts → index-_Y4oNOOb.d.ts} +1 -1
- package/dist/index.d.ts +76 -81
- package/dist/index.js +66 -31
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-DWl3z9tl.d.ts → insight-report-BnRjTibG.d.ts} +1 -1
- package/dist/{kind-factory-0BhLSI27.d.ts → kind-factory-X3eDYbKn.d.ts} +2 -3
- package/dist/matrix/index.d.ts +1 -1
- package/dist/meta-eval/index.d.ts +3 -2
- package/dist/multishot/index.d.ts +4 -4
- package/dist/multishot/index.js.map +1 -1
- package/dist/openapi.json +1 -1
- package/dist/{pre-registration-mAnCugl9.d.ts → pre-registration-nfUdc9EQ.d.ts} +2 -42
- package/dist/{provenance-P-bCL2Fo.d.ts → provenance-CncDq9qE.d.ts} +26 -41
- package/dist/{release-report-BEbWmVYj.d.ts → release-report-pidWUMZ2.d.ts} +2 -2
- package/dist/reporting.d.ts +5 -4
- package/dist/{researcher-B0C2_fVO.d.ts → researcher-Jr8ME1dZ.d.ts} +2 -2
- package/dist/rl.d.ts +516 -515
- package/dist/rl.js +612 -612
- package/dist/rl.js.map +1 -1
- package/dist/{rubric-predictive-validity-Cy_W-hWZ.d.ts → rubric-predictive-validity-C2hDKM8Z.d.ts} +1 -1
- package/dist/{run-campaign-7WNXMDSN.js → run-campaign-WXY7KI67.js} +2 -2
- package/dist/{run-record-e7vj1uZQ.d.ts → run-record-CP2ObebC.d.ts} +14 -18
- package/dist/{runtime-trajectory-BDgfGZSr.d.ts → runtime-trajectory-BOUUjI0y.d.ts} +1 -1
- package/dist/{semantic-concept-judge-B9MgmBnM.d.ts → semantic-concept-judge-DSBB2Cfp.d.ts} +2 -2
- package/dist/{summary-report-BDOFevaT.d.ts → summary-report-CInXwsza.d.ts} +1 -1
- package/dist/testing-C21CHsq2.d.ts +20 -0
- package/dist/testing.d.ts +1 -0
- package/dist/testing.js +8 -0
- package/dist/testing.js.map +1 -0
- package/dist/traces.d.ts +26 -10
- package/dist/traces.js +41 -11
- package/dist/{types-Ce17tDlG.d.ts → types-B5x54y6n.d.ts} +1 -1
- package/dist/{types-mn5Aqk7x.d.ts → types-BUxNaJ8c.d.ts} +2 -4
- package/dist/{types-BU-7W85F.d.ts → types-DQRY8ZT-.d.ts} +60 -58
- package/dist/workflow/index.d.ts +5 -4
- package/dist/workflow/index.js +1 -1
- package/docs/campaign-proposers.md +170 -0
- package/docs/concepts.md +8 -4
- package/docs/customer-journeys.md +15 -13
- package/docs/design/loop-taxonomy.md +34 -66
- package/docs/distributed-driver.md +14 -14
- package/docs/feature-guide.md +1 -1
- package/docs/hosted-ingest-spec.md +2 -3
- package/docs/multi-shot-optimization.md +8 -8
- package/docs/product-eval-adoption.md +1 -1
- package/docs/self-improvement-map.md +33 -29
- package/package.json +8 -14
- package/dist/chunk-2K6UUZ7P.js.map +0 -1
- package/dist/chunk-CTBHKLEU.js.map +0 -1
- package/dist/chunk-E4GH6USR.js.map +0 -1
- package/dist/chunk-EGPMSBEZ.js.map +0 -1
- package/dist/chunk-KWRRMR3J.js.map +0 -1
- package/dist/chunk-MIFZUPEK.js.map +0 -1
- package/dist/chunk-MPQWFX6Y.js.map +0 -1
- package/dist/chunk-Q5LIB7BC.js.map +0 -1
- package/dist/chunk-QMUEXQJS.js.map +0 -1
- package/dist/chunk-SD2YFWQQ.js.map +0 -1
- package/docs/design/external-agent-wedge.md +0 -89
- package/docs/design/phase-d-rfc.md +0 -125
- package/docs/design/phase4-consumer-migration.md +0 -70
- package/docs/design/primitives-integration-spec.md +0 -393
- package/docs/design/product-self-improvement-loop.md +0 -146
- package/docs/design/self-improvement-engine.md +0 -140
- package/docs/design/self-improvement-protocol.md +0 -223
- package/docs/design/self-improvement-roadmap.md +0 -106
- package/docs/design/substrate-gaps.md +0 -118
- package/docs/phase-b-pairing-kit.md +0 -188
- package/docs/phase-b-runbook.md +0 -176
- package/docs/pilot/README.md +0 -62
- package/docs/pilot/customer-checklist.md +0 -90
- package/docs/pilot/integration-foreign-stack.md +0 -296
- package/docs/pilot/integration-tangle-stack.md +0 -248
- package/docs/pilot/one-pager.md +0 -161
- package/docs/pilot/sample-insight-report.json +0 -172
- package/docs/quickstart-external.md +0 -229
- package/docs/research/belief-state-agent-eval-roadmap.md +0 -593
- package/docs/research/research-roadmap.md +0 -205
- package/docs/specs/driver-honest-spec.md +0 -251
- package/docs/specs/hermes-self-improvement-audit.md +0 -93
- package/docs/specs/profile-versioning.md +0 -291
- package/docs/three-package-architecture.md +0 -168
- /package/dist/{chunk-TBDR6PAI.js.map → chunk-IZCEK2HR.js.map} +0 -0
- /package/dist/{chunk-KW53MSA5.js.map → chunk-X74V6ESX.js.map} +0 -0
- /package/dist/{chunk-2KNZHH3P.js.map → chunk-Z6L6YSU6.js.map} +0 -0
- /package/dist/{run-campaign-7WNXMDSN.js.map → run-campaign-WXY7KI67.js.map} +0 -0
|
@@ -3,7 +3,7 @@ import { C as ControlEvalResult, a as ControlRunResult, h as ControlRuntimeConfi
|
|
|
3
3
|
import { T as TraceEmitter } from './emitter-C2rqGH_l.js';
|
|
4
4
|
import { F as FailureClass } from './schema-m0gsnbt3.js';
|
|
5
5
|
import { T as TraceStore } from './store-BcFXE6LG.js';
|
|
6
|
-
import { a as RunSplitTag, b as RunTokenUsage, R as RunRecord } from './run-record-
|
|
6
|
+
import { a as RunSplitTag, b as RunTokenUsage, R as RunRecord } from './run-record-CP2ObebC.js';
|
|
7
7
|
|
|
8
8
|
interface ActionExecutionPolicy {
|
|
9
9
|
allowedTypes?: string[];
|
|
@@ -52,7 +52,7 @@ interface ControlRunToRunRecordOptions extends RunEvidenceMetadata {
|
|
|
52
52
|
* release gates, optimizer tables, and research reports.
|
|
53
53
|
*
|
|
54
54
|
* The control loop owns live execution evidence. The caller still supplies the
|
|
55
|
-
*
|
|
55
|
+
* experiment-cell metadata because prompt/config hashes, split assignment,
|
|
56
56
|
* model snapshot, and commit SHA are product/harness concerns.
|
|
57
57
|
*/
|
|
58
58
|
declare function controlRunToRunRecord<TState, TAction, TActionResult, TEval extends ControlEvalResult = ControlEvalResult>(run: ControlRunResult<TState, TAction, TActionResult, TEval>, options: ControlRunToRunRecordOptions): RunRecord;
|
package/dist/control.d.ts
CHANGED
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
export { A as ActionExecutionPolicy, a as ActionPolicyDecision, C as ControlRunToRunRecordOptions, e as ProposeReviewConfig, f as ProposeReviewControlAction, g as ProposeReviewControlConfig, h as ProposeReviewControlResult, i as ProposeReviewControlState, j as ProposeReviewReport, p as RunEvidenceMetadata, s as controlRunToRunRecord, u as evaluateActionPolicy, x as runProposeReview, y as runProposeReviewAsControlLoop, z as scoreFromEvals } from './control-
|
|
1
|
+
export { A as ActionExecutionPolicy, a as ActionPolicyDecision, C as ControlRunToRunRecordOptions, e as ProposeReviewConfig, f as ProposeReviewControlAction, g as ProposeReviewControlConfig, h as ProposeReviewControlResult, i as ProposeReviewControlState, j as ProposeReviewReport, p as RunEvidenceMetadata, s as controlRunToRunRecord, u as evaluateActionPolicy, x as runProposeReview, y as runProposeReviewAsControlLoop, z as scoreFromEvals } from './control-Doncu-B_.js';
|
|
2
2
|
export { c as ControlActionFailureMode, d as ControlActionOutcome, e as ControlBudget, f as ControlContext, g as ControlDecision, C as ControlEvalResult, a as ControlRunResult, h as ControlRuntimeConfig, i as ControlRuntimeError, j as ControlSeverity, b as ControlStep, k as ControlStopPolicies, S as StopDecision, l as allCriticalPassed, o as objectiveEval, r as runAgentControlLoop, s as stopOnNoProgress, m as stopOnRepeatedAction, n as subjectiveEval } from './control-runtime-Acf9CGhw.js';
|
|
3
3
|
import './feedback-trajectory-BxY0cKfs.js';
|
|
4
4
|
import './dataset-BbGkaN2I.js';
|
|
@@ -6,4 +6,5 @@ import './errors-CzMUYo7b.js';
|
|
|
6
6
|
import './emitter-C2rqGH_l.js';
|
|
7
7
|
import './schema-m0gsnbt3.js';
|
|
8
8
|
import './store-BcFXE6LG.js';
|
|
9
|
-
import './run-record-
|
|
9
|
+
import './run-record-CP2ObebC.js';
|
|
10
|
+
import '@tangle-network/agent-interface';
|
package/dist/control.js
CHANGED
|
@@ -4,7 +4,7 @@ import {
|
|
|
4
4
|
runProposeReview,
|
|
5
5
|
runProposeReviewAsControlLoop,
|
|
6
6
|
scoreFromEvals
|
|
7
|
-
} from "./chunk-
|
|
7
|
+
} from "./chunk-NZEQVRH5.js";
|
|
8
8
|
import {
|
|
9
9
|
allCriticalPassed,
|
|
10
10
|
objectiveEval,
|
|
@@ -13,7 +13,7 @@ import {
|
|
|
13
13
|
stopOnRepeatedAction,
|
|
14
14
|
subjectiveEval
|
|
15
15
|
} from "./chunk-YEHAEDUD.js";
|
|
16
|
-
import "./chunk-
|
|
16
|
+
import "./chunk-LO6IOIJ2.js";
|
|
17
17
|
import "./chunk-TVVP3ZZQ.js";
|
|
18
18
|
import "./chunk-VSMTAMNK.js";
|
|
19
19
|
import "./chunk-3BFEG2F6.js";
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import { AxAIService } from '@ax-llm/ax';
|
|
2
|
-
import { T as TraceAnalystKindSpec } from './kind-factory-
|
|
3
|
-
import { A as Analyst, a as AnalystContext, b as AnalystRunSummary, c as AnalystFinding, d as AnalystRunResult, C as ChatClient, e as AnalystRunInputs, f as AnalystRunEvent } from './types-
|
|
2
|
+
import { T as TraceAnalystKindSpec } from './kind-factory-X3eDYbKn.js';
|
|
3
|
+
import { A as Analyst, a as AnalystContext, b as AnalystRunSummary, c as AnalystFinding, d as AnalystRunResult, C as ChatClient, e as AnalystRunInputs, f as AnalystRunEvent } from './types-B5x54y6n.js';
|
|
4
4
|
|
|
5
5
|
/**
|
|
6
6
|
* AnalystRegistry — orchestrate N analysts against one run.
|
|
@@ -152,4 +152,4 @@ interface DefaultAnalystRegistryOptions {
|
|
|
152
152
|
}
|
|
153
153
|
declare function buildDefaultAnalystRegistry(opts?: DefaultAnalystRegistryOptions): AnalystRegistry;
|
|
154
154
|
|
|
155
|
-
export { AnalystRegistry as A, type BudgetPolicy as B, type DefaultAnalystRegistryOptions as D, type RegistryRunOpts as R, type AnalystHooks as a,
|
|
155
|
+
export { AnalystRegistry as A, type BudgetPolicy as B, type DefaultAnalystRegistryOptions as D, type RegistryRunOpts as R, type AnalystHooks as a, type AnalystRegistryOptions as b, buildDefaultAnalystRegistry as c };
|
package/dist/diagnose.d.ts
CHANGED
|
@@ -3,9 +3,9 @@ export { c as CounterfactualContext, d as CounterfactualResult } from './counter
|
|
|
3
3
|
import { S as Span } from './schema-m0gsnbt3.js';
|
|
4
4
|
import { T as TraceStore } from './store-BcFXE6LG.js';
|
|
5
5
|
import { a as TrajectoryStep, T as Trajectory } from './trajectory-2TkpSEVh.js';
|
|
6
|
-
import { h as AnalystSeverity, c as AnalystFinding } from './types-
|
|
7
|
-
import { C as CorpusRecord } from './corpus-
|
|
8
|
-
import { R as RunRecord } from './run-record-
|
|
6
|
+
import { h as AnalystSeverity, c as AnalystFinding } from './types-B5x54y6n.js';
|
|
7
|
+
import { C as CorpusRecord } from './corpus-D4YW9UoJ.js';
|
|
8
|
+
import { R as RunRecord } from './run-record-CP2ObebC.js';
|
|
9
9
|
import './emitter-C2rqGH_l.js';
|
|
10
10
|
import './store-C1YxJDEK.js';
|
|
11
11
|
import './types-C7DGg5ex.js';
|
|
@@ -13,6 +13,7 @@ import '@tangle-network/tcloud';
|
|
|
13
13
|
import './llm-client-Bj7g0rqu.js';
|
|
14
14
|
import './errors-CzMUYo7b.js';
|
|
15
15
|
import './raw-provider-sink-C46HDghv.js';
|
|
16
|
+
import '@tangle-network/agent-interface';
|
|
16
17
|
|
|
17
18
|
/**
|
|
18
19
|
* Causal sweep — WHY did this run fail?
|
package/dist/diagnose.js
CHANGED
|
@@ -1,96 +1,7 @@
|
|
|
1
|
+
import { S as Scenario, C as CampaignResult, l as GateResult, D as DispatchFn, a as JudgeConfig, L as LabeledScenarioStore, i as CampaignTraceWriter, n as GenerationRecord, M as MutableSurface, P as ParetoParent, c as SurfaceProposer, G as Gate } from './types-DQRY8ZT-.js';
|
|
1
2
|
import { L as LlmClientOptions } from './llm-client-Bj7g0rqu.js';
|
|
2
|
-
import { I as ImprovementDriver, S as Scenario, g as CampaignResult, k as GateResult, D as DispatchFn, a as JudgeConfig, L as LabeledScenarioStore, h as CampaignTraceWriter, m as GenerationRecord, M as MutableSurface, P as ParetoParent, G as Gate } from './types-BU-7W85F.js';
|
|
3
3
|
|
|
4
4
|
/**
|
|
5
|
-
* @experimental
|
|
6
|
-
*
|
|
7
|
-
* `gepaDriver` — a reflective `ImprovementDriver` for prompt-tier surfaces.
|
|
8
|
-
* Each generation it reflects on the prior best candidate's per-scenario
|
|
9
|
-
* scores + weakest dimensions, asks an LLM to propose targeted rewrites of
|
|
10
|
-
* the current surface, and returns them as the next population.
|
|
11
|
-
*
|
|
12
|
-
* Maps onto the GEPA paper (Agrawal et al., arXiv:2507.19457):
|
|
13
|
-
* - *Reflection*: each generation reflects on the best parent's weakest
|
|
14
|
-
* dimensions + per-scenario top/bottom scores to propose targeted rewrites.
|
|
15
|
-
* - *Pareto frontier*: `runOptimization` maintains the non-dominated set of
|
|
16
|
-
* surfaces across generations (per-scenario objective vectors) and supplies
|
|
17
|
-
* it as `ctx.paretoParents`. A surface uniquely best on one hard scenario
|
|
18
|
-
* survives even when its mean composite is lower.
|
|
19
|
-
* - *Combine complementary lessons*: when the frontier has >1 member, the
|
|
20
|
-
* first population slot is a merge of those parents' strengths (one LLM
|
|
21
|
-
* call citing each parent's winning scenarios). Toggle via `combineParents`.
|
|
22
|
-
* Dominance is computed by the package-canonical `paretoFrontier` (`pareto.ts`).
|
|
23
|
-
*
|
|
24
|
-
* Optional `constraints` move structured-doc guards into the driver
|
|
25
|
-
* (preserve H2 section headings, cap sentence-level edits) — useful when
|
|
26
|
-
* the surface IS a structured procedure like a SKILL.md / runbook /
|
|
27
|
-
* judge rubric. When `constraints` is omitted, behavior is unchanged.
|
|
28
|
-
*
|
|
29
|
-
* The driver is surface-agnostic — any string surface in any consumer opts
|
|
30
|
-
* in by selecting it. Reuses the generic reflection primitive
|
|
31
|
-
* (`buildReflectionPrompt` / `parseReflectionResponse`) and the router client.
|
|
32
|
-
*
|
|
33
|
-
* Earns its keep where there is real per-instance signal (which the
|
|
34
|
-
* dimensional + per-scenario evidence + the `LabeledScenarioStore` flywheel
|
|
35
|
-
* now provide). For thin-signal surfaces it degrades to plain reflection.
|
|
36
|
-
* On generation 0 (no history) it reflects on the current surface against
|
|
37
|
-
* the mutation primitives alone.
|
|
38
|
-
*/
|
|
39
|
-
|
|
40
|
-
interface GepaDriverConstraints {
|
|
41
|
-
/** H2 section headings that MUST appear unchanged in every candidate.
|
|
42
|
-
* When set, the driver auto-detects current H2s if this is empty AND
|
|
43
|
-
* rejects any candidate that drops or renames a preserved heading.
|
|
44
|
-
* Use when the surface is a structured doc (SKILL.md, runbook,
|
|
45
|
-
* sectioned system prompt, judge rubric). */
|
|
46
|
-
preserveSections?: string[];
|
|
47
|
-
/** Maximum sentence-level edits per candidate vs the parent surface.
|
|
48
|
-
* Rejection threshold = maxSentenceEdits × 2 (counts adds + removes).
|
|
49
|
-
* Inspired by SkillOpt's edit-budget as a "textual learning rate."
|
|
50
|
-
* Cap prevents an LLM rewrite from overwriting useful prior rules. */
|
|
51
|
-
maxSentenceEdits?: number;
|
|
52
|
-
}
|
|
53
|
-
interface GepaDriverOptions {
|
|
54
|
-
/** Router transport (apiKey/baseUrl). */
|
|
55
|
-
llm: LlmClientOptions;
|
|
56
|
-
/** Model that performs the reflection. */
|
|
57
|
-
model: string;
|
|
58
|
-
/** What is being optimized — appears in the reflection prompt for orientation. */
|
|
59
|
-
target: string;
|
|
60
|
-
/** Surface-specific mutation levers offered to the model. */
|
|
61
|
-
mutationPrimitives?: string[];
|
|
62
|
-
/** Top/bottom scenarios surfaced as evidence each generation. Default 3. */
|
|
63
|
-
evidenceK?: number;
|
|
64
|
-
/** Reflection sampling temperature. Default 0.7. */
|
|
65
|
-
temperature?: number;
|
|
66
|
-
/** Reflection max tokens. Default 6000. */
|
|
67
|
-
maxTokens?: number;
|
|
68
|
-
/** Structured-doc constraints. Candidates violating any are rejected
|
|
69
|
-
* post-parse and dropped from the returned population. */
|
|
70
|
-
constraints?: GepaDriverConstraints;
|
|
71
|
-
/** GEPA combine-complementary-lessons: when the loop supplies a Pareto
|
|
72
|
-
* frontier of >1 non-dominated parents (`ctx.paretoParents`), spend one
|
|
73
|
-
* slot of the population on a merge of their strengths. Default `true` —
|
|
74
|
-
* this is the GEPA-faithful behavior; the merge only fires once the
|
|
75
|
-
* frontier has more than one member (generation ≥ 1). Set `false` for
|
|
76
|
-
* pure single-parent reflection. */
|
|
77
|
-
combineParents?: boolean;
|
|
78
|
-
/** Cap on how many frontier parents feed one combine prompt (highest
|
|
79
|
-
* composite first), to bound prompt size. Default 4. */
|
|
80
|
-
combineMaxParents?: number;
|
|
81
|
-
}
|
|
82
|
-
declare function gepaDriver(opts: GepaDriverOptions): ImprovementDriver;
|
|
83
|
-
/** Extract H2 headings (`## Foo`) from a markdown surface. Exported for
|
|
84
|
-
* consumers building custom mutators that share the same invariant. */
|
|
85
|
-
declare function extractH2Sections(text: string): string[];
|
|
86
|
-
/** Sentence-level edit distance — count distinct add/remove ops between
|
|
87
|
-
* two surfaces via a normalised line-by-line set diff. Treats trivial
|
|
88
|
-
* whitespace as identical. Exported for tests + consumer-side validators. */
|
|
89
|
-
declare function countSentenceEdits(baseline: string, candidate: string): number;
|
|
90
|
-
|
|
91
|
-
/**
|
|
92
|
-
* @experimental
|
|
93
|
-
*
|
|
94
5
|
* `openAutoPr` — thin shell-out helper for the `runImprovementLoop` preset's
|
|
95
6
|
* `autoOnPromote: 'pr'` mode. Substitutes for the per-product PR-opening
|
|
96
7
|
* code consumers duplicated 4 times. The PR body includes the campaign's
|
|
@@ -137,8 +48,6 @@ interface OpenAutoPrResult {
|
|
|
137
48
|
declare function openAutoPr<TArtifact, TScenario extends Scenario>(options: OpenAutoPrOptions<TArtifact, TScenario>): OpenAutoPrResult;
|
|
138
49
|
|
|
139
50
|
/**
|
|
140
|
-
* @experimental
|
|
141
|
-
*
|
|
142
51
|
* `CampaignStorage` — the filesystem seam `runCampaign` writes through
|
|
143
52
|
* (run/cell dirs, the resumability cache, per-cell artifacts, trace spans).
|
|
144
53
|
*
|
|
@@ -177,8 +86,6 @@ declare function fsCampaignStorage(): CampaignStorage;
|
|
|
177
86
|
declare function inMemoryCampaignStorage(): CampaignStorage;
|
|
178
87
|
|
|
179
88
|
/**
|
|
180
|
-
* @experimental
|
|
181
|
-
*
|
|
182
89
|
* `runCampaign` — Pass A substrate primitive. ONE function that orchestrates
|
|
183
90
|
* scenarios → dispatch → artifacts → judges → aggregates, with full
|
|
184
91
|
* reproducibility (seed + manifest hash), cell-level resumability, bootstrap
|
|
@@ -268,51 +175,47 @@ interface RunCampaignOptions<TScenario extends Scenario, TArtifact> {
|
|
|
268
175
|
declare function runCampaign<TScenario extends Scenario, TArtifact>(opts: RunCampaignOptions<TScenario, TArtifact>): Promise<CampaignResult<TArtifact, TScenario>>;
|
|
269
176
|
|
|
270
177
|
/**
|
|
271
|
-
* @experimental
|
|
272
|
-
*
|
|
273
178
|
* `runOptimization` — the improvement loop body. Runs N generations: the
|
|
274
|
-
* `
|
|
179
|
+
* `SurfaceProposer` proposes K candidate surfaces per generation, each
|
|
275
180
|
* candidate runs a campaign (the measurement), top-scoring promote to the
|
|
276
|
-
* next generation.
|
|
277
|
-
* population mutator (`
|
|
278
|
-
*
|
|
279
|
-
* how `propose()` picks candidates.
|
|
181
|
+
* next generation. Proposer-agnostic — the same loop runs an evolutionary
|
|
182
|
+
* population mutator (`evolutionaryProposer`) or any reflective / agentic
|
|
183
|
+
* proposer; they differ only in how `propose()` picks candidates.
|
|
280
184
|
*
|
|
281
185
|
* This is `runLoop`'s shape (plan → measure → decide) specialized to surface
|
|
282
|
-
* improvement: `
|
|
283
|
-
* runs the worker behind `dispatch`), the mean-composite ranking = the
|
|
284
|
-
* validator, `
|
|
186
|
+
* improvement: `proposer.propose` = plan, `runCampaign` = the measurement
|
|
187
|
+
* (which runs the worker behind `dispatch`), the mean-composite ranking = the
|
|
188
|
+
* validator, `proposer.decide` = the stop check.
|
|
285
189
|
*
|
|
286
190
|
* The gated-promotion shell (`runImprovementLoop`) wraps this with a holdout
|
|
287
191
|
* re-score + release gate + optional PR.
|
|
288
192
|
*/
|
|
289
193
|
|
|
290
|
-
interface
|
|
194
|
+
interface RunOptimizationBaseOptions<TScenario extends Scenario, TArtifact> extends Omit<RunCampaignOptions<TScenario, TArtifact>, 'dispatch'> {
|
|
291
195
|
/** Initial mutable surface (typically system prompt or addendum). */
|
|
292
196
|
baselineSurface: MutableSurface;
|
|
293
197
|
/** Dispatcher that takes the CURRENT surface + scenario → artifact. */
|
|
294
198
|
dispatchWithSurface: (surface: MutableSurface, scenario: TScenario, ctx: Parameters<RunCampaignOptions<TScenario, TArtifact>['dispatch']>[1]) => Promise<TArtifact>;
|
|
295
|
-
/** The
|
|
296
|
-
* `
|
|
297
|
-
*
|
|
298
|
-
|
|
199
|
+
/** The candidate-generation strategy. Wrap a population `Mutator` via
|
|
200
|
+
* `evolutionaryProposer({ mutator })`, or pass any reflective / agentic
|
|
201
|
+
* proposer that implements `SurfaceProposer`. */
|
|
202
|
+
proposer: SurfaceProposer;
|
|
299
203
|
populationSize: number;
|
|
300
204
|
maxGenerations: number;
|
|
301
205
|
/** How many top-scoring candidates carry to the next generation. Default 2. */
|
|
302
206
|
promoteTopK?: number;
|
|
303
|
-
/** DEPTH knob forwarded to the
|
|
207
|
+
/** DEPTH knob forwarded to the proposer's `propose()` — max iterations the
|
|
304
208
|
* agentic generator may take per candidate. */
|
|
305
209
|
maxImprovementShots?: number;
|
|
306
|
-
/**
|
|
307
|
-
*
|
|
210
|
+
/** Optional analysis report forwarded to `propose()`. Opaque here; the
|
|
211
|
+
* proposer types it. */
|
|
308
212
|
report?: unknown;
|
|
309
213
|
/** Structured findings forwarded to `propose()` as `ctx.findings`. A
|
|
310
|
-
* findings producer
|
|
311
|
-
* generation's traces; findings-grounded
|
|
312
|
-
*
|
|
313
|
-
* the driver types its `TFindings`. Empty when no producer is wired. */
|
|
214
|
+
* findings producer emits these from the
|
|
215
|
+
* generation's traces; findings-grounded proposers consume them. Opaque here;
|
|
216
|
+
* the proposer types its `TFindings`. Empty when no producer is wired. */
|
|
314
217
|
findings?: unknown[];
|
|
315
|
-
/** Per-generation findings producer
|
|
218
|
+
/** Per-generation findings producer. After each
|
|
316
219
|
* generation's candidates are scored, this is called with that generation's
|
|
317
220
|
* results; whatever it returns REPLACES `ctx.findings` for the NEXT
|
|
318
221
|
* generation's `propose()`, so the diagnosis is refreshed each round instead
|
|
@@ -331,6 +234,7 @@ interface RunOptimizationOptions<TScenario extends Scenario, TArtifact> extends
|
|
|
331
234
|
history: GenerationRecord[];
|
|
332
235
|
}) => Promise<unknown[]>;
|
|
333
236
|
}
|
|
237
|
+
type RunOptimizationOptions<TScenario extends Scenario, TArtifact> = RunOptimizationBaseOptions<TScenario, TArtifact>;
|
|
334
238
|
interface RunOptimizationResult<TArtifact, TScenario extends Scenario> {
|
|
335
239
|
generations: Array<{
|
|
336
240
|
record: GenerationRecord;
|
|
@@ -342,11 +246,11 @@ interface RunOptimizationResult<TArtifact, TScenario extends Scenario> {
|
|
|
342
246
|
}>;
|
|
343
247
|
winnerSurface: MutableSurface;
|
|
344
248
|
winnerSurfaceHash: string;
|
|
345
|
-
/**
|
|
346
|
-
* candidate came from a `ProposedCandidate` (a reflective
|
|
249
|
+
/** Proposer label for the promoted surface. Present when the winning
|
|
250
|
+
* candidate came from a `ProposedCandidate` (a reflective proposer);
|
|
347
251
|
* absent when the winner is the baseline or a bare-surface mutator. */
|
|
348
252
|
winnerLabel?: string;
|
|
349
|
-
/**
|
|
253
|
+
/** Proposer rationale for the promoted surface — the "because Z" that
|
|
350
254
|
* motivated the winning change. Survives to `SelfImproveResult` and the
|
|
351
255
|
* emitted provenance record. Absent when the winner is the baseline. */
|
|
352
256
|
winnerRationale?: string;
|
|
@@ -362,15 +266,13 @@ declare function runOptimization<TScenario extends Scenario, TArtifact>(opts: Ru
|
|
|
362
266
|
declare function surfaceHash(surface: MutableSurface): string;
|
|
363
267
|
|
|
364
268
|
/**
|
|
365
|
-
* @experimental
|
|
366
|
-
*
|
|
367
269
|
* `runImprovementLoop` — the gated-promotion shell around the improvement
|
|
368
|
-
* loop body (`runOptimization`).
|
|
369
|
-
* `
|
|
270
|
+
* loop body (`runOptimization`). Proposes candidate surfaces via the
|
|
271
|
+
* `SurfaceProposer`, re-scores the winner against the baseline on a
|
|
370
272
|
* holdout set, runs the release gate, and optionally opens a PR.
|
|
371
273
|
*
|
|
372
274
|
* Role vocabulary (see docs/design/loop-taxonomy.md):
|
|
373
|
-
* -
|
|
275
|
+
* - PROPOSER = the `SurfaceProposer` (evolutionary GEPA mutator OR
|
|
374
276
|
* reflective analyst). Proposes candidate SURFACES — the
|
|
375
277
|
* worker's system prompt / tool config — NOT conversation
|
|
376
278
|
* turns.
|
|
@@ -380,16 +282,16 @@ declare function surfaceHash(surface: MutableSurface): string;
|
|
|
380
282
|
* topology-opaque `dispatch` seam — never referenced here.
|
|
381
283
|
*
|
|
382
284
|
* Distinct from `runLoop` in `@tangle-network/agent-runtime`, which is the
|
|
383
|
-
* INNER conversation loop (driver↔workers in a sandbox). `runImprovementLoop`
|
|
285
|
+
* INNER conversation loop (execution driver ↔ workers in a sandbox). `runImprovementLoop`
|
|
384
286
|
* is the OUTER loop: it improves the surface that those workers run.
|
|
385
287
|
*
|
|
386
288
|
* Hard-refuses unsafe configurations:
|
|
387
|
-
* - `tracing: 'off'` when a
|
|
289
|
+
* - `tracing: 'off'` when a proposer is wired (improvement is unattributable)
|
|
388
290
|
* - `autoOnPromote: 'config'` — DEFERRED to Pass B; v0.40 only ships
|
|
389
291
|
* `'pr'` and `'none'`.
|
|
390
292
|
*/
|
|
391
293
|
|
|
392
|
-
|
|
294
|
+
type RunImprovementLoopOptions<TScenario extends Scenario, TArtifact> = RunOptimizationOptions<TScenario, TArtifact> & {
|
|
393
295
|
/** Holdout scenarios kept OUT of the training optimization pool — used
|
|
394
296
|
* ONLY to score baseline vs winner for the gate. */
|
|
395
297
|
holdoutScenarios: TScenario[];
|
|
@@ -409,7 +311,7 @@ interface RunImprovementLoopOptions<TScenario extends Scenario, TArtifact> exten
|
|
|
409
311
|
/** Optional render override — substrate writes a diff-shaped surface; pass
|
|
410
312
|
* a function to format the promoted surface differently. */
|
|
411
313
|
renderPromotedDiff?: (winnerSurface: MutableSurface, baselineSurface: MutableSurface) => string;
|
|
412
|
-
}
|
|
314
|
+
};
|
|
413
315
|
interface RunImprovementLoopResult<TArtifact, TScenario extends Scenario> extends RunOptimizationResult<TArtifact, TScenario> {
|
|
414
316
|
baselineOnHoldout: CampaignResult<TArtifact, TScenario>;
|
|
415
317
|
winnerOnHoldout: CampaignResult<TArtifact, TScenario>;
|
|
@@ -424,4 +326,89 @@ interface RunImprovementLoopResult<TArtifact, TScenario extends Scenario> extend
|
|
|
424
326
|
declare function runImprovementLoop<TScenario extends Scenario, TArtifact>(opts: RunImprovementLoopOptions<TScenario, TArtifact>): Promise<RunImprovementLoopResult<TArtifact, TScenario>>;
|
|
425
327
|
declare function defaultRenderDiff(winnerSurface: MutableSurface, baselineSurface: MutableSurface): string;
|
|
426
328
|
|
|
427
|
-
|
|
329
|
+
/**
|
|
330
|
+
* `gepaProposer` — a reflective `SurfaceProposer` for prompt-tier surfaces.
|
|
331
|
+
* Each generation it reflects on the prior best candidate's per-scenario
|
|
332
|
+
* scores + weakest dimensions, asks an LLM to propose targeted rewrites of
|
|
333
|
+
* the current surface, and returns them as the next population.
|
|
334
|
+
*
|
|
335
|
+
* Maps onto the GEPA paper (Agrawal et al., arXiv:2507.19457):
|
|
336
|
+
* - *Reflection*: each generation reflects on the best parent's weakest
|
|
337
|
+
* dimensions + per-scenario top/bottom scores to propose targeted rewrites.
|
|
338
|
+
* - *Pareto frontier*: `runOptimization` maintains the non-dominated set of
|
|
339
|
+
* surfaces across generations (per-scenario objective vectors) and supplies
|
|
340
|
+
* it as `ctx.paretoParents`. A surface uniquely best on one hard scenario
|
|
341
|
+
* survives even when its mean composite is lower.
|
|
342
|
+
* - *Combine complementary lessons*: when the frontier has >1 member, the
|
|
343
|
+
* first population slot is a merge of those parents' strengths (one LLM
|
|
344
|
+
* call citing each parent's winning scenarios). Toggle via `combineParents`.
|
|
345
|
+
* Dominance is computed by the package-canonical `paretoFrontier` (`pareto.ts`).
|
|
346
|
+
*
|
|
347
|
+
* Optional `constraints` move structured-doc guards into the proposer
|
|
348
|
+
* (preserve H2 section headings, cap sentence-level edits) — useful when
|
|
349
|
+
* the surface IS a structured procedure like a SKILL.md / runbook /
|
|
350
|
+
* judge rubric. When `constraints` is omitted, behavior is unchanged.
|
|
351
|
+
*
|
|
352
|
+
* The proposer is surface-agnostic — any string surface in any consumer opts
|
|
353
|
+
* in by selecting it. Reuses the generic reflection primitive
|
|
354
|
+
* (`buildReflectionPrompt` / `parseReflectionResponse`) and the router client.
|
|
355
|
+
*
|
|
356
|
+
* Earns its keep where there is real per-instance signal (which the
|
|
357
|
+
* dimensional + per-scenario evidence + the `LabeledScenarioStore` flywheel
|
|
358
|
+
* now provide). For thin-signal surfaces it degrades to plain reflection.
|
|
359
|
+
* On generation 0 (no history) it reflects on the current surface against
|
|
360
|
+
* the mutation primitives alone.
|
|
361
|
+
*/
|
|
362
|
+
|
|
363
|
+
interface GepaProposerConstraints {
|
|
364
|
+
/** H2 section headings that MUST appear unchanged in every candidate.
|
|
365
|
+
* When set, the proposer auto-detects current H2s if this is empty AND
|
|
366
|
+
* rejects any candidate that drops or renames a preserved heading.
|
|
367
|
+
* Use when the surface is a structured doc (SKILL.md, runbook,
|
|
368
|
+
* sectioned system prompt, judge rubric). */
|
|
369
|
+
preserveSections?: string[];
|
|
370
|
+
/** Maximum sentence-level edits per candidate vs the parent surface.
|
|
371
|
+
* Rejection threshold = maxSentenceEdits × 2 (counts adds + removes).
|
|
372
|
+
* Inspired by SkillOpt's edit-budget as a "textual learning rate."
|
|
373
|
+
* Cap prevents an LLM rewrite from overwriting useful prior rules. */
|
|
374
|
+
maxSentenceEdits?: number;
|
|
375
|
+
}
|
|
376
|
+
interface GepaProposerOptions {
|
|
377
|
+
/** Router transport (apiKey/baseUrl). */
|
|
378
|
+
llm: LlmClientOptions;
|
|
379
|
+
/** Model that performs the reflection. */
|
|
380
|
+
model: string;
|
|
381
|
+
/** What is being optimized — appears in the reflection prompt for orientation. */
|
|
382
|
+
target: string;
|
|
383
|
+
/** Surface-specific mutation levers offered to the model. */
|
|
384
|
+
mutationPrimitives?: string[];
|
|
385
|
+
/** Top/bottom scenarios surfaced as evidence each generation. Default 3. */
|
|
386
|
+
evidenceK?: number;
|
|
387
|
+
/** Reflection sampling temperature. Default 0.7. */
|
|
388
|
+
temperature?: number;
|
|
389
|
+
/** Reflection max tokens. Default 6000. */
|
|
390
|
+
maxTokens?: number;
|
|
391
|
+
/** Structured-doc constraints. Candidates violating any are rejected
|
|
392
|
+
* post-parse and dropped from the returned population. */
|
|
393
|
+
constraints?: GepaProposerConstraints;
|
|
394
|
+
/** GEPA combine-complementary-lessons: when the loop supplies a Pareto
|
|
395
|
+
* frontier of >1 non-dominated parents (`ctx.paretoParents`), spend one
|
|
396
|
+
* slot of the population on a merge of their strengths. Default `true` —
|
|
397
|
+
* this is the GEPA-faithful behavior; the merge only fires once the
|
|
398
|
+
* frontier has more than one member (generation ≥ 1). Set `false` for
|
|
399
|
+
* pure single-parent reflection. */
|
|
400
|
+
combineParents?: boolean;
|
|
401
|
+
/** Cap on how many frontier parents feed one combine prompt (highest
|
|
402
|
+
* composite first), to bound prompt size. Default 4. */
|
|
403
|
+
combineMaxParents?: number;
|
|
404
|
+
}
|
|
405
|
+
declare function gepaProposer(opts: GepaProposerOptions): SurfaceProposer;
|
|
406
|
+
/** Extract H2 headings (`## Foo`) from a markdown surface. Exported for
|
|
407
|
+
* consumers building custom mutators that share the same invariant. */
|
|
408
|
+
declare function extractH2Sections(text: string): string[];
|
|
409
|
+
/** Sentence-level edit distance — count distinct add/remove ops between
|
|
410
|
+
* two surfaces via a normalised line-by-line set diff. Treats trivial
|
|
411
|
+
* whitespace as identical. Exported for tests + consumer-side validators. */
|
|
412
|
+
declare function countSentenceEdits(baseline: string, candidate: string): number;
|
|
413
|
+
|
|
414
|
+
export { type CampaignStorage as C, type GepaProposerOptions as G, type OpenAutoPrOptions as O, type RunOptimizationOptions as R, type RunImprovementLoopResult as a, type RunCampaignOptions as b, type RunImprovementLoopOptions as c, runImprovementLoop as d, type GepaProposerConstraints as e, fsCampaignStorage as f, gepaProposer as g, type OpenAutoPrResult as h, inMemoryCampaignStorage as i, type RunOptimizationResult as j, countSentenceEdits as k, defaultRenderDiff as l, extractH2Sections as m, runOptimization as n, openAutoPr as o, runCampaign as r, surfaceHash as s };
|
package/dist/hosted/index.d.ts
CHANGED
|
@@ -1,9 +1,10 @@
|
|
|
1
|
-
import { M as MutableSurface,
|
|
2
|
-
import { I as InsightReport } from '../insight-report-
|
|
3
|
-
import '../run-record-
|
|
1
|
+
import { M as MutableSurface, d as GateDecision } from '../types-DQRY8ZT-.js';
|
|
2
|
+
import { I as InsightReport } from '../insight-report-BnRjTibG.js';
|
|
3
|
+
import '../run-record-CP2ObebC.js';
|
|
4
|
+
import '@tangle-network/agent-interface';
|
|
4
5
|
import '../errors-CzMUYo7b.js';
|
|
5
6
|
import '../schema-m0gsnbt3.js';
|
|
6
|
-
import '../summary-report-
|
|
7
|
+
import '../summary-report-CInXwsza.js';
|
|
7
8
|
import '../failure-cluster-DH9Flgcf.js';
|
|
8
9
|
import '../store-BcFXE6LG.js';
|
|
9
10
|
import '../judge-calibration-0p2QcWNE.js';
|