@tangle-network/agent-eval 0.109.1 → 0.110.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +9 -0
- package/dist/analyst/index.d.ts +10 -12
- package/dist/analyst/index.js +8 -11
- package/dist/analyst/index.js.map +1 -1
- package/dist/{analyze-runs-DJYpep3L.d.ts → analyze-runs-Dmz6LA9e.d.ts} +4 -4
- package/dist/{baseline-Bbid3WoO.d.ts → baseline-DsNteOgR.d.ts} +32 -2
- package/dist/belief-state/index.d.ts +6 -6
- package/dist/benchmarks/index.d.ts +4 -4
- package/dist/benchmarks/index.js +7 -8
- package/dist/builder-eval/index.d.ts +4 -4
- package/dist/builder-eval/index.js +1 -2
- package/dist/builder-eval/index.js.map +1 -1
- package/dist/{calibration-BPmzuVPk.d.ts → calibration-Dz8TQV4y.d.ts} +2 -2
- package/dist/campaign/index.d.ts +62 -20
- package/dist/campaign/index.js +9 -8
- package/dist/{chunk-LOJ2QVCE.js → chunk-2IY4ILP4.js} +2 -2
- package/dist/{chunk-LIEJUH2I.js → chunk-6PL5MGDL.js} +9 -9
- package/dist/{chunk-2OGPXHOB.js → chunk-7NX6ZSBG.js} +36 -7
- package/dist/chunk-7NX6ZSBG.js.map +1 -0
- package/dist/{chunk-R6D7NEYJ.js → chunk-GBI5J5DB.js} +81 -11
- package/dist/chunk-GBI5J5DB.js.map +1 -0
- package/dist/{chunk-YEHAEDUD.js → chunk-IMWDSFUM.js} +604 -2
- package/dist/chunk-IMWDSFUM.js.map +1 -0
- package/dist/{chunk-OVPVM4JC.js → chunk-J4AKLZEV.js} +15 -4
- package/dist/{chunk-OVPVM4JC.js.map → chunk-J4AKLZEV.js.map} +1 -1
- package/dist/{chunk-JZXGWLK5.js → chunk-MHNQWM4I.js} +62 -6
- package/dist/chunk-MHNQWM4I.js.map +1 -0
- package/dist/{chunk-QRVS7MX4.js → chunk-OW47B5WA.js} +3 -5
- package/dist/{chunk-QRVS7MX4.js.map → chunk-OW47B5WA.js.map} +1 -1
- package/dist/{chunk-DBDRR6GF.js → chunk-PLOMR3HP.js} +48 -2
- package/dist/chunk-PLOMR3HP.js.map +1 -0
- package/dist/{chunk-GDZAWO2I.js → chunk-QFGTU7MT.js} +2 -2
- package/dist/{chunk-V7HNA47Z.js → chunk-RSVSSZKF.js} +5 -5
- package/dist/{chunk-5PK3626Q.js → chunk-XRGOKCMO.js} +88 -17
- package/dist/chunk-XRGOKCMO.js.map +1 -0
- package/dist/{code-agent-session-rnJKlqmT.d.ts → code-agent-session-yitf9I-F.d.ts} +1 -1
- package/dist/contract/index.d.ts +26 -28
- package/dist/contract/index.js +11 -13
- package/dist/contract/index.js.map +1 -1
- package/dist/{control-B8UthSBL.d.ts → control-U8LBKUES.d.ts} +5 -6
- package/dist/control.d.ts +8 -9
- package/dist/control.js +6 -8
- package/dist/{dataset-DS7ytHZU.d.ts → dataset-NENEzRgk.d.ts} +1 -1
- package/dist/{default-registry-BswHCXnU.d.ts → default-registry-Bcf1uKVI.d.ts} +1 -2
- package/dist/{emitter-C2rqGH_l.d.ts → emitter-BRchAAAx.d.ts} +2 -2
- package/dist/{failure-cluster-DH9Flgcf.d.ts → failure-cluster-C48PiReX.d.ts} +2 -2
- package/dist/feedback-trajectory-pDcz1lQ1.d.ts +348 -0
- package/dist/{gepa-B3x5Ulcv.d.ts → gepa-T8T215nw.d.ts} +149 -6
- package/dist/hosted/index.d.ts +7 -7
- package/dist/{index-pPtfoIJO.d.ts → index-Dc3VLGhp.d.ts} +2 -2
- package/dist/index.d.ts +645 -61
- package/dist/index.js +1282 -190
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-B4xrdwEK.d.ts → insight-report-D4cXFsLt.d.ts} +1 -1
- package/dist/{integrity-DqGZg3st.d.ts → integrity-qemeBAyx.d.ts} +1 -1
- package/dist/{types-D1ytG0Yg.d.ts → kind-factory-20hcaYpf.d.ts} +169 -2
- package/dist/meta-eval/index.d.ts +5 -5
- package/dist/meta-eval/index.js +1 -2
- package/dist/meta-eval/index.js.map +1 -1
- package/dist/{multi-layer-verifier-CI4jdX-q.d.ts → multi-layer-verifier-BsqKuLyN.d.ts} +1 -1
- package/dist/multishot/index.d.ts +3 -3
- package/dist/openapi.json +1 -1
- package/dist/pipelines/index.d.ts +6 -7
- package/dist/pipelines/index.js +3 -6
- package/dist/pipelines/index.js.map +1 -1
- package/dist/{policy-edit-DQUXYMDm.d.ts → policy-edit-D2bBDZDf.d.ts} +2 -2
- package/dist/{pre-registration-BUhVPzE7.d.ts → pre-registration-BepVVa6P.d.ts} +3 -3
- package/dist/{provenance-DdDhf6cg.d.ts → provenance-CyxkvEi9.d.ts} +3 -5
- package/dist/{query-0aTmbmQe.d.ts → query-Ck190MOd.d.ts} +2 -2
- package/dist/{release-report-DeJpsBiA.d.ts → release-report-oBfOz8ku.d.ts} +3 -3
- package/dist/reporting.d.ts +8 -8
- package/dist/{researcher-Wc7dx6GM.d.ts → researcher-CaH0CwFC.d.ts} +6 -6
- package/dist/rl.d.ts +568 -15
- package/dist/rl.js +4 -4
- package/dist/{rubric-predictive-validity-DPnyG-CE.d.ts → rubric-predictive-validity-C-fMteAW.d.ts} +1 -1
- package/dist/{run-record-I-Z3JNvO.d.ts → run-record-DksGsfgv.d.ts} +1 -1
- package/dist/{runtime-trajectory-iW9IhV3e.d.ts → runtime-trajectory-h5i0SZUj.d.ts} +1 -1
- package/dist/{schema-m0gsnbt3.d.ts → schema-SGWcK9wa.d.ts} +1 -1
- package/dist/{semantic-concept-judge-BmNZPB_j.d.ts → semantic-concept-judge-D7z6JCLZ.d.ts} +57 -4
- package/dist/{store-BcFXE6LG.d.ts → store-BsVi7ncX.d.ts} +1 -1
- package/dist/storyboard/index.d.ts +1 -1
- package/dist/{summary-report-QMZVe3P-.d.ts → summary-report-Bz-0-t8v.d.ts} +2 -2
- package/dist/{test-graded-scenario-DeODGLra.d.ts → test-graded-scenario-mzYBKspu.d.ts} +3 -3
- package/dist/traces.d.ts +54 -11
- package/dist/traces.js +25 -27
- package/dist/{types-BdIv5dvA.d.ts → types-v--ctu-b.d.ts} +2 -2
- package/dist/wire/index.d.ts +5 -6
- package/docs/improvement-glossary.md +14 -13
- package/package.json +1 -71
- package/dist/adapters/http.d.ts +0 -142
- package/dist/adapters/http.js +0 -203
- package/dist/adapters/http.js.map +0 -1
- package/dist/adapters/langchain.d.ts +0 -95
- package/dist/adapters/langchain.js +0 -34
- package/dist/adapters/langchain.js.map +0 -1
- package/dist/adapters/otel.d.ts +0 -112
- package/dist/adapters/otel.js +0 -110
- package/dist/adapters/otel.js.map +0 -1
- package/dist/chunk-2OGPXHOB.js.map +0 -1
- package/dist/chunk-45EEMHTC.js +0 -35
- package/dist/chunk-45EEMHTC.js.map +0 -1
- package/dist/chunk-5BKGXME7.js +0 -65
- package/dist/chunk-5BKGXME7.js.map +0 -1
- package/dist/chunk-5PK3626Q.js.map +0 -1
- package/dist/chunk-6SK5VFYK.js +0 -100
- package/dist/chunk-6SK5VFYK.js.map +0 -1
- package/dist/chunk-DBDRR6GF.js.map +0 -1
- package/dist/chunk-DJWX3GVS.js +0 -81
- package/dist/chunk-DJWX3GVS.js.map +0 -1
- package/dist/chunk-FOUG2VVS.js +0 -855
- package/dist/chunk-FOUG2VVS.js.map +0 -1
- package/dist/chunk-JZXGWLK5.js.map +0 -1
- package/dist/chunk-K7QEIHHJ.js +0 -613
- package/dist/chunk-K7QEIHHJ.js.map +0 -1
- package/dist/chunk-KKHDIONI.js +0 -414
- package/dist/chunk-KKHDIONI.js.map +0 -1
- package/dist/chunk-KMPRBJK4.js +0 -74
- package/dist/chunk-KMPRBJK4.js.map +0 -1
- package/dist/chunk-Q2JRAWRI.js +0 -196
- package/dist/chunk-Q2JRAWRI.js.map +0 -1
- package/dist/chunk-R6D7NEYJ.js.map +0 -1
- package/dist/chunk-RZTMDUO7.js +0 -49
- package/dist/chunk-RZTMDUO7.js.map +0 -1
- package/dist/chunk-STGVSCDH.js +0 -202
- package/dist/chunk-STGVSCDH.js.map +0 -1
- package/dist/chunk-YEHAEDUD.js.map +0 -1
- package/dist/control-runtime-Acf9CGhw.d.ts +0 -182
- package/dist/corpus-eBVwhCp1.d.ts +0 -560
- package/dist/counterfactual-DlOz8PBx.d.ts +0 -85
- package/dist/diagnose.d.ts +0 -252
- package/dist/diagnose.js +0 -382
- package/dist/diagnose.js.map +0 -1
- package/dist/feedback-trajectory-C9KCo8ag.d.ts +0 -169
- package/dist/governance/index.d.ts +0 -135
- package/dist/governance/index.js +0 -18
- package/dist/governance/index.js.map +0 -1
- package/dist/groundedness/index.d.ts +0 -112
- package/dist/groundedness/index.js +0 -77
- package/dist/groundedness/index.js.map +0 -1
- package/dist/harness-optimizer-mOl9XX_O.d.ts +0 -106
- package/dist/kind-factory-DvIGo_cP.d.ts +0 -171
- package/dist/knowledge/index.d.ts +0 -103
- package/dist/knowledge/index.js +0 -18
- package/dist/knowledge/index.js.map +0 -1
- package/dist/pareto-E-pembql.d.ts +0 -81
- package/dist/perf/index.d.ts +0 -123
- package/dist/perf/index.js +0 -18
- package/dist/perf/index.js.map +0 -1
- package/dist/prm/index.d.ts +0 -104
- package/dist/prm/index.js +0 -265
- package/dist/prm/index.js.map +0 -1
- package/dist/product-benchmark/index.d.ts +0 -247
- package/dist/product-benchmark/index.js +0 -37
- package/dist/product-benchmark/index.js.map +0 -1
- package/dist/red-team-KmmiqBlY.d.ts +0 -63
- package/dist/redact-B40YG2M_.d.ts +0 -45
- package/dist/rubric-Cc6UHvUb.d.ts +0 -73
- package/dist/run-critic-CmMf05uV.d.ts +0 -56
- package/dist/sink-fetch-B1Yg4Til.d.ts +0 -101
- package/dist/telemetry/file.d.ts +0 -19
- package/dist/telemetry/file.js +0 -45
- package/dist/telemetry/file.js.map +0 -1
- package/dist/telemetry/index.d.ts +0 -38
- package/dist/telemetry/index.js +0 -130
- package/dist/telemetry/index.js.map +0 -1
- package/dist/testing-C21CHsq2.d.ts +0 -20
- package/dist/testing.d.ts +0 -1
- package/dist/testing.js +0 -8
- package/dist/testing.js.map +0 -1
- package/dist/trajectory-2TkpSEVh.d.ts +0 -33
- package/dist/workflow/index.d.ts +0 -496
- package/dist/workflow/index.js +0 -2178
- package/dist/workflow/index.js.map +0 -1
- /package/dist/{chunk-LOJ2QVCE.js.map → chunk-2IY4ILP4.js.map} +0 -0
- /package/dist/{chunk-LIEJUH2I.js.map → chunk-6PL5MGDL.js.map} +0 -0
- /package/dist/{chunk-GDZAWO2I.js.map → chunk-QFGTU7MT.js.map} +0 -0
- /package/dist/{chunk-V7HNA47Z.js.map → chunk-RSVSSZKF.js.map} +0 -0
|
@@ -196,4 +196,4 @@ declare function isRetrievalSpan(s: Span): s is RetrievalSpan;
|
|
|
196
196
|
declare function isJudgeSpan(s: Span): s is JudgeSpan;
|
|
197
197
|
declare function isSandboxSpan(s: Span): s is SandboxSpan;
|
|
198
198
|
|
|
199
|
-
export { type Artifact as A, type BudgetLedgerEntry as B, type EventKind as E, type FailureClass as F, type GenericSpan as G, type JudgeSpan as J, type LlmSpan as L, type Message as M, type Run as R, type Span as S, type ToolSpan as T, type
|
|
199
|
+
export { type Artifact as A, type BudgetLedgerEntry as B, type EventKind as E, type FailureClass as F, type GenericSpan as G, type JudgeSpan as J, type LlmSpan as L, type Message as M, type Run as R, type Span as S, type ToolSpan as T, type RunOutcome as a, type SpanKind as b, type RetrievalSpan as c, type SandboxSpan as d, type TraceEvent as e, type RunStatus as f, type RunLayer as g, type BudgetSpec as h, FAILURE_CLASSES as i, type SpanBase as j, type SpanStatus as k, TRACE_SCHEMA_VERSION as l, isJudgeSpan as m, isLlmSpan as n, isRetrievalSpan as o, isSandboxSpan as p, isToolSpan as q };
|
|
@@ -1,10 +1,11 @@
|
|
|
1
1
|
import { AxAIService } from '@ax-llm/ax';
|
|
2
2
|
import { z } from 'zod';
|
|
3
|
-
import { A as AnalystFinding, a as Analyst, b as AnalystContext } from './
|
|
4
|
-
import { T as TraceAnalystKindSpec } from './kind-factory-DvIGo_cP.js';
|
|
3
|
+
import { A as AnalystFinding, T as TraceAnalystKindSpec, a as Analyst, b as AnalystContext } from './kind-factory-20hcaYpf.js';
|
|
5
4
|
import { a as TraceAnalystSpan } from './store-C1YxJDEK.js';
|
|
5
|
+
import { R as Run, S as Span, e as TraceEvent, A as Artifact, B as BudgetLedgerEntry } from './schema-SGWcK9wa.js';
|
|
6
|
+
import { T as TraceStore } from './store-BsVi7ncX.js';
|
|
6
7
|
import { L as LlmClientOptions } from './llm-client-DyqEH4jH.js';
|
|
7
|
-
import { S as Severity } from './multi-layer-verifier-
|
|
8
|
+
import { S as Severity } from './multi-layer-verifier-BsqKuLyN.js';
|
|
8
9
|
|
|
9
10
|
interface CreateAnalystAiConfig {
|
|
10
11
|
/** OpenAI-compatible API key forwarded as `Authorization: Bearer`.
|
|
@@ -487,6 +488,58 @@ interface BehavioralMetrics {
|
|
|
487
488
|
*/
|
|
488
489
|
declare function computeTraceMetrics(spans: readonly TraceAnalystSpan[]): BehavioralMetrics;
|
|
489
490
|
|
|
491
|
+
interface RunScore {
|
|
492
|
+
success: number;
|
|
493
|
+
goalProgress: number;
|
|
494
|
+
repoGroundedness: number;
|
|
495
|
+
driftPenalty: number;
|
|
496
|
+
toolUseQuality: number;
|
|
497
|
+
patchQuality: number;
|
|
498
|
+
testReality: number;
|
|
499
|
+
finalGate: number;
|
|
500
|
+
reviewerBlockers: number;
|
|
501
|
+
costUsd: number;
|
|
502
|
+
wallSeconds: number;
|
|
503
|
+
notes?: string[];
|
|
504
|
+
}
|
|
505
|
+
interface RunScoreWeights {
|
|
506
|
+
success: number;
|
|
507
|
+
goalProgress: number;
|
|
508
|
+
repoGroundedness: number;
|
|
509
|
+
driftPenalty: number;
|
|
510
|
+
toolUseQuality: number;
|
|
511
|
+
patchQuality: number;
|
|
512
|
+
testReality: number;
|
|
513
|
+
finalGate: number;
|
|
514
|
+
reviewerBlockers: number;
|
|
515
|
+
costUsd: number;
|
|
516
|
+
wallSeconds: number;
|
|
517
|
+
}
|
|
518
|
+
declare const DEFAULT_RUN_SCORE_WEIGHTS: RunScoreWeights;
|
|
519
|
+
declare function aggregateRunScore(score: RunScore, weights?: Partial<RunScoreWeights>): number;
|
|
520
|
+
declare function clamp01(value: number): number;
|
|
521
|
+
|
|
522
|
+
interface RunTrace {
|
|
523
|
+
run: Run;
|
|
524
|
+
spans: Span[];
|
|
525
|
+
events: TraceEvent[];
|
|
526
|
+
artifacts: Artifact[];
|
|
527
|
+
budget: BudgetLedgerEntry[];
|
|
528
|
+
}
|
|
529
|
+
interface RunCriticOptions {
|
|
530
|
+
weights?: Partial<RunScoreWeights>;
|
|
531
|
+
driftPatterns?: RegExp[];
|
|
532
|
+
}
|
|
533
|
+
declare class RunCritic {
|
|
534
|
+
private readonly weights?;
|
|
535
|
+
private readonly driftPatterns;
|
|
536
|
+
constructor(options?: RunCriticOptions);
|
|
537
|
+
score(store: TraceStore, runId: string): Promise<RunScore>;
|
|
538
|
+
scoreTrace(trace: RunTrace): RunScore;
|
|
539
|
+
rank(score: RunScore): number;
|
|
540
|
+
private isDrift;
|
|
541
|
+
}
|
|
542
|
+
|
|
490
543
|
/**
|
|
491
544
|
* Semantic concept judge — "does the built artifact actually implement
|
|
492
545
|
* the features the user asked for?"
|
|
@@ -621,4 +674,4 @@ declare function runSemanticConceptJudge(input: SemanticConceptJudgeInput, optio
|
|
|
621
674
|
*/
|
|
622
675
|
declare function createSemanticConceptJudge(options?: SemanticConceptJudgeOptions): (input: SemanticConceptJudgeInput) => Promise<SemanticConceptJudgeResult>;
|
|
623
676
|
|
|
624
|
-
export { type
|
|
677
|
+
export { type ConceptComplexity as A, type BehavioralMetrics as B, type CreateAnalystAiConfig as C, DEFAULT_TRACE_ANALYST_KINDS as D, type ConceptFinding as E, FAILURE_MODE_KIND_SPEC as F, type ConceptSpec as G, type ConceptWeightStrategy as H, IMPROVEMENT_KIND_SPEC as I, DEFAULT_COMPLEXITY_WEIGHTS as J, KIND_EXPECTED_SUBJECTS as K, DEFAULT_RUN_SCORE_WEIGHTS as L, type RunCriticOptions as M, SEMANTIC_CONCEPT_JUDGE_VERSION as N, type SemanticConceptJudgeResult as O, type PersistedFinding as P, type SuboptimalCode as Q, RunCritic as R, type SemanticConceptJudgeOptions as S, type SuboptimalSignal as T, aggregateRunScore as U, clamp01 as V, computeTraceMetrics as W, createSemanticConceptJudge as X, runSemanticConceptJudge as Y, type RunTrace as a, type SemanticConceptJudgeInput as b, type DiffPolicy as c, FINDING_SUBJECT_GRAMMAR_PROMPT as d, FINDING_SUBJECT_KINDS as e, type FindingSubject as f, type FindingSubjectKind as g, FindingSubjectStringSchema as h, type FindingsDiff as i, FindingsStore as j, KNOWLEDGE_GAP_KIND_SPEC as k, KNOWLEDGE_POISONING_KIND_SPEC as l, SKILL_USAGE_ANALYST as m, SkillUsageAnalyst as n, type SkillUsageRecord as o, type SkillUsageReport as p, type SkillUsageScanConfig as q, buildSkillUsageReport as r, createAnalystAi as s, defaultIsMaterial as t, diffFindings as u, emitSkillUsageFindings as v, parseFindingSubject as w, renderFindingSubject as x, type RunScore as y, type RunScoreWeights as z };
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { R as Run, S as Span,
|
|
1
|
+
import { R as Run, S as Span, e as TraceEvent, A as Artifact, B as BudgetLedgerEntry, f as RunStatus, g as RunLayer, b as SpanKind, E as EventKind } from './schema-SGWcK9wa.js';
|
|
2
2
|
|
|
3
3
|
interface RunFilter {
|
|
4
4
|
scenarioId?: string;
|
|
@@ -1,5 +1,5 @@
|
|
|
1
|
-
import { R as RunRecord } from './run-record-
|
|
2
|
-
import { F as FailureClusterReport } from './failure-cluster-
|
|
1
|
+
import { R as RunRecord } from './run-record-DksGsfgv.js';
|
|
2
|
+
import { F as FailureClusterReport } from './failure-cluster-C48PiReX.js';
|
|
3
3
|
|
|
4
4
|
/**
|
|
5
5
|
* HeldOutGate — first-class held-out paired-delta promotion gate.
|
|
@@ -1,6 +1,6 @@
|
|
|
1
|
-
import { T as TraceEmitter } from './emitter-
|
|
2
|
-
import { R as Run, F as FailureClass } from './schema-
|
|
3
|
-
import { T as TraceStore } from './store-
|
|
1
|
+
import { T as TraceEmitter } from './emitter-BRchAAAx.js';
|
|
2
|
+
import { R as Run, F as FailureClass } from './schema-SGWcK9wa.js';
|
|
3
|
+
import { T as TraceStore } from './store-BsVi7ncX.js';
|
|
4
4
|
|
|
5
5
|
/**
|
|
6
6
|
* SandboxHarness — executes a scenario in an isolated environment and
|
package/dist/traces.d.ts
CHANGED
|
@@ -1,20 +1,19 @@
|
|
|
1
1
|
import { N as NotFoundError, R as ReplayError } from './errors-oeQrLqXC.js';
|
|
2
2
|
import { P as ProviderRedactor, R as RawProviderSink, d as RawProviderEvent } from './raw-provider-sink-C46HDghv.js';
|
|
3
3
|
export { F as FileSystemRawProviderSink, a as FileSystemRawProviderSinkOptions, I as InMemoryRawProviderSink, b as InMemoryRawProviderSinkOptions, N as NoopRawProviderSink, c as RawProviderDirection, e as RawProviderSinkFilter, f as defaultProviderRedactor, p as providerFromBaseUrl } from './raw-provider-sink-C46HDghv.js';
|
|
4
|
-
import { a as RunCompleteHookContext, R as RunCompleteHook } from './emitter-
|
|
5
|
-
export { S as SpanHandle, T as TraceEmitter, b as TraceEmitterOptions, l as llmSpanFromProvider } from './emitter-
|
|
6
|
-
export { b as RunIntegrityError, R as RunIntegrityExpectations, c as RunIntegrityIssue, d as RunIntegrityIssueCode, a as RunIntegrityReport, e as assertRunCaptured, t as throwIfRunIncomplete } from './integrity-
|
|
7
|
-
import { T as TraceStore } from './store-
|
|
8
|
-
export { E as EventFilter, F as FileSystemTraceStore, a as FileSystemTraceStoreOptions, I as InMemoryTraceStore, R as RunFilter, S as SpanFilter } from './store-
|
|
9
|
-
export { a as aggregateLlm, b as argHash, g as groupBy, j as judgeSpans, l as llmSpans, r as runFailureClass, c as runsForScenario, t as toolSpans } from './query-
|
|
10
|
-
|
|
11
|
-
|
|
12
|
-
export { A as Artifact, B as BudgetLedgerEntry, h as BudgetSpec, E as EventKind, i as FAILURE_CLASSES, F as FailureClass, G as GenericSpan, J as JudgeSpan, L as LlmSpan, M as Message, d as RetrievalSpan, g as RunLayer, b as RunOutcome, f as RunStatus, e as SandboxSpan, S as Span, j as SpanBase, c as SpanKind, k as SpanStatus, l as TRACE_SCHEMA_VERSION, T as ToolSpan, a as TraceEvent, m as isJudgeSpan, n as isLlmSpan, o as isRetrievalSpan, p as isSandboxSpan, q as isToolSpan } from './schema-m0gsnbt3.js';
|
|
4
|
+
import { a as RunCompleteHookContext, R as RunCompleteHook } from './emitter-BRchAAAx.js';
|
|
5
|
+
export { S as SpanHandle, T as TraceEmitter, b as TraceEmitterOptions, l as llmSpanFromProvider } from './emitter-BRchAAAx.js';
|
|
6
|
+
export { b as RunIntegrityError, R as RunIntegrityExpectations, c as RunIntegrityIssue, d as RunIntegrityIssueCode, a as RunIntegrityReport, e as assertRunCaptured, t as throwIfRunIncomplete } from './integrity-qemeBAyx.js';
|
|
7
|
+
import { T as TraceStore } from './store-BsVi7ncX.js';
|
|
8
|
+
export { E as EventFilter, F as FileSystemTraceStore, a as FileSystemTraceStoreOptions, I as InMemoryTraceStore, R as RunFilter, S as SpanFilter } from './store-BsVi7ncX.js';
|
|
9
|
+
export { a as aggregateLlm, b as argHash, g as groupBy, j as judgeSpans, l as llmSpans, r as runFailureClass, c as runsForScenario, t as toolSpans } from './query-Ck190MOd.js';
|
|
10
|
+
import { R as Run } from './schema-SGWcK9wa.js';
|
|
11
|
+
export { A as Artifact, B as BudgetLedgerEntry, h as BudgetSpec, E as EventKind, i as FAILURE_CLASSES, F as FailureClass, G as GenericSpan, J as JudgeSpan, L as LlmSpan, M as Message, c as RetrievalSpan, g as RunLayer, a as RunOutcome, f as RunStatus, d as SandboxSpan, S as Span, j as SpanBase, b as SpanKind, k as SpanStatus, l as TRACE_SCHEMA_VERSION, T as ToolSpan, e as TraceEvent, m as isJudgeSpan, n as isLlmSpan, o as isRetrievalSpan, p as isSandboxSpan, q as isToolSpan } from './schema-SGWcK9wa.js';
|
|
13
12
|
import { A as AnalyzeTracesOptions, b as AnalyzeTracesResult } from './analyst-C8HHvfJp.js';
|
|
14
13
|
export { a as AnalyzeTracesInput, c as AnalyzeTracesTurnSnapshot, d as analyzeTraces } from './analyst-C8HHvfJp.js';
|
|
15
14
|
import { h as TraceAnalystSpanKind, i as TraceAnalystSpanStatus, T as TraceAnalysisStore, g as TraceAnalystFilters, b as DatasetOverview, Q as QueryTracesPage, l as ViewTraceResult, V as ViewSpansResult, c as SearchTraceResult, S as SearchSpanResult } from './store-C1YxJDEK.js';
|
|
16
15
|
export { D as DEFAULT_TRACE_ANALYST_BUDGETS, E as ErrorCluster, d as SpanMatchRecord, e as TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, f as TraceAnalystByteBudgets, a as TraceAnalystSpan, j as TraceAnalystTraceSummary, k as ViewTraceOversized } from './store-C1YxJDEK.js';
|
|
17
|
-
import { b as RunSplitTag, c as RunTokenUsage, R as RunRecord } from './run-record-
|
|
16
|
+
import { b as RunSplitTag, c as RunTokenUsage, R as RunRecord } from './run-record-DksGsfgv.js';
|
|
18
17
|
import { AxFunction } from '@ax-llm/ax';
|
|
19
18
|
import '@tangle-network/agent-interface';
|
|
20
19
|
|
|
@@ -782,6 +781,50 @@ declare function otelRunCompleteHook(exporter: OtelExporter): RunCompleteHook;
|
|
|
782
781
|
*/
|
|
783
782
|
declare function createOtelTracingStore(inner: TraceStore, exporter: OtelExporter, traceId: string): TraceStore;
|
|
784
783
|
|
|
784
|
+
/**
|
|
785
|
+
* Redaction — remove PII / secrets from trace payloads before persist.
|
|
786
|
+
*
|
|
787
|
+
* Pre-persistence rules mean raw traces in storage are already scrubbed.
|
|
788
|
+
* Unredacted variants (for debugging / post-mortems) live in a separate
|
|
789
|
+
* storage layer with stricter access controls; this module only covers
|
|
790
|
+
* the default scrub-then-persist path.
|
|
791
|
+
*
|
|
792
|
+
* Rules compose: pass an array of `RedactionRule`, each is applied in
|
|
793
|
+
* order. Strings that match get replaced with a tagged sentinel so the
|
|
794
|
+
* eval framework can count how many redactions happened per run
|
|
795
|
+
* (surfaced via `redaction_applied` events).
|
|
796
|
+
*/
|
|
797
|
+
interface RedactionRule {
|
|
798
|
+
id: string;
|
|
799
|
+
pattern: RegExp;
|
|
800
|
+
/** Replacement — e.g. '[PII:email]'. Defaults to `[redacted:{id}]`. */
|
|
801
|
+
replacement?: string;
|
|
802
|
+
}
|
|
803
|
+
interface RedactionReport {
|
|
804
|
+
redactionCount: number;
|
|
805
|
+
byRule: Record<string, number>;
|
|
806
|
+
}
|
|
807
|
+
/** OWASP / common-sense defaults — extend per-domain. */
|
|
808
|
+
declare const DEFAULT_REDACTION_RULES: RedactionRule[];
|
|
809
|
+
declare const REDACTION_VERSION = "1.0.0";
|
|
810
|
+
/**
|
|
811
|
+
* Redact a single string. Returns the new string and a per-rule count of
|
|
812
|
+
* how many substitutions fired.
|
|
813
|
+
*/
|
|
814
|
+
declare function redactString(input: string, rules?: RedactionRule[]): {
|
|
815
|
+
output: string;
|
|
816
|
+
report: RedactionReport;
|
|
817
|
+
};
|
|
818
|
+
/**
|
|
819
|
+
* Walk a JSON-ish value applying `redactString` to every string leaf.
|
|
820
|
+
* Arrays and plain objects are recursed; other types pass through
|
|
821
|
+
* untouched. Circular references throw — traces should be tree-shaped.
|
|
822
|
+
*/
|
|
823
|
+
declare function redactValue(value: unknown, rules?: RedactionRule[], report?: RedactionReport): {
|
|
824
|
+
value: unknown;
|
|
825
|
+
report: RedactionReport;
|
|
826
|
+
};
|
|
827
|
+
|
|
785
828
|
/**
|
|
786
829
|
* Convert agent-eval's internal trace shape (`FileSystemTraceStore` → `Run`,
|
|
787
830
|
* `Span`, `TraceEvent`) into the OTLP-flat JSONL the trace analyst
|
|
@@ -973,4 +1016,4 @@ declare function iterateRawCalls(sink: RawProviderSink, filter?: {
|
|
|
973
1016
|
spanId?: string;
|
|
974
1017
|
}): AsyncGenerator<ReplayCacheEntry>;
|
|
975
1018
|
|
|
976
|
-
export { AnalyzeTracesOptions, AnalyzeTracesResult, type CaptureFetchContext, type CaptureFetchOptions, DatasetOverview, type ExportableSpan, type ExtractedUsage, type FlattenOtlpOptions, LLM_CACHED_TOKENS, LLM_CACHED_TOKEN_ATTR_KEYS, LLM_COST_ATTR_KEYS, LLM_COST_USD, LLM_INPUT_TOKENS, LLM_INPUT_TOKEN_ATTR_KEYS, LLM_MODEL_ATTR_KEYS, LLM_MODEL_NAME, LLM_OUTPUT_TOKENS, LLM_OUTPUT_TOKEN_ATTR_KEYS, OPENINFERENCE_SPAN_KIND, OTEL_AGENT_EVAL_SCOPE, type OtelExportConfig, type OtelExporter, type OtlpExport, OtlpFileTraceStore, type OtlpFileTraceStoreOptions, type OtlpFlatLine, type OtlpResourceSpans, type OtlpSpan, type OtlpToRunRecordsOptions, type OtlpTraceRunRecord, type ProjectedOtlpSpan, ProviderRedactor, QueryTracesPage, RawProviderEvent, RawProviderSink, ReplayCache, type ReplayCacheEntry, ReplayCacheMissError, type ReplayCacheStats, type ReplayFetchOptions, Run, RunCompleteHook, RunCompleteHookContext, SPAN_KIND_ATTR_KEYS, SearchSpanResult, SearchTraceResult, SpanNotFoundError, TOOL_NAME, TOOL_NAME_ATTR_KEYS, TRACE_ANALYST_ACTOR_DESCRIPTION, TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION, TRACE_ANALYST_SUBAGENT_DESCRIPTION, type TraceAggregate, TraceAnalysisStore, TraceAnalystFilters, type TraceAnalystHookOptions, TraceAnalystSpanKind, TraceAnalystSpanStatus, TraceFileMissingError, type TraceInsightContext, type TraceInsightFinding, type TraceInsightPanelRole, type TraceInsightPromptInput, type TraceInsightQualityGate, type TraceInsightQuestion, type TraceInsightReadiness, type TraceInsightSuite, type TraceInsightTask, TraceNotFoundError, TraceStore, type TraceStoreSource, type TraceStoreToOtlpOptions, type TracesToOtlpResult, ViewSpansResult, ViewTraceResult, asNumber, asString, buildTraceAnalystTools, buildTraceInsightContext, buildTraceInsightPrompt, captureFetchToRawSink, convertTraceStoresToOtlp, createOtelExporter, createOtelTracingStore, createReplayFetch, defaultTraceInsightPanel, describeTraceInsightScope, domainEvidencePattern, exportRunAsOtlp, extractOtlpAttributes, extractUsage, extractUsageFromResponse, extractUsageFromSse, firstNumberAttr, firstStringAttr, flattenOtlpExportToNdjson, inferDomainKeywords, inferOtlpKind, iterateRawCalls, otelRunCompleteHook, otlpToRunRecords, otlpToTraceRunRecords, planTraceInsightQuestions, projectOtlpFlatLine, readOtlpStatus, scoreTraceInsightReadiness, stringField, tokenizeDomainWords, traceAnalystFunctionGroup, traceAnalystOnRunComplete, traceSpanKindToOpenInferenceKind };
|
|
1019
|
+
export { AnalyzeTracesOptions, AnalyzeTracesResult, type CaptureFetchContext, type CaptureFetchOptions, DEFAULT_REDACTION_RULES, DatasetOverview, type ExportableSpan, type ExtractedUsage, type FlattenOtlpOptions, LLM_CACHED_TOKENS, LLM_CACHED_TOKEN_ATTR_KEYS, LLM_COST_ATTR_KEYS, LLM_COST_USD, LLM_INPUT_TOKENS, LLM_INPUT_TOKEN_ATTR_KEYS, LLM_MODEL_ATTR_KEYS, LLM_MODEL_NAME, LLM_OUTPUT_TOKENS, LLM_OUTPUT_TOKEN_ATTR_KEYS, OPENINFERENCE_SPAN_KIND, OTEL_AGENT_EVAL_SCOPE, type OtelExportConfig, type OtelExporter, type OtlpExport, OtlpFileTraceStore, type OtlpFileTraceStoreOptions, type OtlpFlatLine, type OtlpResourceSpans, type OtlpSpan, type OtlpToRunRecordsOptions, type OtlpTraceRunRecord, type ProjectedOtlpSpan, ProviderRedactor, QueryTracesPage, REDACTION_VERSION, RawProviderEvent, RawProviderSink, type RedactionReport, type RedactionRule, ReplayCache, type ReplayCacheEntry, ReplayCacheMissError, type ReplayCacheStats, type ReplayFetchOptions, Run, RunCompleteHook, RunCompleteHookContext, SPAN_KIND_ATTR_KEYS, SearchSpanResult, SearchTraceResult, SpanNotFoundError, TOOL_NAME, TOOL_NAME_ATTR_KEYS, TRACE_ANALYST_ACTOR_DESCRIPTION, TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION, TRACE_ANALYST_SUBAGENT_DESCRIPTION, type TraceAggregate, TraceAnalysisStore, TraceAnalystFilters, type TraceAnalystHookOptions, TraceAnalystSpanKind, TraceAnalystSpanStatus, TraceFileMissingError, type TraceInsightContext, type TraceInsightFinding, type TraceInsightPanelRole, type TraceInsightPromptInput, type TraceInsightQualityGate, type TraceInsightQuestion, type TraceInsightReadiness, type TraceInsightSuite, type TraceInsightTask, TraceNotFoundError, TraceStore, type TraceStoreSource, type TraceStoreToOtlpOptions, type TracesToOtlpResult, ViewSpansResult, ViewTraceResult, asNumber, asString, buildTraceAnalystTools, buildTraceInsightContext, buildTraceInsightPrompt, captureFetchToRawSink, convertTraceStoresToOtlp, createOtelExporter, createOtelTracingStore, createReplayFetch, defaultTraceInsightPanel, describeTraceInsightScope, domainEvidencePattern, exportRunAsOtlp, extractOtlpAttributes, extractUsage, extractUsageFromResponse, extractUsageFromSse, firstNumberAttr, firstStringAttr, flattenOtlpExportToNdjson, inferDomainKeywords, inferOtlpKind, iterateRawCalls, otelRunCompleteHook, otlpToRunRecords, otlpToTraceRunRecords, planTraceInsightQuestions, projectOtlpFlatLine, readOtlpStatus, redactString, redactValue, scoreTraceInsightReadiness, stringField, tokenizeDomainWords, traceAnalystFunctionGroup, traceAnalystOnRunComplete, traceSpanKindToOpenInferenceKind };
|
package/dist/traces.js
CHANGED
|
@@ -28,7 +28,24 @@ import {
|
|
|
28
28
|
scoreTraceInsightReadiness,
|
|
29
29
|
tokenizeDomainWords,
|
|
30
30
|
traceAnalystOnRunComplete
|
|
31
|
-
} from "./chunk-
|
|
31
|
+
} from "./chunk-RSVSSZKF.js";
|
|
32
|
+
import {
|
|
33
|
+
FAILURE_CLASSES,
|
|
34
|
+
TRACE_SCHEMA_VERSION,
|
|
35
|
+
aggregateLlm,
|
|
36
|
+
argHash,
|
|
37
|
+
groupBy,
|
|
38
|
+
isJudgeSpan,
|
|
39
|
+
isLlmSpan,
|
|
40
|
+
isRetrievalSpan,
|
|
41
|
+
isSandboxSpan,
|
|
42
|
+
isToolSpan,
|
|
43
|
+
judgeSpans,
|
|
44
|
+
llmSpans,
|
|
45
|
+
runFailureClass,
|
|
46
|
+
runsForScenario,
|
|
47
|
+
toolSpans
|
|
48
|
+
} from "./chunk-MHNQWM4I.js";
|
|
32
49
|
import {
|
|
33
50
|
TRACE_ANALYST_ACTOR_DESCRIPTION,
|
|
34
51
|
TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION,
|
|
@@ -41,25 +58,6 @@ import {
|
|
|
41
58
|
redactString,
|
|
42
59
|
redactValue
|
|
43
60
|
} from "./chunk-GGE4NNQT.js";
|
|
44
|
-
import {
|
|
45
|
-
aggregateLlm,
|
|
46
|
-
argHash,
|
|
47
|
-
groupBy,
|
|
48
|
-
judgeSpans,
|
|
49
|
-
llmSpans,
|
|
50
|
-
runFailureClass,
|
|
51
|
-
runsForScenario,
|
|
52
|
-
toolSpans
|
|
53
|
-
} from "./chunk-JZXGWLK5.js";
|
|
54
|
-
import {
|
|
55
|
-
FAILURE_CLASSES,
|
|
56
|
-
TRACE_SCHEMA_VERSION,
|
|
57
|
-
isJudgeSpan,
|
|
58
|
-
isLlmSpan,
|
|
59
|
-
isRetrievalSpan,
|
|
60
|
-
isSandboxSpan,
|
|
61
|
-
isToolSpan
|
|
62
|
-
} from "./chunk-5BKGXME7.js";
|
|
63
61
|
import {
|
|
64
62
|
DEFAULT_TRACE_ANALYST_BUDGETS,
|
|
65
63
|
LLM_CACHED_TOKENS,
|
|
@@ -99,6 +97,13 @@ import {
|
|
|
99
97
|
assertRunCaptured,
|
|
100
98
|
throwIfRunIncomplete
|
|
101
99
|
} from "./chunk-TT4KNT67.js";
|
|
100
|
+
import {
|
|
101
|
+
TraceEmitter,
|
|
102
|
+
llmSpanFromProvider
|
|
103
|
+
} from "./chunk-TVVP3ZZQ.js";
|
|
104
|
+
import "./chunk-VK6HBGAE.js";
|
|
105
|
+
import "./chunk-XJYR7XFV.js";
|
|
106
|
+
import "./chunk-VSMTAMNK.js";
|
|
102
107
|
import {
|
|
103
108
|
FileSystemRawProviderSink,
|
|
104
109
|
InMemoryRawProviderSink,
|
|
@@ -106,13 +111,6 @@ import {
|
|
|
106
111
|
defaultProviderRedactor,
|
|
107
112
|
providerFromBaseUrl
|
|
108
113
|
} from "./chunk-PC4UYEBM.js";
|
|
109
|
-
import "./chunk-VK6HBGAE.js";
|
|
110
|
-
import {
|
|
111
|
-
TraceEmitter,
|
|
112
|
-
llmSpanFromProvider
|
|
113
|
-
} from "./chunk-TVVP3ZZQ.js";
|
|
114
|
-
import "./chunk-XJYR7XFV.js";
|
|
115
|
-
import "./chunk-VSMTAMNK.js";
|
|
116
114
|
import "./chunk-ONWEPEDO.js";
|
|
117
115
|
import "./chunk-PZ5AY32C.js";
|
|
118
116
|
export {
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { c as RunTokenUsage } from './run-record-
|
|
1
|
+
import { c as RunTokenUsage } from './run-record-DksGsfgv.js';
|
|
2
2
|
|
|
3
3
|
/**
|
|
4
4
|
* Pass A substrate types — `runCampaign` is the one primitive every
|
|
@@ -557,4 +557,4 @@ interface CampaignResult<TArtifact = unknown, TScenario extends Scenario = Scena
|
|
|
557
557
|
scenarios: Array<Pick<TScenario, 'id' | 'kind'>>;
|
|
558
558
|
}
|
|
559
559
|
|
|
560
|
-
export { type JudgeAggregate as A, type ScenarioAggregate as B, type CampaignResult as C, type
|
|
560
|
+
export { type JudgeAggregate as A, type ScenarioAggregate as B, type CampaignResult as C, type DispatchContext as D, isProposedCandidate as E, labelTrustRank as F, type GateResult as G, type JudgeScore as J, type LabeledScenarioStore as L, type MutableSurface as M, type OptimizationProposer as O, type ParetoParent as P, type RedactionStatus as R, type Scenario as S, type TraceSpan as T, type JudgeDimension as a, type JudgeConfig as b, type DispatchFn as c, type CampaignTraceWriter as d, type GenerationRecord as e, type SurfaceProposer as f, type Gate as g, type GateDecision as h, type CampaignAggregates as i, type CampaignArtifactWriter as j, type CampaignCellResult as k, type CampaignCostMeter as l, type CodeSurface as m, type GateContext as n, type GenerationCandidate as o, type Mutator as p, type OptimizerConfig as q, type SessionScript as r, type LabeledScenarioWrite as s, type LabeledScenarioSampleArgs as t, type LabeledScenarioRecord as u, type LabelTrust as v, type ProposedCandidate as w, type ProposeContext as x, type LabeledScenarioSource as y, type CampaignTokenUsage as z };
|
package/dist/wire/index.d.ts
CHANGED
|
@@ -1,14 +1,13 @@
|
|
|
1
|
-
import { F as FeedbackTrajectoryStore } from '../feedback-trajectory-
|
|
2
|
-
import { T as TraceStore } from '../store-
|
|
1
|
+
import { F as FeedbackTrajectoryStore } from '../feedback-trajectory-pDcz1lQ1.js';
|
|
2
|
+
import { T as TraceStore } from '../store-BsVi7ncX.js';
|
|
3
3
|
import { z } from 'zod';
|
|
4
4
|
import { OpenAPIObject } from 'openapi3-ts/oas31';
|
|
5
5
|
import * as hono_types from 'hono/types';
|
|
6
6
|
import { ServerType } from '@hono/node-server';
|
|
7
7
|
import { Hono } from 'hono';
|
|
8
|
-
import '../
|
|
9
|
-
import '../
|
|
10
|
-
import '../
|
|
11
|
-
import '../dataset-DS7ytHZU.js';
|
|
8
|
+
import '../emitter-BRchAAAx.js';
|
|
9
|
+
import '../schema-SGWcK9wa.js';
|
|
10
|
+
import '../dataset-NENEzRgk.js';
|
|
12
11
|
import '../errors-oeQrLqXC.js';
|
|
13
12
|
|
|
14
13
|
declare const RubricDimensionSchema: z.ZodObject<{
|
|
@@ -61,25 +61,26 @@ Start at the top row and only move down when the row's *"reach for it when"* mat
|
|
|
61
61
|
| `gepaProposer` | You want the strong default: reflective full-surface prompt rewrites, grounded in findings, keeping a Pareto frontier of complementary winners. | prompt string | **Yes — `improve({ surface: 'prompt' })` default.** Proven live. |
|
|
62
62
|
| `skillOptProposer` | You are editing a structured `SKILL.md`/runbook and want small anchored add/delete/replace patches that preserve earlier rules. | skill/prompt string | **Yes — `improve({ surface: 'skills' })` default.** Not yet proven live. |
|
|
63
63
|
| `parameterSweepProposer` | The likely fix is a config knob, not words — `retrieval.k`, `temperature`, `max_tokens`. You give it candidate patches; it applies them to a JSON surface. | JSON config string | Yes, but you supply the candidate list. |
|
|
64
|
-
| `fapoProposer` | You want *evidence to decide when to escalate*: try prompt edits first, move to parameters, then to structural code — one scoped change per cycle, only escalating when the cheaper level is exhausted.
|
|
64
|
+
| `fapoProposer` | You want *evidence to decide when to escalate*: try prompt edits first, move to parameters, then to structural code — one scoped change per cycle, only escalating when the cheaper level is exhausted. | whatever its level proposers return | Exported; you wire the level proposers. |
|
|
65
|
+
| `compositeProposer` | You want several proposers to share one candidate-generation budget in the same round. It allocates the population by declared weights, preserves member provenance, deduplicates surfaces, and isolates a member failure unless every member fails. | whatever its member proposers return | Exported; you wire the member proposers. |
|
|
65
66
|
| `aceProposer` | You are accumulating hard-won lessons into a playbook and must **never** summarize an old lesson away (append-only, provenance-tagged). | playbook string | Exported. |
|
|
66
67
|
| `memoryCurationProposer` | Same as ACE but you want a compact, deduped, re-ranked memory instead of append-only growth. | memory string | Exported. |
|
|
67
68
|
| `evolutionaryProposer` | You want blind population search (mutate → measure → select) with no reflection over findings — a cheap control or a baseline to beat. | any string | Exported. |
|
|
68
69
|
| `traceAnalystProposer` | Bench-only: race our trace-analysis evidence engine head-to-head inside `compareProposers`. | prompt string | Bench-only. |
|
|
69
70
|
| `haloProposer` | Bench-only: race the external `halo-engine` analysis against ours. | prompt string | Bench-only, external. |
|
|
70
71
|
|
|
71
|
-
Default path: `gepaProposer` for prompts; add `parameterSweepProposer` when a config knob is the suspect; wrap
|
|
72
|
+
Default path: `gepaProposer` for prompts; add `parameterSweepProposer` when a config knob is the suspect; wrap levels in `fapoProposer` when the loop should decide *when* to escalate, or use `compositeProposer` when multiple proposer families must split one fixed population budget.
|
|
72
73
|
|
|
73
|
-
## Composing proposers —
|
|
74
|
+
## Composing proposers — four distinct shapes
|
|
74
75
|
|
|
75
|
-
|
|
76
|
-
Composition happens in exactly three shapes:
|
|
76
|
+
Choose the shape that matches the experiment:
|
|
77
77
|
|
|
78
|
-
1. **
|
|
79
|
-
2. **
|
|
80
|
-
3. **
|
|
78
|
+
1. **Portfolio** — `compositeProposer` splits one generation's population across member proposers by fixed weights and returns one provenance-labelled pool.
|
|
79
|
+
2. **Escalate** — `fapoProposer` wraps prompt + parameter + structural levels into one proposer and spends on the cheapest level until evidence says to escalate.
|
|
80
|
+
3. **Race** — `compareProposers` gives proposers separate loops, then re-scores their winners on one holdout and returns per-proposer lift intervals plus pairwise results.
|
|
81
|
+
4. **Plug in** — hand any proposer to `runImprovementLoop({ proposer })`, or use `improve({ surface, generator })` in `@tangle-network/agent-runtime`.
|
|
81
82
|
|
|
82
|
-
###
|
|
83
|
+
### 2 + 4 — compose by escalation, then run the improvement loop
|
|
83
84
|
|
|
84
85
|
```ts
|
|
85
86
|
import {
|
|
@@ -93,8 +94,8 @@ import {
|
|
|
93
94
|
const llm = { baseUrl: process.env.TANGLE_BASE_URL, apiKey: process.env.TANGLE_API_KEY }
|
|
94
95
|
const model = 'deepseek-v4-flash'
|
|
95
96
|
|
|
96
|
-
// Compose: prompt edits first (GEPA),
|
|
97
|
-
// prompt-level search plateaus.
|
|
97
|
+
// Compose by escalation: prompt edits first (GEPA), then a config knob only
|
|
98
|
+
// when prompt-level search plateaus. This is distinct from a peer portfolio.
|
|
98
99
|
const proposer = fapoProposer({
|
|
99
100
|
scope: { allowedLevels: ['prompt', 'parameter'] }, // no structural/code tier here
|
|
100
101
|
promptProposer: gepaProposer({ llm, model, target: 'agent system prompt' }),
|
|
@@ -147,7 +148,7 @@ const out = await improve(profile, findings, {
|
|
|
147
148
|
if (out.shipped) deploy(out.profile) // out.lift is the held-out winner − baseline
|
|
148
149
|
```
|
|
149
150
|
|
|
150
|
-
###
|
|
151
|
+
### 3 — race proposers head-to-head for a lift CI
|
|
151
152
|
|
|
152
153
|
```ts
|
|
153
154
|
import {
|
|
@@ -196,7 +197,7 @@ Every entrant is re-scored on the **same** holdout with the **same** judges, so
|
|
|
196
197
|
- Do not put eval logic inside a proposer — scoring lives in `dispatch` + `judges`, proposing lives in the proposer.
|
|
197
198
|
- Do not let a proposer read held-out judge scores — `ProposeContext` makes that a compile error on purpose; a proposer that games the acceptance axis is an oracle, not an optimizer.
|
|
198
199
|
- Do not read `lift` without `result.power`/MDE — a "+4" on a valset too small to detect +4 is noise wearing a number.
|
|
199
|
-
- Do not
|
|
200
|
+
- Do not confuse `compositeProposer` with `fapoProposer`: the former allocates one fixed population across peers, while the latter escalates through ordered levels from cheaper to more structural changes.
|
|
200
201
|
|
|
201
202
|
### neutralizationGate — the placebo / content-causality control
|
|
202
203
|
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@tangle-network/agent-eval",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.110.1",
|
|
4
4
|
"description": "Evaluate and improve AI agents from runs, traces, judges, and feedback. Compare candidates, cluster failures, measure lift, and gate releases.",
|
|
5
5
|
"homepage": "https://github.com/tangle-network/agent-eval#readme",
|
|
6
6
|
"repository": {
|
|
@@ -39,11 +39,6 @@
|
|
|
39
39
|
"import": "./dist/rl.js",
|
|
40
40
|
"default": "./dist/rl.js"
|
|
41
41
|
},
|
|
42
|
-
"./diagnose": {
|
|
43
|
-
"types": "./dist/diagnose.d.ts",
|
|
44
|
-
"import": "./dist/diagnose.js",
|
|
45
|
-
"default": "./dist/diagnose.js"
|
|
46
|
-
},
|
|
47
42
|
"./fuzz": {
|
|
48
43
|
"types": "./dist/fuzz.d.ts",
|
|
49
44
|
"import": "./dist/fuzz.js",
|
|
@@ -54,16 +49,6 @@
|
|
|
54
49
|
"import": "./dist/traces.js",
|
|
55
50
|
"default": "./dist/traces.js"
|
|
56
51
|
},
|
|
57
|
-
"./telemetry": {
|
|
58
|
-
"types": "./dist/telemetry/index.d.ts",
|
|
59
|
-
"import": "./dist/telemetry/index.js",
|
|
60
|
-
"default": "./dist/telemetry/index.js"
|
|
61
|
-
},
|
|
62
|
-
"./telemetry/file": {
|
|
63
|
-
"types": "./dist/telemetry/file.d.ts",
|
|
64
|
-
"import": "./dist/telemetry/file.js",
|
|
65
|
-
"default": "./dist/telemetry/file.js"
|
|
66
|
-
},
|
|
67
52
|
"./wire": {
|
|
68
53
|
"types": "./dist/wire/index.d.ts",
|
|
69
54
|
"import": "./dist/wire/index.js",
|
|
@@ -84,41 +69,16 @@
|
|
|
84
69
|
"import": "./dist/meta-eval/index.js",
|
|
85
70
|
"default": "./dist/meta-eval/index.js"
|
|
86
71
|
},
|
|
87
|
-
"./prm": {
|
|
88
|
-
"types": "./dist/prm/index.d.ts",
|
|
89
|
-
"import": "./dist/prm/index.js",
|
|
90
|
-
"default": "./dist/prm/index.js"
|
|
91
|
-
},
|
|
92
72
|
"./builder-eval": {
|
|
93
73
|
"types": "./dist/builder-eval/index.d.ts",
|
|
94
74
|
"import": "./dist/builder-eval/index.js",
|
|
95
75
|
"default": "./dist/builder-eval/index.js"
|
|
96
76
|
},
|
|
97
|
-
"./governance": {
|
|
98
|
-
"types": "./dist/governance/index.d.ts",
|
|
99
|
-
"import": "./dist/governance/index.js",
|
|
100
|
-
"default": "./dist/governance/index.js"
|
|
101
|
-
},
|
|
102
|
-
"./knowledge": {
|
|
103
|
-
"types": "./dist/knowledge/index.d.ts",
|
|
104
|
-
"import": "./dist/knowledge/index.js",
|
|
105
|
-
"default": "./dist/knowledge/index.js"
|
|
106
|
-
},
|
|
107
77
|
"./matrix": {
|
|
108
78
|
"types": "./dist/matrix/index.d.ts",
|
|
109
79
|
"import": "./dist/matrix/index.js",
|
|
110
80
|
"default": "./dist/matrix/index.js"
|
|
111
81
|
},
|
|
112
|
-
"./perf": {
|
|
113
|
-
"types": "./dist/perf/index.d.ts",
|
|
114
|
-
"import": "./dist/perf/index.js",
|
|
115
|
-
"default": "./dist/perf/index.js"
|
|
116
|
-
},
|
|
117
|
-
"./product-benchmark": {
|
|
118
|
-
"types": "./dist/product-benchmark/index.d.ts",
|
|
119
|
-
"import": "./dist/product-benchmark/index.js",
|
|
120
|
-
"default": "./dist/product-benchmark/index.js"
|
|
121
|
-
},
|
|
122
82
|
"./multishot": {
|
|
123
83
|
"types": "./dist/multishot/index.d.ts",
|
|
124
84
|
"import": "./dist/multishot/index.js",
|
|
@@ -139,51 +99,21 @@
|
|
|
139
99
|
"import": "./dist/authenticity/index.js",
|
|
140
100
|
"default": "./dist/authenticity/index.js"
|
|
141
101
|
},
|
|
142
|
-
"./groundedness": {
|
|
143
|
-
"types": "./dist/groundedness/index.d.ts",
|
|
144
|
-
"import": "./dist/groundedness/index.js",
|
|
145
|
-
"default": "./dist/groundedness/index.js"
|
|
146
|
-
},
|
|
147
102
|
"./belief-state": {
|
|
148
103
|
"types": "./dist/belief-state/index.d.ts",
|
|
149
104
|
"import": "./dist/belief-state/index.js",
|
|
150
105
|
"default": "./dist/belief-state/index.js"
|
|
151
106
|
},
|
|
152
|
-
"./workflow": {
|
|
153
|
-
"types": "./dist/workflow/index.d.ts",
|
|
154
|
-
"import": "./dist/workflow/index.js",
|
|
155
|
-
"default": "./dist/workflow/index.js"
|
|
156
|
-
},
|
|
157
107
|
"./contract": {
|
|
158
108
|
"types": "./dist/contract/index.d.ts",
|
|
159
109
|
"import": "./dist/contract/index.js",
|
|
160
110
|
"default": "./dist/contract/index.js"
|
|
161
111
|
},
|
|
162
|
-
"./adapters/langchain": {
|
|
163
|
-
"types": "./dist/adapters/langchain.d.ts",
|
|
164
|
-
"import": "./dist/adapters/langchain.js",
|
|
165
|
-
"default": "./dist/adapters/langchain.js"
|
|
166
|
-
},
|
|
167
|
-
"./adapters/http": {
|
|
168
|
-
"types": "./dist/adapters/http.d.ts",
|
|
169
|
-
"import": "./dist/adapters/http.js",
|
|
170
|
-
"default": "./dist/adapters/http.js"
|
|
171
|
-
},
|
|
172
|
-
"./adapters/otel": {
|
|
173
|
-
"types": "./dist/adapters/otel.d.ts",
|
|
174
|
-
"import": "./dist/adapters/otel.js",
|
|
175
|
-
"default": "./dist/adapters/otel.js"
|
|
176
|
-
},
|
|
177
112
|
"./hosted": {
|
|
178
113
|
"types": "./dist/hosted/index.d.ts",
|
|
179
114
|
"import": "./dist/hosted/index.js",
|
|
180
115
|
"default": "./dist/hosted/index.js"
|
|
181
116
|
},
|
|
182
|
-
"./testing": {
|
|
183
|
-
"types": "./dist/testing.d.ts",
|
|
184
|
-
"import": "./dist/testing.js",
|
|
185
|
-
"default": "./dist/testing.js"
|
|
186
|
-
},
|
|
187
117
|
"./openapi.json": {
|
|
188
118
|
"default": "./dist/openapi.json"
|
|
189
119
|
}
|