@tangle-network/agent-eval 0.117.1 → 0.118.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +30 -0
- package/dist/analyst/index.d.ts +2772 -21
- package/dist/analyst/index.js +7 -6
- package/dist/analyst/index.js.map +1 -1
- package/dist/belief-state/index.d.ts +706 -10
- package/dist/belief-state/index.js +2 -1
- package/dist/belief-state/index.js.map +1 -1
- package/dist/benchmarks/index.d.ts +958 -14
- package/dist/benchmarks/index.js +12 -10
- package/dist/builder-eval/index.d.ts +449 -4
- package/dist/builder-eval/index.js +4 -3
- package/dist/builder-eval/index.js.map +1 -1
- package/dist/campaign/index.d.ts +4275 -73
- package/dist/campaign/index.js +12 -10
- package/dist/{chunk-VF3XSYTI.js → chunk-33JA4TFA.js} +6 -6
- package/dist/{chunk-4JLWXDYA.js → chunk-3EHHMC6E.js} +2 -2
- package/dist/{chunk-CCZIVI3F.js → chunk-BFW56GTT.js} +2 -2
- package/dist/{chunk-YZPO4UHR.js → chunk-FTUMG2U7.js} +124 -149
- package/dist/chunk-FTUMG2U7.js.map +1 -0
- package/dist/{chunk-E4BUPP7Z.js → chunk-HKUCJ437.js} +38 -66
- package/dist/chunk-HKUCJ437.js.map +1 -0
- package/dist/{chunk-HZHNRYHK.js → chunk-K6N6XJJX.js} +2 -2
- package/dist/chunk-KSDQVPLR.js +286 -0
- package/dist/chunk-KSDQVPLR.js.map +1 -0
- package/dist/chunk-MA6HLL3S.js +65 -0
- package/dist/chunk-MA6HLL3S.js.map +1 -0
- package/dist/{chunk-MGEHEHSN.js → chunk-OIIMMLRB.js} +11 -11
- package/dist/{chunk-DXZRATT5.js → chunk-OYZAPX5G.js} +3 -3
- package/dist/chunk-PXE2VKMX.js +140 -0
- package/dist/chunk-PXE2VKMX.js.map +1 -0
- package/dist/{chunk-JSJZ4PJ6.js → chunk-Q442S5AS.js} +17 -17
- package/dist/{chunk-ODVOOEWQ.js → chunk-QBRSJK47.js} +2 -2
- package/dist/{chunk-S2F4J57L.js → chunk-QKEGNI5B.js} +77 -32
- package/dist/chunk-QKEGNI5B.js.map +1 -0
- package/dist/{chunk-5UF54T55.js → chunk-S3UZOQ5Y.js} +34 -7
- package/dist/chunk-S3UZOQ5Y.js.map +1 -0
- package/dist/{chunk-FQNLDL4D.js → chunk-SVH2ANFD.js} +136 -4
- package/dist/chunk-SVH2ANFD.js.map +1 -0
- package/dist/{chunk-GQCZRZ7L.js → chunk-U5CHZ5M3.js} +9 -9
- package/dist/{chunk-TVVP3ZZQ.js → chunk-VQMK5FMP.js} +2 -1
- package/dist/{chunk-TVVP3ZZQ.js.map → chunk-VQMK5FMP.js.map} +1 -1
- package/dist/chunk-WDHBCA3M.js +31 -0
- package/dist/chunk-WDHBCA3M.js.map +1 -0
- package/dist/{chunk-HQPHZGL6.js → chunk-YLMUS4MM.js} +9 -9
- package/dist/{chunk-LQUTGLOZ.js → chunk-ZET2UAYW.js} +15 -65
- package/dist/chunk-ZET2UAYW.js.map +1 -0
- package/dist/contract/index.d.ts +4012 -38
- package/dist/contract/index.js +56 -29
- package/dist/contract/index.js.map +1 -1
- package/dist/control.d.ts +1013 -9
- package/dist/control.js +4 -3
- package/dist/fuzz.d.ts +194 -4
- package/dist/fuzz.js +3 -3
- package/dist/hosted/index.d.ts +498 -17
- package/dist/index.d.ts +10992 -1288
- package/dist/index.js +101 -80
- package/dist/index.js.map +1 -1
- package/dist/matrix/index.d.ts +139 -4
- package/dist/meta-eval/index.d.ts +862 -15
- package/dist/meta-eval/index.js +2 -1
- package/dist/meta-eval/index.js.map +1 -1
- package/dist/multishot/index.d.ts +214 -14
- package/dist/openapi.json +1 -1
- package/dist/pipelines/index.d.ts +392 -7
- package/dist/pipelines/index.js +5 -3
- package/dist/pipelines/index.js.map +1 -1
- package/dist/reporting.d.ts +1277 -17
- package/dist/rl.d.ts +2359 -28
- package/dist/rl.js +6 -5
- package/dist/rl.js.map +1 -1
- package/dist/storyboard/index.d.ts +86 -1
- package/dist/trace-attributes.d.ts +16 -0
- package/dist/trace-attributes.js +32 -0
- package/dist/trace-attributes.js.map +1 -0
- package/dist/traces.d.ts +1978 -697
- package/dist/traces.js +53 -32
- package/dist/wire/index.d.ts +655 -9
- package/docs/insight-report.md +44 -0
- package/package.json +9 -3
- package/dist/adversarial-B7loGVVX.d.ts +0 -19
- package/dist/analyst-C8HHvfJp.d.ts +0 -88
- package/dist/analyze-runs--2x39HZ7.d.ts +0 -81
- package/dist/baseline-DKq3gJpP.d.ts +0 -141
- package/dist/calibration-C8MTS7cw.d.ts +0 -101
- package/dist/chunk-5UF54T55.js.map +0 -1
- package/dist/chunk-E4BUPP7Z.js.map +0 -1
- package/dist/chunk-FQNLDL4D.js.map +0 -1
- package/dist/chunk-LQUTGLOZ.js.map +0 -1
- package/dist/chunk-S2F4J57L.js.map +0 -1
- package/dist/chunk-YZPO4UHR.js.map +0 -1
- package/dist/code-agent-session-CjZsVd19.d.ts +0 -87
- package/dist/control-6vuGfmDH.d.ts +0 -258
- package/dist/cost-ledger-DWy3XdJc.d.ts +0 -183
- package/dist/dataset-NENEzRgk.d.ts +0 -115
- package/dist/default-registry-DaK8b3fv.d.ts +0 -155
- package/dist/emitter-CjD7vUwv.d.ts +0 -122
- package/dist/errors-oeQrLqXC.d.ts +0 -74
- package/dist/failure-cluster-DOAcSJ87.d.ts +0 -76
- package/dist/feedback-trajectory-BUnM58xL.d.ts +0 -348
- package/dist/gepa-eESocoDi.d.ts +0 -642
- package/dist/index-PdX4VnPA.d.ts +0 -423
- package/dist/insight-report-DY4nDW9Q.d.ts +0 -310
- package/dist/integrity-DqlBiLyK.d.ts +0 -81
- package/dist/judge-calibration-7C-IDmKr.d.ts +0 -145
- package/dist/kind-factory-ClZmO25A.d.ts +0 -171
- package/dist/llm-client-qoDd18Qz.d.ts +0 -289
- package/dist/multi-layer-verifier-BsqKuLyN.d.ts +0 -150
- package/dist/off-policy-DiwuKKg7.d.ts +0 -132
- package/dist/outcome-store-rnXLEqSn.d.ts +0 -63
- package/dist/policy-edit-wG9uFEFm.d.ts +0 -455
- package/dist/pre-registration-BWQhJ3vz.d.ts +0 -761
- package/dist/provenance-DpjwyseI.d.ts +0 -541
- package/dist/query-CF7PG61p.d.ts +0 -35
- package/dist/raw-provider-sink-C46HDghv.d.ts +0 -132
- package/dist/release-report-C8G2i5Xi.d.ts +0 -236
- package/dist/researcher-C8XyxQsu.d.ts +0 -387
- package/dist/rubric-predictive-validity-p49lLVrE.d.ts +0 -105
- package/dist/run-record-BDH49H2E.d.ts +0 -360
- package/dist/runtime-trajectory-DGBIUt4B.d.ts +0 -49
- package/dist/schema-B3Q3l9Z_.d.ts +0 -201
- package/dist/semantic-concept-judge-CXnPEJbf.d.ts +0 -723
- package/dist/sequential-5iSVfzl2.d.ts +0 -139
- package/dist/series-convergence-D5OWMBg6.d.ts +0 -33
- package/dist/statistics-KUnG73jH.d.ts +0 -494
- package/dist/storage-DrX3v_5B.d.ts +0 -50
- package/dist/store-C1YxJDEK.d.ts +0 -248
- package/dist/store-DGqD0Pyo.d.ts +0 -116
- package/dist/summary-report-C5bKFfm-.d.ts +0 -445
- package/dist/test-graded-scenario-B0ybnPY7.d.ts +0 -166
- package/dist/types-BSw1rOUB.d.ts +0 -634
- package/dist/types-BUxNaJ8c.d.ts +0 -108
- package/dist/types-BkfcQnxV.d.ts +0 -313
- package/dist/verdict-C9MlYujm.d.ts +0 -35
- /package/dist/{chunk-VF3XSYTI.js.map → chunk-33JA4TFA.js.map} +0 -0
- /package/dist/{chunk-4JLWXDYA.js.map → chunk-3EHHMC6E.js.map} +0 -0
- /package/dist/{chunk-CCZIVI3F.js.map → chunk-BFW56GTT.js.map} +0 -0
- /package/dist/{chunk-HZHNRYHK.js.map → chunk-K6N6XJJX.js.map} +0 -0
- /package/dist/{chunk-MGEHEHSN.js.map → chunk-OIIMMLRB.js.map} +0 -0
- /package/dist/{chunk-DXZRATT5.js.map → chunk-OYZAPX5G.js.map} +0 -0
- /package/dist/{chunk-JSJZ4PJ6.js.map → chunk-Q442S5AS.js.map} +0 -0
- /package/dist/{chunk-ODVOOEWQ.js.map → chunk-QBRSJK47.js.map} +0 -0
- /package/dist/{chunk-GQCZRZ7L.js.map → chunk-U5CHZ5M3.js.map} +0 -0
- /package/dist/{chunk-HQPHZGL6.js.map → chunk-YLMUS4MM.js.map} +0 -0
|
@@ -1,155 +0,0 @@
|
|
|
1
|
-
import { AxAIService } from '@ax-llm/ax';
|
|
2
|
-
import { T as TraceAnalystKindSpec } from './kind-factory-ClZmO25A.js';
|
|
3
|
-
import { A as Analyst, a as AnalystContext, b as AnalystRunSummary, c as AnalystFinding, d as AnalystRunResult, C as ChatClient, e as AnalystRunInputs, f as AnalystRunEvent } from './policy-edit-wG9uFEFm.js';
|
|
4
|
-
|
|
5
|
-
/**
|
|
6
|
-
* AnalystRegistry — orchestrate N analysts against one run.
|
|
7
|
-
*
|
|
8
|
-
* Owns three responsibilities and only three:
|
|
9
|
-
* 1. Registration — ids must be unique; bad registrations fail loudly
|
|
10
|
-
* at register-time, not run-time.
|
|
11
|
-
* 2. Routing — each analyst declares its `inputKind`; the registry
|
|
12
|
-
* picks the matching field from AnalystRunInputs and skips the
|
|
13
|
-
* analyst with a logged reason if it's missing.
|
|
14
|
-
* 3. Isolation — one analyst's exception MUST NOT stop other analysts.
|
|
15
|
-
* Failed analysts produce zero findings + a 'failed' summary row.
|
|
16
|
-
*
|
|
17
|
-
* Cross-cutting concerns (telemetry, error → finding conversion, cost
|
|
18
|
-
* ingestion, storage rotation) live in `AnalystHooks`. Budget shaping
|
|
19
|
-
* (equal split vs weighted vs custom) lives in `BudgetPolicy`. Both
|
|
20
|
-
* have sensible defaults; consumers override only what they need.
|
|
21
|
-
*/
|
|
22
|
-
|
|
23
|
-
interface AnalystHooks {
|
|
24
|
-
/** Before analyze() — last chance to mutate ctx (e.g. inject tags, override budget). */
|
|
25
|
-
onBeforeAnalyze?(args: {
|
|
26
|
-
analyst: Analyst;
|
|
27
|
-
ctx: AnalystContext;
|
|
28
|
-
runId: string;
|
|
29
|
-
}): void | Promise<void>;
|
|
30
|
-
/** After every analyst (ok | failed | skipped). Use for telemetry, ingestion, rotation. */
|
|
31
|
-
onAfterAnalyze?(args: {
|
|
32
|
-
analyst: Analyst;
|
|
33
|
-
summary: AnalystRunSummary;
|
|
34
|
-
findings: AnalystFinding[];
|
|
35
|
-
runId: string;
|
|
36
|
-
}): void | Promise<void>;
|
|
37
|
-
/**
|
|
38
|
-
* On analyst exception. Hook MAY return findings to convert the
|
|
39
|
-
* error into structured findings; the summary still reports 'failed'.
|
|
40
|
-
* Return void to keep the default empty-findings behavior.
|
|
41
|
-
*/
|
|
42
|
-
onError?(args: {
|
|
43
|
-
analyst: Analyst;
|
|
44
|
-
error: Error;
|
|
45
|
-
runId: string;
|
|
46
|
-
}): AnalystFinding[] | undefined | Promise<AnalystFinding[] | undefined>;
|
|
47
|
-
/** Once after registry.run() completes. Use for final aggregation, persistence. */
|
|
48
|
-
onComplete?(args: {
|
|
49
|
-
result: AnalystRunResult;
|
|
50
|
-
}): void | Promise<void>;
|
|
51
|
-
}
|
|
52
|
-
interface BudgetPolicy {
|
|
53
|
-
/** Overall USD cap across the registry.run(). */
|
|
54
|
-
totalUsd?: number;
|
|
55
|
-
/** Per-analyst weight for the default allocator. Missing ids get weight 1. */
|
|
56
|
-
weights?: Record<string, number>;
|
|
57
|
-
/**
|
|
58
|
-
* Custom allocator — receives the analyst, remaining/total budget, and
|
|
59
|
-
* the count of analysts that will run. Returns the per-analyst budget
|
|
60
|
-
* (or undefined to leave it uncapped). Overrides weights when set.
|
|
61
|
-
*/
|
|
62
|
-
allocate?: (args: {
|
|
63
|
-
analyst: Analyst;
|
|
64
|
-
totalUsd: number | undefined;
|
|
65
|
-
remainingUsd: number | undefined;
|
|
66
|
-
runningCount: number;
|
|
67
|
-
}) => number | undefined;
|
|
68
|
-
}
|
|
69
|
-
interface AnalystRegistryOptions {
|
|
70
|
-
/** Shared chat client passed to every LLM analyst via AnalystContext. */
|
|
71
|
-
chat?: ChatClient;
|
|
72
|
-
/** Logger callback. Defaults to a no-op. */
|
|
73
|
-
log?: (msg: string, fields?: Record<string, unknown>) => void;
|
|
74
|
-
/** Hooks invoked around analyze() — observability + customization seam. */
|
|
75
|
-
hooks?: AnalystHooks;
|
|
76
|
-
/** Default budget when run() doesn't override. */
|
|
77
|
-
defaultBudget?: BudgetPolicy;
|
|
78
|
-
}
|
|
79
|
-
interface RegistryRunOpts {
|
|
80
|
-
/** Restrict to a subset of registered analysts by id. */
|
|
81
|
-
only?: string[];
|
|
82
|
-
/** Skip these analysts even if registered. Useful for cheap iteration. */
|
|
83
|
-
skip?: string[];
|
|
84
|
-
/** Budget policy — totalUsd + optional weights/allocator. Falls back to options.defaultBudget. */
|
|
85
|
-
budget?: BudgetPolicy;
|
|
86
|
-
/** Wall-clock cap. Analysts SHOULD honor `ctx.deadlineMs`. */
|
|
87
|
-
timeoutMs?: number;
|
|
88
|
-
/** Abort signal — forwarded into every analyst's context. */
|
|
89
|
-
signal?: AbortSignal;
|
|
90
|
-
/** Tags echoed into AnalystContext.tags — useful for tracking environment/version in findings. */
|
|
91
|
-
tags?: Record<string, string>;
|
|
92
|
-
/**
|
|
93
|
-
* Prior-run findings made available as retrieval context to every
|
|
94
|
-
* analyst via `ctx.priorFindings`. The registry forwards the slice
|
|
95
|
-
* whose `analyst_id` matches each registered analyst so a kind sees
|
|
96
|
-
* only its own history. Pass `{ '*': findings }` to broadcast to
|
|
97
|
-
* every analyst (useful for cross-kind chaining where the improvement
|
|
98
|
-
* analyst consumes upstream failure findings).
|
|
99
|
-
*/
|
|
100
|
-
priorFindings?: ReadonlyArray<AnalystFinding> | Record<string, ReadonlyArray<AnalystFinding>>;
|
|
101
|
-
}
|
|
102
|
-
declare class AnalystRegistry {
|
|
103
|
-
private readonly analysts;
|
|
104
|
-
private readonly options;
|
|
105
|
-
constructor(options?: AnalystRegistryOptions);
|
|
106
|
-
register(analyst: Analyst): void;
|
|
107
|
-
list(): ReadonlyArray<{
|
|
108
|
-
id: string;
|
|
109
|
-
description: string;
|
|
110
|
-
version: string;
|
|
111
|
-
cost: Analyst['cost'];
|
|
112
|
-
}>;
|
|
113
|
-
run(runId: string, inputs: AnalystRunInputs, runOpts?: RegistryRunOpts): Promise<AnalystRunResult>;
|
|
114
|
-
/**
|
|
115
|
-
* Streaming counterpart to `run()`. Emits `AnalystRunEvent` values
|
|
116
|
-
* in real time — `run-started`, then per-analyst `skipped` /
|
|
117
|
-
* `started` / `completed`, then a terminal `run-completed` whose
|
|
118
|
-
* payload is the full `AnalystRunResult`. UIs use this to render
|
|
119
|
-
* progress; persistence consumers use `run()` and read the result.
|
|
120
|
-
*
|
|
121
|
-
* Hooks (`onBeforeAnalyze` / `onAfterAnalyze` / `onError` /
|
|
122
|
-
* `onComplete`) fire as before — streaming is additive, not a hook
|
|
123
|
-
* replacement.
|
|
124
|
-
*/
|
|
125
|
-
runStream(runId: string, inputs: AnalystRunInputs, runOpts?: RegistryRunOpts): AsyncGenerator<AnalystRunEvent, void, void>;
|
|
126
|
-
private selectAnalysts;
|
|
127
|
-
private routeInput;
|
|
128
|
-
}
|
|
129
|
-
|
|
130
|
-
/**
|
|
131
|
-
* `buildDefaultAnalystRegistry` — the canonical analyst suite, so consumers
|
|
132
|
-
* stop hand-wiring `new AnalystRegistry()` + per-kind `createTraceAnalystKind`.
|
|
133
|
-
*
|
|
134
|
-
* The deterministic `behavioralAnalyst` is ALWAYS registered (it needs no
|
|
135
|
-
* model and is model-agnostic by construction). The agentic RLM kinds are
|
|
136
|
-
* registered only when an `ai` service is supplied — so a caller with no LLM
|
|
137
|
-
* still gets the full behavioral/efficiency diagnosis, and the substrate's
|
|
138
|
-
* "any model (including no model)" guarantee holds at the suite level.
|
|
139
|
-
*/
|
|
140
|
-
|
|
141
|
-
interface DefaultAnalystRegistryOptions {
|
|
142
|
-
/** Ax service for the agentic RLM kinds. Omit → only the deterministic analyst. */
|
|
143
|
-
ai?: AxAIService;
|
|
144
|
-
/** Model for the agentic kinds (falls back to the ai service default). */
|
|
145
|
-
model?: string;
|
|
146
|
-
/** Which agentic kinds to register when `ai` is present. Default = the shipped suite. */
|
|
147
|
-
kinds?: readonly TraceAnalystKindSpec[];
|
|
148
|
-
/** Set false to omit the deterministic behavioral analyst (default: include). */
|
|
149
|
-
includeBehavioral?: boolean;
|
|
150
|
-
/** Forwarded to the AnalystRegistry constructor (signal, tags, priorFindings). */
|
|
151
|
-
registry?: AnalystRegistryOptions;
|
|
152
|
-
}
|
|
153
|
-
declare function buildDefaultAnalystRegistry(opts?: DefaultAnalystRegistryOptions): AnalystRegistry;
|
|
154
|
-
|
|
155
|
-
export { AnalystRegistry as A, type BudgetPolicy as B, type DefaultAnalystRegistryOptions as D, type RegistryRunOpts as R, type AnalystHooks as a, type AnalystRegistryOptions as b, buildDefaultAnalystRegistry as c };
|
|
@@ -1,122 +0,0 @@
|
|
|
1
|
-
import { a as RunOutcome, R as Run, S as Span, b as SpanKind, L as LlmSpan, T as ToolSpan, c as RetrievalSpan, J as JudgeSpan, d as SandboxSpan, E as EventKind, e as TraceEvent, B as BudgetLedgerEntry, A as Artifact, M as Message } from './schema-B3Q3l9Z_.js';
|
|
2
|
-
import { T as TraceStore } from './store-DGqD0Pyo.js';
|
|
3
|
-
|
|
4
|
-
/**
|
|
5
|
-
* TraceEmitter — hierarchical span builder that auto-parents using an
|
|
6
|
-
* internal stack. One emitter per Run; emitters do NOT share state.
|
|
7
|
-
*
|
|
8
|
-
* Convenience methods (`llm`, `tool`, `retrieval`, `judge`, `sandbox`)
|
|
9
|
-
* return a `SpanHandle` with `.end()` / `.fail()` so callers don't
|
|
10
|
-
* have to thread spanIds manually. For async workflows that can't use
|
|
11
|
-
* the stack (e.g. fan-out parallel calls), pass `parentSpanId`
|
|
12
|
-
* explicitly.
|
|
13
|
-
*/
|
|
14
|
-
|
|
15
|
-
interface SpanHandle<S extends Span = Span> {
|
|
16
|
-
span: S;
|
|
17
|
-
end(patch?: Partial<S>): Promise<void>;
|
|
18
|
-
fail(error: string | Error, patch?: Partial<S>): Promise<void>;
|
|
19
|
-
}
|
|
20
|
-
interface RunCompleteHookContext {
|
|
21
|
-
runId: string;
|
|
22
|
-
emitter: TraceEmitter;
|
|
23
|
-
store: TraceStore;
|
|
24
|
-
/** Outcome the caller passed to `endRun` (undefined for `abortRun`). */
|
|
25
|
-
outcome?: RunOutcome;
|
|
26
|
-
/** Final run status. */
|
|
27
|
-
status: 'completed' | 'failed' | 'aborted';
|
|
28
|
-
}
|
|
29
|
-
type RunCompleteHook = (ctx: RunCompleteHookContext) => Promise<void> | void;
|
|
30
|
-
interface TraceEmitterOptions {
|
|
31
|
-
runId?: string;
|
|
32
|
-
/** Inject a clock for deterministic tests. */
|
|
33
|
-
now?: () => number;
|
|
34
|
-
/** Inject an id generator for deterministic tests. */
|
|
35
|
-
id?: () => string;
|
|
36
|
-
/**
|
|
37
|
-
* Hooks fired after `endRun` / `abortRun` writes the final run state.
|
|
38
|
-
* Designed for trace-analyst auto-execution, integrity assertions, and
|
|
39
|
-
* outbound notifications. Hooks run sequentially in the order supplied.
|
|
40
|
-
*
|
|
41
|
-
* By default a hook that throws is swallowed and logged as a `note` event
|
|
42
|
-
* on the run — auto-orchestration must not crash the underlying flow.
|
|
43
|
-
* Set `hookErrors: 'throw'` to propagate.
|
|
44
|
-
*/
|
|
45
|
-
onRunComplete?: RunCompleteHook[];
|
|
46
|
-
/** `'swallow'` (default) | `'throw'`. */
|
|
47
|
-
hookErrors?: 'swallow' | 'throw';
|
|
48
|
-
}
|
|
49
|
-
declare class TraceEmitter {
|
|
50
|
-
private store;
|
|
51
|
-
private stack;
|
|
52
|
-
private _runId;
|
|
53
|
-
private now;
|
|
54
|
-
private id;
|
|
55
|
-
private hooks;
|
|
56
|
-
private hookErrors;
|
|
57
|
-
constructor(store: TraceStore, options?: TraceEmitterOptions);
|
|
58
|
-
get runId(): string;
|
|
59
|
-
get traceStore(): TraceStore;
|
|
60
|
-
/** Append a hook after construction (e.g. attach the trace analyst). */
|
|
61
|
-
addRunCompleteHook(hook: RunCompleteHook): void;
|
|
62
|
-
/**
|
|
63
|
-
* Begin a Run.
|
|
64
|
-
*
|
|
65
|
-
* `scenarioId` is required on the persisted Run shape — every Run downstream
|
|
66
|
-
* gets a non-empty scenarioId so filters and aggregations stay simple — but
|
|
67
|
-
* the INPUT here accepts it as optional. When omitted, startRun substitutes
|
|
68
|
-
* a sensible default (`run.layer ?? run.tags?.['kind'] ?? 'runtime'`) so
|
|
69
|
-
* runtime / operator / meta-eval runs that have no curated-scenario corpus
|
|
70
|
-
* to anchor to don't have to invent placeholder strings at the call site.
|
|
71
|
-
*/
|
|
72
|
-
startRun(run: Omit<Run, 'runId' | 'scenarioId' | 'startedAt' | 'status'> & {
|
|
73
|
-
scenarioId?: string;
|
|
74
|
-
}): Promise<Run>;
|
|
75
|
-
endRun(outcome?: RunOutcome): Promise<void>;
|
|
76
|
-
abortRun(reason: string): Promise<void>;
|
|
77
|
-
private runHooks;
|
|
78
|
-
span<S extends Span = Span>(init: {
|
|
79
|
-
kind: SpanKind;
|
|
80
|
-
name: string;
|
|
81
|
-
parentSpanId?: string;
|
|
82
|
-
attributes?: Record<string, unknown>;
|
|
83
|
-
} & Partial<Omit<S, 'spanId' | 'runId' | 'startedAt' | 'kind' | 'name'>>): Promise<SpanHandle<S>>;
|
|
84
|
-
private handle;
|
|
85
|
-
private pop;
|
|
86
|
-
llm(init: Omit<LlmSpan, 'spanId' | 'runId' | 'kind' | 'startedAt'>): Promise<SpanHandle<LlmSpan>>;
|
|
87
|
-
tool(init: Omit<ToolSpan, 'spanId' | 'runId' | 'kind' | 'startedAt'>): Promise<SpanHandle<ToolSpan>>;
|
|
88
|
-
retrieval(init: Omit<RetrievalSpan, 'spanId' | 'runId' | 'kind' | 'startedAt'>): Promise<SpanHandle<RetrievalSpan>>;
|
|
89
|
-
recordJudge(verdict: Omit<JudgeSpan, 'spanId' | 'runId' | 'kind' | 'startedAt' | 'endedAt'>): Promise<JudgeSpan>;
|
|
90
|
-
sandbox(init: Omit<SandboxSpan, 'spanId' | 'runId' | 'kind' | 'startedAt'>): Promise<SpanHandle<SandboxSpan>>;
|
|
91
|
-
emit(event: {
|
|
92
|
-
kind: EventKind;
|
|
93
|
-
spanId?: string;
|
|
94
|
-
payload?: Record<string, unknown>;
|
|
95
|
-
}): Promise<TraceEvent>;
|
|
96
|
-
recordBudget(entry: Omit<BudgetLedgerEntry, 'runId' | 'timestamp'> & {
|
|
97
|
-
timestamp?: number;
|
|
98
|
-
}): Promise<BudgetLedgerEntry>;
|
|
99
|
-
recordArtifact(artifact: Omit<Artifact, 'artifactId' | 'runId'>): Promise<Artifact>;
|
|
100
|
-
/**
|
|
101
|
-
* Runs `fn` inside a span; auto-ends on success, auto-fails on throw.
|
|
102
|
-
* Returns the fn's return value. Use this for the 95% case.
|
|
103
|
-
*/
|
|
104
|
-
within<T>(init: Parameters<TraceEmitter['span']>[0], fn: (handle: SpanHandle) => Promise<T>): Promise<T>;
|
|
105
|
-
}
|
|
106
|
-
/** Helper to build an LLM span handle args object from a provider-shaped response. */
|
|
107
|
-
declare function llmSpanFromProvider(args: {
|
|
108
|
-
name?: string;
|
|
109
|
-
model: string;
|
|
110
|
-
messages: Message[];
|
|
111
|
-
output: string;
|
|
112
|
-
usage?: {
|
|
113
|
-
inputTokens?: number;
|
|
114
|
-
outputTokens?: number;
|
|
115
|
-
cachedTokens?: number;
|
|
116
|
-
reasoningTokens?: number;
|
|
117
|
-
};
|
|
118
|
-
costUsd?: number;
|
|
119
|
-
finishReason?: string;
|
|
120
|
-
}): Omit<LlmSpan, 'spanId' | 'runId' | 'kind' | 'startedAt'>;
|
|
121
|
-
|
|
122
|
-
export { type RunCompleteHook as R, type SpanHandle as S, TraceEmitter as T, type RunCompleteHookContext as a, type TraceEmitterOptions as b, llmSpanFromProvider as l };
|
|
@@ -1,74 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* Error taxonomy for `@tangle-network/agent-eval`.
|
|
3
|
-
*
|
|
4
|
-
* Every error this package throws as part of its *public contract* extends
|
|
5
|
-
* `AgentEvalError`. Consumers can pattern-match by `instanceof <Subclass>` or
|
|
6
|
-
* by the stable string `code` carried on the base class.
|
|
7
|
-
*
|
|
8
|
-
* The codes are stable across minor versions; new codes can be added, but
|
|
9
|
-
* existing codes never change meaning. New subclasses are non-breaking.
|
|
10
|
-
*
|
|
11
|
-
* Internal invariant guards (`throw new Error('this should never happen')`)
|
|
12
|
-
* remain plain `Error`s on purpose — they're programmer-mistake assertions,
|
|
13
|
-
* not consumer-catchable contract failures.
|
|
14
|
-
*/
|
|
15
|
-
type AgentEvalErrorCode = 'validation' | 'not_found' | 'config' | 'capture_integrity' | 'judge' | 'verification' | 'replay' | 'backend_integrity' | 'profile_matrix';
|
|
16
|
-
/**
|
|
17
|
-
* Base class for every contract error this package throws — carries the stable
|
|
18
|
-
* string `code` taxonomy so consumers can `instanceof`-match or switch on `code`.
|
|
19
|
-
*/
|
|
20
|
-
declare class AgentEvalError extends Error {
|
|
21
|
-
/** Stable string code. Survives minification; safe to switch on. */
|
|
22
|
-
readonly code: AgentEvalErrorCode;
|
|
23
|
-
constructor(code: AgentEvalErrorCode, message: string, options?: {
|
|
24
|
-
cause?: unknown;
|
|
25
|
-
});
|
|
26
|
-
}
|
|
27
|
-
/** Caller passed invalid arguments (out of range, mutually-exclusive options, bad shape). */
|
|
28
|
-
declare class ValidationError extends AgentEvalError {
|
|
29
|
-
constructor(message: string, options?: {
|
|
30
|
-
cause?: unknown;
|
|
31
|
-
});
|
|
32
|
-
}
|
|
33
|
-
/** A named resource (run, span, rubric, scenario, dataset row, route) does not exist. */
|
|
34
|
-
declare class NotFoundError extends AgentEvalError {
|
|
35
|
-
constructor(message: string, options?: {
|
|
36
|
-
cause?: unknown;
|
|
37
|
-
});
|
|
38
|
-
}
|
|
39
|
-
/** Configuration missing or malformed (`HOME` unset, required image not supplied, env var absent). */
|
|
40
|
-
declare class ConfigError extends AgentEvalError {
|
|
41
|
-
constructor(message: string, options?: {
|
|
42
|
-
cause?: unknown;
|
|
43
|
-
});
|
|
44
|
-
}
|
|
45
|
-
/**
|
|
46
|
-
* A run is missing the artifacts a launch-grade check requires:
|
|
47
|
-
* raw HTTP capture absent, no LLM spans, route assertion failed, run-end
|
|
48
|
-
* assertion tripped. Block ship on this; do not catch and move on.
|
|
49
|
-
*/
|
|
50
|
-
declare class CaptureIntegrityError extends AgentEvalError {
|
|
51
|
-
constructor(message: string, options?: {
|
|
52
|
-
cause?: unknown;
|
|
53
|
-
});
|
|
54
|
-
}
|
|
55
|
-
/** A judge call failed in a way that's not retryable: schema parse failure, bad rubric, conflicting dimensions. */
|
|
56
|
-
declare class JudgeError extends AgentEvalError {
|
|
57
|
-
constructor(message: string, options?: {
|
|
58
|
-
cause?: unknown;
|
|
59
|
-
});
|
|
60
|
-
}
|
|
61
|
-
/** A verifier signalled a hard failure (compile, test, schema) — distinct from a low judge score. */
|
|
62
|
-
declare class VerificationError extends AgentEvalError {
|
|
63
|
-
constructor(message: string, options?: {
|
|
64
|
-
cause?: unknown;
|
|
65
|
-
});
|
|
66
|
-
}
|
|
67
|
-
/** Replay cache cannot satisfy a request: miss with no fallback, sink lacks list(), unsupported URL. */
|
|
68
|
-
declare class ReplayError extends AgentEvalError {
|
|
69
|
-
constructor(message: string, options?: {
|
|
70
|
-
cause?: unknown;
|
|
71
|
-
});
|
|
72
|
-
}
|
|
73
|
-
|
|
74
|
-
export { AgentEvalError as A, CaptureIntegrityError as C, JudgeError as J, NotFoundError as N, ReplayError as R, ValidationError as V, ConfigError as a, type AgentEvalErrorCode as b, VerificationError as c };
|
|
@@ -1,76 +0,0 @@
|
|
|
1
|
-
import { R as Run, S as Span, e as TraceEvent, F as FailureClass } from './schema-B3Q3l9Z_.js';
|
|
2
|
-
import { T as TraceStore } from './store-DGqD0Pyo.js';
|
|
3
|
-
|
|
4
|
-
/**
|
|
5
|
-
* Failure taxonomy — canonical classes + a default classifier.
|
|
6
|
-
*
|
|
7
|
-
* Every failed run should end up in a named class. The classifier here
|
|
8
|
-
* is rule-based (fast, deterministic); an LLM fallback can be added by
|
|
9
|
-
* the consumer for novel cases and trained into the rule base over time.
|
|
10
|
-
*
|
|
11
|
-
* Consumers call `classifyFailure(run, spans, events)` and persist the
|
|
12
|
-
* returned class as `Run.outcome.failureClass`.
|
|
13
|
-
*/
|
|
14
|
-
|
|
15
|
-
interface FailureContext {
|
|
16
|
-
run: Run;
|
|
17
|
-
spans: Span[];
|
|
18
|
-
events: TraceEvent[];
|
|
19
|
-
}
|
|
20
|
-
interface FailureClassification {
|
|
21
|
-
failureClass: FailureClass;
|
|
22
|
-
reason: string;
|
|
23
|
-
triggerSpanId?: string;
|
|
24
|
-
triggerEventId?: string;
|
|
25
|
-
}
|
|
26
|
-
/** Ordered rules — first match wins. */
|
|
27
|
-
interface FailureRule {
|
|
28
|
-
id: string;
|
|
29
|
-
match: (ctx: FailureContext) => {
|
|
30
|
-
failureClass: FailureClass;
|
|
31
|
-
reason: string;
|
|
32
|
-
triggerSpanId?: string;
|
|
33
|
-
triggerEventId?: string;
|
|
34
|
-
} | null;
|
|
35
|
-
}
|
|
36
|
-
declare const DEFAULT_RULES: FailureRule[];
|
|
37
|
-
/** Classify the failure mode of a run using an ordered rule list. */
|
|
38
|
-
declare function classifyFailure(ctx: FailureContext, rules?: FailureRule[]): FailureClassification;
|
|
39
|
-
|
|
40
|
-
/**
|
|
41
|
-
* FailureClusterView — groups failed runs by (failureClass, triggerTool,
|
|
42
|
-
* argHash-prefix) so weekly reviews can prioritize the top-N clusters.
|
|
43
|
-
*
|
|
44
|
-
* Each cluster includes: N runs, scenarios affected, representative
|
|
45
|
-
* error message, a proposed mitigation hint (rule → action table).
|
|
46
|
-
*/
|
|
47
|
-
|
|
48
|
-
interface FailureCluster {
|
|
49
|
-
failureClass: FailureClass;
|
|
50
|
-
/** Tool name when the trigger was a tool span, else undefined. */
|
|
51
|
-
toolName?: string;
|
|
52
|
-
/** First 16 chars of argHash — clusters similar args. */
|
|
53
|
-
argPrefix?: string;
|
|
54
|
-
/**
|
|
55
|
-
* Source dimension when the trigger was a judge span (e.g. `'format'`,
|
|
56
|
-
* `'safety'`, `'correctness'`). Lets cross-template aggregators
|
|
57
|
-
* group failures by the dimension that fired without overloading
|
|
58
|
-
* `argPrefix`. Optional — clusters without this field deserialize cleanly.
|
|
59
|
-
*/
|
|
60
|
-
dimension?: string;
|
|
61
|
-
runCount: number;
|
|
62
|
-
scenarioIds: string[];
|
|
63
|
-
exampleError?: string;
|
|
64
|
-
exampleRunId: string;
|
|
65
|
-
}
|
|
66
|
-
interface FailureClusterReport {
|
|
67
|
-
clusters: FailureCluster[];
|
|
68
|
-
totalFailures: number;
|
|
69
|
-
totalRuns: number;
|
|
70
|
-
}
|
|
71
|
-
declare function failureClusterView(store: TraceStore, options?: {
|
|
72
|
-
rules?: FailureRule[];
|
|
73
|
-
minClusterSize?: number;
|
|
74
|
-
}): Promise<FailureClusterReport>;
|
|
75
|
-
|
|
76
|
-
export { DEFAULT_RULES as D, type FailureClusterReport as F, type FailureCluster as a, type FailureClassification as b, type FailureContext as c, type FailureRule as d, classifyFailure as e, failureClusterView as f };
|