@tangle-network/agent-eval 0.117.1 → 0.118.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (143) hide show
  1. package/CHANGELOG.md +30 -0
  2. package/dist/analyst/index.d.ts +2772 -21
  3. package/dist/analyst/index.js +7 -6
  4. package/dist/analyst/index.js.map +1 -1
  5. package/dist/belief-state/index.d.ts +706 -10
  6. package/dist/belief-state/index.js +2 -1
  7. package/dist/belief-state/index.js.map +1 -1
  8. package/dist/benchmarks/index.d.ts +958 -14
  9. package/dist/benchmarks/index.js +12 -10
  10. package/dist/builder-eval/index.d.ts +449 -4
  11. package/dist/builder-eval/index.js +4 -3
  12. package/dist/builder-eval/index.js.map +1 -1
  13. package/dist/campaign/index.d.ts +4275 -73
  14. package/dist/campaign/index.js +12 -10
  15. package/dist/{chunk-VF3XSYTI.js → chunk-33JA4TFA.js} +6 -6
  16. package/dist/{chunk-4JLWXDYA.js → chunk-3EHHMC6E.js} +2 -2
  17. package/dist/{chunk-CCZIVI3F.js → chunk-BFW56GTT.js} +2 -2
  18. package/dist/{chunk-YZPO4UHR.js → chunk-FTUMG2U7.js} +124 -149
  19. package/dist/chunk-FTUMG2U7.js.map +1 -0
  20. package/dist/{chunk-E4BUPP7Z.js → chunk-HKUCJ437.js} +38 -66
  21. package/dist/chunk-HKUCJ437.js.map +1 -0
  22. package/dist/{chunk-HZHNRYHK.js → chunk-K6N6XJJX.js} +2 -2
  23. package/dist/chunk-KSDQVPLR.js +286 -0
  24. package/dist/chunk-KSDQVPLR.js.map +1 -0
  25. package/dist/chunk-MA6HLL3S.js +65 -0
  26. package/dist/chunk-MA6HLL3S.js.map +1 -0
  27. package/dist/{chunk-MGEHEHSN.js → chunk-OIIMMLRB.js} +11 -11
  28. package/dist/{chunk-DXZRATT5.js → chunk-OYZAPX5G.js} +3 -3
  29. package/dist/chunk-PXE2VKMX.js +140 -0
  30. package/dist/chunk-PXE2VKMX.js.map +1 -0
  31. package/dist/{chunk-JSJZ4PJ6.js → chunk-Q442S5AS.js} +17 -17
  32. package/dist/{chunk-ODVOOEWQ.js → chunk-QBRSJK47.js} +2 -2
  33. package/dist/{chunk-S2F4J57L.js → chunk-QKEGNI5B.js} +77 -32
  34. package/dist/chunk-QKEGNI5B.js.map +1 -0
  35. package/dist/{chunk-5UF54T55.js → chunk-S3UZOQ5Y.js} +34 -7
  36. package/dist/chunk-S3UZOQ5Y.js.map +1 -0
  37. package/dist/{chunk-FQNLDL4D.js → chunk-SVH2ANFD.js} +136 -4
  38. package/dist/chunk-SVH2ANFD.js.map +1 -0
  39. package/dist/{chunk-GQCZRZ7L.js → chunk-U5CHZ5M3.js} +9 -9
  40. package/dist/{chunk-TVVP3ZZQ.js → chunk-VQMK5FMP.js} +2 -1
  41. package/dist/{chunk-TVVP3ZZQ.js.map → chunk-VQMK5FMP.js.map} +1 -1
  42. package/dist/chunk-WDHBCA3M.js +31 -0
  43. package/dist/chunk-WDHBCA3M.js.map +1 -0
  44. package/dist/{chunk-HQPHZGL6.js → chunk-YLMUS4MM.js} +9 -9
  45. package/dist/{chunk-LQUTGLOZ.js → chunk-ZET2UAYW.js} +15 -65
  46. package/dist/chunk-ZET2UAYW.js.map +1 -0
  47. package/dist/contract/index.d.ts +4012 -38
  48. package/dist/contract/index.js +56 -29
  49. package/dist/contract/index.js.map +1 -1
  50. package/dist/control.d.ts +1013 -9
  51. package/dist/control.js +4 -3
  52. package/dist/fuzz.d.ts +194 -4
  53. package/dist/fuzz.js +3 -3
  54. package/dist/hosted/index.d.ts +498 -17
  55. package/dist/index.d.ts +10992 -1288
  56. package/dist/index.js +101 -80
  57. package/dist/index.js.map +1 -1
  58. package/dist/matrix/index.d.ts +139 -4
  59. package/dist/meta-eval/index.d.ts +862 -15
  60. package/dist/meta-eval/index.js +2 -1
  61. package/dist/meta-eval/index.js.map +1 -1
  62. package/dist/multishot/index.d.ts +214 -14
  63. package/dist/openapi.json +1 -1
  64. package/dist/pipelines/index.d.ts +392 -7
  65. package/dist/pipelines/index.js +5 -3
  66. package/dist/pipelines/index.js.map +1 -1
  67. package/dist/reporting.d.ts +1277 -17
  68. package/dist/rl.d.ts +2359 -28
  69. package/dist/rl.js +6 -5
  70. package/dist/rl.js.map +1 -1
  71. package/dist/storyboard/index.d.ts +86 -1
  72. package/dist/trace-attributes.d.ts +16 -0
  73. package/dist/trace-attributes.js +32 -0
  74. package/dist/trace-attributes.js.map +1 -0
  75. package/dist/traces.d.ts +1978 -697
  76. package/dist/traces.js +53 -32
  77. package/dist/wire/index.d.ts +655 -9
  78. package/docs/insight-report.md +44 -0
  79. package/package.json +9 -3
  80. package/dist/adversarial-B7loGVVX.d.ts +0 -19
  81. package/dist/analyst-C8HHvfJp.d.ts +0 -88
  82. package/dist/analyze-runs--2x39HZ7.d.ts +0 -81
  83. package/dist/baseline-DKq3gJpP.d.ts +0 -141
  84. package/dist/calibration-C8MTS7cw.d.ts +0 -101
  85. package/dist/chunk-5UF54T55.js.map +0 -1
  86. package/dist/chunk-E4BUPP7Z.js.map +0 -1
  87. package/dist/chunk-FQNLDL4D.js.map +0 -1
  88. package/dist/chunk-LQUTGLOZ.js.map +0 -1
  89. package/dist/chunk-S2F4J57L.js.map +0 -1
  90. package/dist/chunk-YZPO4UHR.js.map +0 -1
  91. package/dist/code-agent-session-CjZsVd19.d.ts +0 -87
  92. package/dist/control-6vuGfmDH.d.ts +0 -258
  93. package/dist/cost-ledger-DWy3XdJc.d.ts +0 -183
  94. package/dist/dataset-NENEzRgk.d.ts +0 -115
  95. package/dist/default-registry-DaK8b3fv.d.ts +0 -155
  96. package/dist/emitter-CjD7vUwv.d.ts +0 -122
  97. package/dist/errors-oeQrLqXC.d.ts +0 -74
  98. package/dist/failure-cluster-DOAcSJ87.d.ts +0 -76
  99. package/dist/feedback-trajectory-BUnM58xL.d.ts +0 -348
  100. package/dist/gepa-eESocoDi.d.ts +0 -642
  101. package/dist/index-PdX4VnPA.d.ts +0 -423
  102. package/dist/insight-report-DY4nDW9Q.d.ts +0 -310
  103. package/dist/integrity-DqlBiLyK.d.ts +0 -81
  104. package/dist/judge-calibration-7C-IDmKr.d.ts +0 -145
  105. package/dist/kind-factory-ClZmO25A.d.ts +0 -171
  106. package/dist/llm-client-qoDd18Qz.d.ts +0 -289
  107. package/dist/multi-layer-verifier-BsqKuLyN.d.ts +0 -150
  108. package/dist/off-policy-DiwuKKg7.d.ts +0 -132
  109. package/dist/outcome-store-rnXLEqSn.d.ts +0 -63
  110. package/dist/policy-edit-wG9uFEFm.d.ts +0 -455
  111. package/dist/pre-registration-BWQhJ3vz.d.ts +0 -761
  112. package/dist/provenance-DpjwyseI.d.ts +0 -541
  113. package/dist/query-CF7PG61p.d.ts +0 -35
  114. package/dist/raw-provider-sink-C46HDghv.d.ts +0 -132
  115. package/dist/release-report-C8G2i5Xi.d.ts +0 -236
  116. package/dist/researcher-C8XyxQsu.d.ts +0 -387
  117. package/dist/rubric-predictive-validity-p49lLVrE.d.ts +0 -105
  118. package/dist/run-record-BDH49H2E.d.ts +0 -360
  119. package/dist/runtime-trajectory-DGBIUt4B.d.ts +0 -49
  120. package/dist/schema-B3Q3l9Z_.d.ts +0 -201
  121. package/dist/semantic-concept-judge-CXnPEJbf.d.ts +0 -723
  122. package/dist/sequential-5iSVfzl2.d.ts +0 -139
  123. package/dist/series-convergence-D5OWMBg6.d.ts +0 -33
  124. package/dist/statistics-KUnG73jH.d.ts +0 -494
  125. package/dist/storage-DrX3v_5B.d.ts +0 -50
  126. package/dist/store-C1YxJDEK.d.ts +0 -248
  127. package/dist/store-DGqD0Pyo.d.ts +0 -116
  128. package/dist/summary-report-C5bKFfm-.d.ts +0 -445
  129. package/dist/test-graded-scenario-B0ybnPY7.d.ts +0 -166
  130. package/dist/types-BSw1rOUB.d.ts +0 -634
  131. package/dist/types-BUxNaJ8c.d.ts +0 -108
  132. package/dist/types-BkfcQnxV.d.ts +0 -313
  133. package/dist/verdict-C9MlYujm.d.ts +0 -35
  134. /package/dist/{chunk-VF3XSYTI.js.map → chunk-33JA4TFA.js.map} +0 -0
  135. /package/dist/{chunk-4JLWXDYA.js.map → chunk-3EHHMC6E.js.map} +0 -0
  136. /package/dist/{chunk-CCZIVI3F.js.map → chunk-BFW56GTT.js.map} +0 -0
  137. /package/dist/{chunk-HZHNRYHK.js.map → chunk-K6N6XJJX.js.map} +0 -0
  138. /package/dist/{chunk-MGEHEHSN.js.map → chunk-OIIMMLRB.js.map} +0 -0
  139. /package/dist/{chunk-DXZRATT5.js.map → chunk-OYZAPX5G.js.map} +0 -0
  140. /package/dist/{chunk-JSJZ4PJ6.js.map → chunk-Q442S5AS.js.map} +0 -0
  141. /package/dist/{chunk-ODVOOEWQ.js.map → chunk-QBRSJK47.js.map} +0 -0
  142. /package/dist/{chunk-GQCZRZ7L.js.map → chunk-U5CHZ5M3.js.map} +0 -0
  143. /package/dist/{chunk-HQPHZGL6.js.map → chunk-YLMUS4MM.js.map} +0 -0
@@ -1,155 +0,0 @@
1
- import { AxAIService } from '@ax-llm/ax';
2
- import { T as TraceAnalystKindSpec } from './kind-factory-ClZmO25A.js';
3
- import { A as Analyst, a as AnalystContext, b as AnalystRunSummary, c as AnalystFinding, d as AnalystRunResult, C as ChatClient, e as AnalystRunInputs, f as AnalystRunEvent } from './policy-edit-wG9uFEFm.js';
4
-
5
- /**
6
- * AnalystRegistry — orchestrate N analysts against one run.
7
- *
8
- * Owns three responsibilities and only three:
9
- * 1. Registration — ids must be unique; bad registrations fail loudly
10
- * at register-time, not run-time.
11
- * 2. Routing — each analyst declares its `inputKind`; the registry
12
- * picks the matching field from AnalystRunInputs and skips the
13
- * analyst with a logged reason if it's missing.
14
- * 3. Isolation — one analyst's exception MUST NOT stop other analysts.
15
- * Failed analysts produce zero findings + a 'failed' summary row.
16
- *
17
- * Cross-cutting concerns (telemetry, error → finding conversion, cost
18
- * ingestion, storage rotation) live in `AnalystHooks`. Budget shaping
19
- * (equal split vs weighted vs custom) lives in `BudgetPolicy`. Both
20
- * have sensible defaults; consumers override only what they need.
21
- */
22
-
23
- interface AnalystHooks {
24
- /** Before analyze() — last chance to mutate ctx (e.g. inject tags, override budget). */
25
- onBeforeAnalyze?(args: {
26
- analyst: Analyst;
27
- ctx: AnalystContext;
28
- runId: string;
29
- }): void | Promise<void>;
30
- /** After every analyst (ok | failed | skipped). Use for telemetry, ingestion, rotation. */
31
- onAfterAnalyze?(args: {
32
- analyst: Analyst;
33
- summary: AnalystRunSummary;
34
- findings: AnalystFinding[];
35
- runId: string;
36
- }): void | Promise<void>;
37
- /**
38
- * On analyst exception. Hook MAY return findings to convert the
39
- * error into structured findings; the summary still reports 'failed'.
40
- * Return void to keep the default empty-findings behavior.
41
- */
42
- onError?(args: {
43
- analyst: Analyst;
44
- error: Error;
45
- runId: string;
46
- }): AnalystFinding[] | undefined | Promise<AnalystFinding[] | undefined>;
47
- /** Once after registry.run() completes. Use for final aggregation, persistence. */
48
- onComplete?(args: {
49
- result: AnalystRunResult;
50
- }): void | Promise<void>;
51
- }
52
- interface BudgetPolicy {
53
- /** Overall USD cap across the registry.run(). */
54
- totalUsd?: number;
55
- /** Per-analyst weight for the default allocator. Missing ids get weight 1. */
56
- weights?: Record<string, number>;
57
- /**
58
- * Custom allocator — receives the analyst, remaining/total budget, and
59
- * the count of analysts that will run. Returns the per-analyst budget
60
- * (or undefined to leave it uncapped). Overrides weights when set.
61
- */
62
- allocate?: (args: {
63
- analyst: Analyst;
64
- totalUsd: number | undefined;
65
- remainingUsd: number | undefined;
66
- runningCount: number;
67
- }) => number | undefined;
68
- }
69
- interface AnalystRegistryOptions {
70
- /** Shared chat client passed to every LLM analyst via AnalystContext. */
71
- chat?: ChatClient;
72
- /** Logger callback. Defaults to a no-op. */
73
- log?: (msg: string, fields?: Record<string, unknown>) => void;
74
- /** Hooks invoked around analyze() — observability + customization seam. */
75
- hooks?: AnalystHooks;
76
- /** Default budget when run() doesn't override. */
77
- defaultBudget?: BudgetPolicy;
78
- }
79
- interface RegistryRunOpts {
80
- /** Restrict to a subset of registered analysts by id. */
81
- only?: string[];
82
- /** Skip these analysts even if registered. Useful for cheap iteration. */
83
- skip?: string[];
84
- /** Budget policy — totalUsd + optional weights/allocator. Falls back to options.defaultBudget. */
85
- budget?: BudgetPolicy;
86
- /** Wall-clock cap. Analysts SHOULD honor `ctx.deadlineMs`. */
87
- timeoutMs?: number;
88
- /** Abort signal — forwarded into every analyst's context. */
89
- signal?: AbortSignal;
90
- /** Tags echoed into AnalystContext.tags — useful for tracking environment/version in findings. */
91
- tags?: Record<string, string>;
92
- /**
93
- * Prior-run findings made available as retrieval context to every
94
- * analyst via `ctx.priorFindings`. The registry forwards the slice
95
- * whose `analyst_id` matches each registered analyst so a kind sees
96
- * only its own history. Pass `{ '*': findings }` to broadcast to
97
- * every analyst (useful for cross-kind chaining where the improvement
98
- * analyst consumes upstream failure findings).
99
- */
100
- priorFindings?: ReadonlyArray<AnalystFinding> | Record<string, ReadonlyArray<AnalystFinding>>;
101
- }
102
- declare class AnalystRegistry {
103
- private readonly analysts;
104
- private readonly options;
105
- constructor(options?: AnalystRegistryOptions);
106
- register(analyst: Analyst): void;
107
- list(): ReadonlyArray<{
108
- id: string;
109
- description: string;
110
- version: string;
111
- cost: Analyst['cost'];
112
- }>;
113
- run(runId: string, inputs: AnalystRunInputs, runOpts?: RegistryRunOpts): Promise<AnalystRunResult>;
114
- /**
115
- * Streaming counterpart to `run()`. Emits `AnalystRunEvent` values
116
- * in real time — `run-started`, then per-analyst `skipped` /
117
- * `started` / `completed`, then a terminal `run-completed` whose
118
- * payload is the full `AnalystRunResult`. UIs use this to render
119
- * progress; persistence consumers use `run()` and read the result.
120
- *
121
- * Hooks (`onBeforeAnalyze` / `onAfterAnalyze` / `onError` /
122
- * `onComplete`) fire as before — streaming is additive, not a hook
123
- * replacement.
124
- */
125
- runStream(runId: string, inputs: AnalystRunInputs, runOpts?: RegistryRunOpts): AsyncGenerator<AnalystRunEvent, void, void>;
126
- private selectAnalysts;
127
- private routeInput;
128
- }
129
-
130
- /**
131
- * `buildDefaultAnalystRegistry` — the canonical analyst suite, so consumers
132
- * stop hand-wiring `new AnalystRegistry()` + per-kind `createTraceAnalystKind`.
133
- *
134
- * The deterministic `behavioralAnalyst` is ALWAYS registered (it needs no
135
- * model and is model-agnostic by construction). The agentic RLM kinds are
136
- * registered only when an `ai` service is supplied — so a caller with no LLM
137
- * still gets the full behavioral/efficiency diagnosis, and the substrate's
138
- * "any model (including no model)" guarantee holds at the suite level.
139
- */
140
-
141
- interface DefaultAnalystRegistryOptions {
142
- /** Ax service for the agentic RLM kinds. Omit → only the deterministic analyst. */
143
- ai?: AxAIService;
144
- /** Model for the agentic kinds (falls back to the ai service default). */
145
- model?: string;
146
- /** Which agentic kinds to register when `ai` is present. Default = the shipped suite. */
147
- kinds?: readonly TraceAnalystKindSpec[];
148
- /** Set false to omit the deterministic behavioral analyst (default: include). */
149
- includeBehavioral?: boolean;
150
- /** Forwarded to the AnalystRegistry constructor (signal, tags, priorFindings). */
151
- registry?: AnalystRegistryOptions;
152
- }
153
- declare function buildDefaultAnalystRegistry(opts?: DefaultAnalystRegistryOptions): AnalystRegistry;
154
-
155
- export { AnalystRegistry as A, type BudgetPolicy as B, type DefaultAnalystRegistryOptions as D, type RegistryRunOpts as R, type AnalystHooks as a, type AnalystRegistryOptions as b, buildDefaultAnalystRegistry as c };
@@ -1,122 +0,0 @@
1
- import { a as RunOutcome, R as Run, S as Span, b as SpanKind, L as LlmSpan, T as ToolSpan, c as RetrievalSpan, J as JudgeSpan, d as SandboxSpan, E as EventKind, e as TraceEvent, B as BudgetLedgerEntry, A as Artifact, M as Message } from './schema-B3Q3l9Z_.js';
2
- import { T as TraceStore } from './store-DGqD0Pyo.js';
3
-
4
- /**
5
- * TraceEmitter — hierarchical span builder that auto-parents using an
6
- * internal stack. One emitter per Run; emitters do NOT share state.
7
- *
8
- * Convenience methods (`llm`, `tool`, `retrieval`, `judge`, `sandbox`)
9
- * return a `SpanHandle` with `.end()` / `.fail()` so callers don't
10
- * have to thread spanIds manually. For async workflows that can't use
11
- * the stack (e.g. fan-out parallel calls), pass `parentSpanId`
12
- * explicitly.
13
- */
14
-
15
- interface SpanHandle<S extends Span = Span> {
16
- span: S;
17
- end(patch?: Partial<S>): Promise<void>;
18
- fail(error: string | Error, patch?: Partial<S>): Promise<void>;
19
- }
20
- interface RunCompleteHookContext {
21
- runId: string;
22
- emitter: TraceEmitter;
23
- store: TraceStore;
24
- /** Outcome the caller passed to `endRun` (undefined for `abortRun`). */
25
- outcome?: RunOutcome;
26
- /** Final run status. */
27
- status: 'completed' | 'failed' | 'aborted';
28
- }
29
- type RunCompleteHook = (ctx: RunCompleteHookContext) => Promise<void> | void;
30
- interface TraceEmitterOptions {
31
- runId?: string;
32
- /** Inject a clock for deterministic tests. */
33
- now?: () => number;
34
- /** Inject an id generator for deterministic tests. */
35
- id?: () => string;
36
- /**
37
- * Hooks fired after `endRun` / `abortRun` writes the final run state.
38
- * Designed for trace-analyst auto-execution, integrity assertions, and
39
- * outbound notifications. Hooks run sequentially in the order supplied.
40
- *
41
- * By default a hook that throws is swallowed and logged as a `note` event
42
- * on the run — auto-orchestration must not crash the underlying flow.
43
- * Set `hookErrors: 'throw'` to propagate.
44
- */
45
- onRunComplete?: RunCompleteHook[];
46
- /** `'swallow'` (default) | `'throw'`. */
47
- hookErrors?: 'swallow' | 'throw';
48
- }
49
- declare class TraceEmitter {
50
- private store;
51
- private stack;
52
- private _runId;
53
- private now;
54
- private id;
55
- private hooks;
56
- private hookErrors;
57
- constructor(store: TraceStore, options?: TraceEmitterOptions);
58
- get runId(): string;
59
- get traceStore(): TraceStore;
60
- /** Append a hook after construction (e.g. attach the trace analyst). */
61
- addRunCompleteHook(hook: RunCompleteHook): void;
62
- /**
63
- * Begin a Run.
64
- *
65
- * `scenarioId` is required on the persisted Run shape — every Run downstream
66
- * gets a non-empty scenarioId so filters and aggregations stay simple — but
67
- * the INPUT here accepts it as optional. When omitted, startRun substitutes
68
- * a sensible default (`run.layer ?? run.tags?.['kind'] ?? 'runtime'`) so
69
- * runtime / operator / meta-eval runs that have no curated-scenario corpus
70
- * to anchor to don't have to invent placeholder strings at the call site.
71
- */
72
- startRun(run: Omit<Run, 'runId' | 'scenarioId' | 'startedAt' | 'status'> & {
73
- scenarioId?: string;
74
- }): Promise<Run>;
75
- endRun(outcome?: RunOutcome): Promise<void>;
76
- abortRun(reason: string): Promise<void>;
77
- private runHooks;
78
- span<S extends Span = Span>(init: {
79
- kind: SpanKind;
80
- name: string;
81
- parentSpanId?: string;
82
- attributes?: Record<string, unknown>;
83
- } & Partial<Omit<S, 'spanId' | 'runId' | 'startedAt' | 'kind' | 'name'>>): Promise<SpanHandle<S>>;
84
- private handle;
85
- private pop;
86
- llm(init: Omit<LlmSpan, 'spanId' | 'runId' | 'kind' | 'startedAt'>): Promise<SpanHandle<LlmSpan>>;
87
- tool(init: Omit<ToolSpan, 'spanId' | 'runId' | 'kind' | 'startedAt'>): Promise<SpanHandle<ToolSpan>>;
88
- retrieval(init: Omit<RetrievalSpan, 'spanId' | 'runId' | 'kind' | 'startedAt'>): Promise<SpanHandle<RetrievalSpan>>;
89
- recordJudge(verdict: Omit<JudgeSpan, 'spanId' | 'runId' | 'kind' | 'startedAt' | 'endedAt'>): Promise<JudgeSpan>;
90
- sandbox(init: Omit<SandboxSpan, 'spanId' | 'runId' | 'kind' | 'startedAt'>): Promise<SpanHandle<SandboxSpan>>;
91
- emit(event: {
92
- kind: EventKind;
93
- spanId?: string;
94
- payload?: Record<string, unknown>;
95
- }): Promise<TraceEvent>;
96
- recordBudget(entry: Omit<BudgetLedgerEntry, 'runId' | 'timestamp'> & {
97
- timestamp?: number;
98
- }): Promise<BudgetLedgerEntry>;
99
- recordArtifact(artifact: Omit<Artifact, 'artifactId' | 'runId'>): Promise<Artifact>;
100
- /**
101
- * Runs `fn` inside a span; auto-ends on success, auto-fails on throw.
102
- * Returns the fn's return value. Use this for the 95% case.
103
- */
104
- within<T>(init: Parameters<TraceEmitter['span']>[0], fn: (handle: SpanHandle) => Promise<T>): Promise<T>;
105
- }
106
- /** Helper to build an LLM span handle args object from a provider-shaped response. */
107
- declare function llmSpanFromProvider(args: {
108
- name?: string;
109
- model: string;
110
- messages: Message[];
111
- output: string;
112
- usage?: {
113
- inputTokens?: number;
114
- outputTokens?: number;
115
- cachedTokens?: number;
116
- reasoningTokens?: number;
117
- };
118
- costUsd?: number;
119
- finishReason?: string;
120
- }): Omit<LlmSpan, 'spanId' | 'runId' | 'kind' | 'startedAt'>;
121
-
122
- export { type RunCompleteHook as R, type SpanHandle as S, TraceEmitter as T, type RunCompleteHookContext as a, type TraceEmitterOptions as b, llmSpanFromProvider as l };
@@ -1,74 +0,0 @@
1
- /**
2
- * Error taxonomy for `@tangle-network/agent-eval`.
3
- *
4
- * Every error this package throws as part of its *public contract* extends
5
- * `AgentEvalError`. Consumers can pattern-match by `instanceof <Subclass>` or
6
- * by the stable string `code` carried on the base class.
7
- *
8
- * The codes are stable across minor versions; new codes can be added, but
9
- * existing codes never change meaning. New subclasses are non-breaking.
10
- *
11
- * Internal invariant guards (`throw new Error('this should never happen')`)
12
- * remain plain `Error`s on purpose — they're programmer-mistake assertions,
13
- * not consumer-catchable contract failures.
14
- */
15
- type AgentEvalErrorCode = 'validation' | 'not_found' | 'config' | 'capture_integrity' | 'judge' | 'verification' | 'replay' | 'backend_integrity' | 'profile_matrix';
16
- /**
17
- * Base class for every contract error this package throws — carries the stable
18
- * string `code` taxonomy so consumers can `instanceof`-match or switch on `code`.
19
- */
20
- declare class AgentEvalError extends Error {
21
- /** Stable string code. Survives minification; safe to switch on. */
22
- readonly code: AgentEvalErrorCode;
23
- constructor(code: AgentEvalErrorCode, message: string, options?: {
24
- cause?: unknown;
25
- });
26
- }
27
- /** Caller passed invalid arguments (out of range, mutually-exclusive options, bad shape). */
28
- declare class ValidationError extends AgentEvalError {
29
- constructor(message: string, options?: {
30
- cause?: unknown;
31
- });
32
- }
33
- /** A named resource (run, span, rubric, scenario, dataset row, route) does not exist. */
34
- declare class NotFoundError extends AgentEvalError {
35
- constructor(message: string, options?: {
36
- cause?: unknown;
37
- });
38
- }
39
- /** Configuration missing or malformed (`HOME` unset, required image not supplied, env var absent). */
40
- declare class ConfigError extends AgentEvalError {
41
- constructor(message: string, options?: {
42
- cause?: unknown;
43
- });
44
- }
45
- /**
46
- * A run is missing the artifacts a launch-grade check requires:
47
- * raw HTTP capture absent, no LLM spans, route assertion failed, run-end
48
- * assertion tripped. Block ship on this; do not catch and move on.
49
- */
50
- declare class CaptureIntegrityError extends AgentEvalError {
51
- constructor(message: string, options?: {
52
- cause?: unknown;
53
- });
54
- }
55
- /** A judge call failed in a way that's not retryable: schema parse failure, bad rubric, conflicting dimensions. */
56
- declare class JudgeError extends AgentEvalError {
57
- constructor(message: string, options?: {
58
- cause?: unknown;
59
- });
60
- }
61
- /** A verifier signalled a hard failure (compile, test, schema) — distinct from a low judge score. */
62
- declare class VerificationError extends AgentEvalError {
63
- constructor(message: string, options?: {
64
- cause?: unknown;
65
- });
66
- }
67
- /** Replay cache cannot satisfy a request: miss with no fallback, sink lacks list(), unsupported URL. */
68
- declare class ReplayError extends AgentEvalError {
69
- constructor(message: string, options?: {
70
- cause?: unknown;
71
- });
72
- }
73
-
74
- export { AgentEvalError as A, CaptureIntegrityError as C, JudgeError as J, NotFoundError as N, ReplayError as R, ValidationError as V, ConfigError as a, type AgentEvalErrorCode as b, VerificationError as c };
@@ -1,76 +0,0 @@
1
- import { R as Run, S as Span, e as TraceEvent, F as FailureClass } from './schema-B3Q3l9Z_.js';
2
- import { T as TraceStore } from './store-DGqD0Pyo.js';
3
-
4
- /**
5
- * Failure taxonomy — canonical classes + a default classifier.
6
- *
7
- * Every failed run should end up in a named class. The classifier here
8
- * is rule-based (fast, deterministic); an LLM fallback can be added by
9
- * the consumer for novel cases and trained into the rule base over time.
10
- *
11
- * Consumers call `classifyFailure(run, spans, events)` and persist the
12
- * returned class as `Run.outcome.failureClass`.
13
- */
14
-
15
- interface FailureContext {
16
- run: Run;
17
- spans: Span[];
18
- events: TraceEvent[];
19
- }
20
- interface FailureClassification {
21
- failureClass: FailureClass;
22
- reason: string;
23
- triggerSpanId?: string;
24
- triggerEventId?: string;
25
- }
26
- /** Ordered rules — first match wins. */
27
- interface FailureRule {
28
- id: string;
29
- match: (ctx: FailureContext) => {
30
- failureClass: FailureClass;
31
- reason: string;
32
- triggerSpanId?: string;
33
- triggerEventId?: string;
34
- } | null;
35
- }
36
- declare const DEFAULT_RULES: FailureRule[];
37
- /** Classify the failure mode of a run using an ordered rule list. */
38
- declare function classifyFailure(ctx: FailureContext, rules?: FailureRule[]): FailureClassification;
39
-
40
- /**
41
- * FailureClusterView — groups failed runs by (failureClass, triggerTool,
42
- * argHash-prefix) so weekly reviews can prioritize the top-N clusters.
43
- *
44
- * Each cluster includes: N runs, scenarios affected, representative
45
- * error message, a proposed mitigation hint (rule → action table).
46
- */
47
-
48
- interface FailureCluster {
49
- failureClass: FailureClass;
50
- /** Tool name when the trigger was a tool span, else undefined. */
51
- toolName?: string;
52
- /** First 16 chars of argHash — clusters similar args. */
53
- argPrefix?: string;
54
- /**
55
- * Source dimension when the trigger was a judge span (e.g. `'format'`,
56
- * `'safety'`, `'correctness'`). Lets cross-template aggregators
57
- * group failures by the dimension that fired without overloading
58
- * `argPrefix`. Optional — clusters without this field deserialize cleanly.
59
- */
60
- dimension?: string;
61
- runCount: number;
62
- scenarioIds: string[];
63
- exampleError?: string;
64
- exampleRunId: string;
65
- }
66
- interface FailureClusterReport {
67
- clusters: FailureCluster[];
68
- totalFailures: number;
69
- totalRuns: number;
70
- }
71
- declare function failureClusterView(store: TraceStore, options?: {
72
- rules?: FailureRule[];
73
- minClusterSize?: number;
74
- }): Promise<FailureClusterReport>;
75
-
76
- export { DEFAULT_RULES as D, type FailureClusterReport as F, type FailureCluster as a, type FailureClassification as b, type FailureContext as c, type FailureRule as d, classifyFailure as e, failureClusterView as f };