@tangle-network/agent-eval 0.115.3 → 0.116.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (68) hide show
  1. package/CHANGELOG.md +24 -0
  2. package/dist/analyst/index.d.ts +8 -10
  3. package/dist/analyst/index.js +27 -22
  4. package/dist/analyst/index.js.map +1 -1
  5. package/dist/{analyst-C8HHvfJp.d.ts → analyst-CFBc14Wc.d.ts} +1 -1
  6. package/dist/{analyze-runs-BYHg6Irm.d.ts → analyze-runs-0rz_m29H.d.ts} +3 -3
  7. package/dist/belief-state/index.d.ts +3 -3
  8. package/dist/benchmarks/index.d.ts +7 -3
  9. package/dist/benchmarks/index.js +4 -4
  10. package/dist/campaign/index.d.ts +212 -23
  11. package/dist/campaign/index.js +19 -4
  12. package/dist/{chunk-ADYLPOSX.js → chunk-3274WNK7.js} +427 -93
  13. package/dist/chunk-3274WNK7.js.map +1 -0
  14. package/dist/{chunk-I6LVHOV3.js → chunk-7GKEAIAD.js} +2 -2
  15. package/dist/{chunk-5S5NJ63F.js → chunk-CIUOICJT.js} +747 -2
  16. package/dist/chunk-CIUOICJT.js.map +1 -0
  17. package/dist/{chunk-KG4TD7EQ.js → chunk-GSW3OBHK.js} +1283 -181
  18. package/dist/chunk-GSW3OBHK.js.map +1 -0
  19. package/dist/chunk-MPHTT5HE.js +74 -0
  20. package/dist/chunk-MPHTT5HE.js.map +1 -0
  21. package/dist/{chunk-WSBUZMBU.js → chunk-NBSS5NDZ.js} +3 -3
  22. package/dist/{code-agent-session-D-g04tcy.d.ts → code-agent-session-CdxteG0y.d.ts} +1 -1
  23. package/dist/contract/index.d.ts +19 -19
  24. package/dist/contract/index.js +5 -3
  25. package/dist/contract/index.js.map +1 -1
  26. package/dist/{control-CcBiAEnn.d.ts → control-DbcDxouY.d.ts} +1 -1
  27. package/dist/control.d.ts +2 -2
  28. package/dist/{default-registry-DltpYR5u.d.ts → default-registry-DDfv22MQ.d.ts} +2 -1
  29. package/dist/{gepa-dne9JDPL.d.ts → gepa-CQelRtuC.d.ts} +10 -8
  30. package/dist/hosted/index.d.ts +8 -4
  31. package/dist/{index-BTEpx9He.d.ts → index-DbCXJfZ1.d.ts} +2 -2
  32. package/dist/index.d.ts +27 -30
  33. package/dist/index.js +26 -20
  34. package/dist/index.js.map +1 -1
  35. package/dist/{insight-report-IwwvqZZv.d.ts → insight-report-oMVxDTxl.d.ts} +1 -1
  36. package/dist/{integrity-qemeBAyx.d.ts → integrity-C6PZ73iC.d.ts} +1 -1
  37. package/dist/kind-factory-DWOvXjR_.d.ts +171 -0
  38. package/dist/meta-eval/index.d.ts +2 -2
  39. package/dist/multishot/index.d.ts +6 -2
  40. package/dist/openapi.json +1 -1
  41. package/dist/policy-edit-Clb2v6Oa.d.ts +708 -0
  42. package/dist/{pre-registration-D8h7ZxNL.d.ts → pre-registration--vU0mMtD.d.ts} +4 -4
  43. package/dist/{provenance-Bibyg1U9.d.ts → provenance-BbVagC68.d.ts} +26 -14
  44. package/dist/{release-report-CCtzajxP.d.ts → release-report-CamNDe90.d.ts} +2 -2
  45. package/dist/reporting.d.ts +4 -4
  46. package/dist/{researcher-Dq-EtpbE.d.ts → researcher-Dwbo_Fxx.d.ts} +5 -5
  47. package/dist/rl.d.ts +11 -9
  48. package/dist/{rubric-predictive-validity-DYTLjGWu.d.ts → rubric-predictive-validity-BIdf9h4R.d.ts} +1 -1
  49. package/dist/{run-record-B7RTi_ix.d.ts → run-record-CZmcpWPo.d.ts} +1 -1
  50. package/dist/{runtime-trajectory-Dws7Kpgi.d.ts → runtime-trajectory-CC0jx9ql.d.ts} +1 -1
  51. package/dist/{semantic-concept-judge-DxJmRkyJ.d.ts → semantic-concept-judge-CKjePUMh.d.ts} +3 -3
  52. package/dist/{store-C1YxJDEK.d.ts → store-9cAScOcb.d.ts} +132 -1
  53. package/dist/{summary-report-BJ5aNwZ1.d.ts → summary-report-DTNgQycC.d.ts} +1 -1
  54. package/dist/traces.d.ts +6 -8
  55. package/dist/{types-C5gJrOVT.d.ts → types-Ca_63YSD.d.ts} +59 -2
  56. package/docs/design/loop-taxonomy.md +1 -2
  57. package/package.json +1 -1
  58. package/dist/chunk-5S5NJ63F.js.map +0 -1
  59. package/dist/chunk-ADYLPOSX.js.map +0 -1
  60. package/dist/chunk-KG4TD7EQ.js.map +0 -1
  61. package/dist/chunk-QMXXSNC4.js +0 -761
  62. package/dist/chunk-QMXXSNC4.js.map +0 -1
  63. package/dist/kind-factory-DcNg13sZ.d.ts +0 -508
  64. package/dist/llm-client-DyqEH4jH.d.ts +0 -265
  65. package/dist/policy-edit-RLn8GWof.d.ts +0 -103
  66. package/dist/raw-provider-sink-C46HDghv.d.ts +0 -132
  67. /package/dist/{chunk-I6LVHOV3.js.map → chunk-7GKEAIAD.js.map} +0 -0
  68. /package/dist/{chunk-WSBUZMBU.js.map → chunk-NBSS5NDZ.js.map} +0 -0
@@ -1,4 +1,4 @@
1
- import { G as GainDistributionBin, P as ParetoFigureSpec } from './summary-report-BJ5aNwZ1.js';
1
+ import { G as GainDistributionBin, P as ParetoFigureSpec } from './summary-report-DTNgQycC.js';
2
2
  import { C as ContinuousAgreement } from './judge-calibration-7C-IDmKr.js';
3
3
 
4
4
  /**
@@ -1,5 +1,5 @@
1
1
  import { C as CaptureIntegrityError } from './errors-oeQrLqXC.js';
2
- import { R as RawProviderSink } from './raw-provider-sink-C46HDghv.js';
2
+ import { R as RawProviderSink } from './store-9cAScOcb.js';
3
3
  import { T as TraceStore } from './store-BsVi7ncX.js';
4
4
 
5
5
  /**
@@ -0,0 +1,171 @@
1
+ import { AxAIService, AxFunction } from '@ax-llm/ax';
2
+ import { T as TraceAnalysisStore } from './store-9cAScOcb.js';
3
+ import { z } from 'zod';
4
+ import { h as AnalystCost, b as AnalystContext, A as Analyst } from './policy-edit-Clb2v6Oa.js';
5
+
6
+ /**
7
+ * Typed Ax output for analyst findings.
8
+ *
9
+ * Replaces the legacy `findings:string[]` pattern (where every bullet
10
+ * became a flat-severity `AnalystFinding`) with a structured object
11
+ * array. Ax binds the field as `findings:json[]` so the provider emits
12
+ * native structured output; at the kind-factory boundary we Zod-validate
13
+ * each emitted finding so malformed rows fail loud instead of being
14
+ * silently lifted with default severity.
15
+ *
16
+ * Why not `f.object().array()` directly in the signature? The Ax
17
+ * signature string `question:string -> findings:json[]` already lets
18
+ * the provider emit JSON arrays. A Zod boundary is required either
19
+ * way (the provider can return any JSON), and Zod gives us a single
20
+ * validation surface independent of which Ax version is installed.
21
+ */
22
+
23
+ declare const ANALYST_SEVERITIES: readonly ["critical", "high", "medium", "low", "info"];
24
+ declare const RawAnalystFindingSchema: z.ZodObject<{
25
+ severity: z.ZodEnum<{
26
+ info: "info";
27
+ critical: "critical";
28
+ medium: "medium";
29
+ low: "low";
30
+ high: "high";
31
+ }>;
32
+ claim: z.ZodString;
33
+ subject: z.ZodOptional<z.ZodString>;
34
+ evidence_uri: z.ZodString;
35
+ evidence_excerpt: z.ZodOptional<z.ZodString>;
36
+ confidence: z.ZodNumber;
37
+ rationale: z.ZodOptional<z.ZodString>;
38
+ recommended_action: z.ZodOptional<z.ZodString>;
39
+ }, z.core.$strict>;
40
+ type RawAnalystFinding = z.infer<typeof RawAnalystFindingSchema>;
41
+ /**
42
+ * Description embedded into the actor prompt so the LLM knows what
43
+ * shape to emit. Kept here so kinds share one source of truth rather
44
+ * than restating the schema in every prompt.
45
+ */
46
+ declare const RAW_FINDING_SCHEMA_PROMPT = "Each finding MUST be a JSON object with these fields:\n - severity: one of \"critical\" | \"high\" | \"medium\" | \"low\" | \"info\"\n - claim: one-sentence statement (max 2000 chars)\n - subject?: the routing locus this finding is about. It MUST be one of the exact subject forms listed in this kind's instructions above (e.g. `system-prompt:<section>`, `agent-knowledge:wiki:<slug>`, `tool-doc:<tool>`). A free phrase, a bare noun, or any form not in that list is REJECTED at parse time and the finding is discarded \u2014 omit subject entirely rather than guess a form.\n - evidence_uri: REQUIRED, never blank. Exactly one of \"span://<trace_id>/<span_id>\" (trace evidence), \"artifact://<relative-path>\" (files), \"metric://<name>\" (named scalars) \u2014 ALWAYS cite a real id surfaced by the tools. If you have no citable id, do not emit the finding.\n - evidence_excerpt?: short quote (<=2000 chars) from the cited span/artifact\n - confidence: number 0..1 \u2014 0.9+ when backed by exact quotes, 0.6-0.8 for inferred patterns, <0.5 for speculative\n - rationale?: one or two sentences explaining the reasoning\n - recommended_action?: concrete change phrased as an imperative (\"Add ...\", \"Replace ...\", \"Stop ...\") \u2014 omit when the finding is purely descriptive\n\nEmit an empty array when the question has no findings to report. Do not fabricate evidence.";
47
+ /**
48
+ * Validate one row emitted by the LLM. Returns the typed finding on
49
+ * success; returns `null` and logs the reason on failure so the kind
50
+ * factory can skip-and-count rather than abort the whole analyst run.
51
+ */
52
+ declare function parseRawFinding(row: unknown, log?: (msg: string, fields?: Record<string, unknown>) => void): RawAnalystFinding | null;
53
+
54
+ /**
55
+ * Analyst-kind factory — the typed way to define trace analysts.
56
+ *
57
+ * A "kind" is a specialized analyst whose actor prompt, tool subset,
58
+ * and Ax recursion config target one failure-mode lens (failure-mode
59
+ * classification, knowledge gap discovery, knowledge poisoning, recursive
60
+ * self-improvement, ...). Kinds emit findings in the typed `RawAnalystFinding`
61
+ * shape via a JSON-array Ax output; the factory validates each row with
62
+ * Zod and lifts it into `AnalystFinding[]` with no shape guessing.
63
+ *
64
+ * Composition rules:
65
+ * - Each kind owns its actor description. No generic "answer this
66
+ * question" prompt — the prompt names the failure lens.
67
+ * - Each kind picks a narrow tool subset from `ANALYST_TOOL_GROUPS`.
68
+ * A kind that never needs full-trace dumps can drop `viewTrace` /
69
+ * `viewSpans` and stay cheap.
70
+ * - Each kind declares its recursion + parallelism budget. Discovery-
71
+ * heavy kinds (failure-mode) get higher `maxDepth`; lens kinds
72
+ * (poisoning) usually stay at 0 since they have a tighter brief.
73
+ *
74
+ * Optimizer hook: kinds may declare `goldens` — labeled examples used
75
+ * by `AxMiPRO` / `AxBootstrapFewShot` / `AxGEPA` to fit the actor
76
+ * description programmatically. Stored on the kind, not the registry,
77
+ * because the right metric is kind-specific.
78
+ */
79
+
80
+ /**
81
+ * Per-kind specification. The factory turns this into a regular
82
+ * `Analyst<TraceAnalysisStore>` ready for `AnalystRegistry.register()`.
83
+ */
84
+ interface TraceAnalystKindSpec {
85
+ /** Stable id. Appears in finding_id, telemetry, and registry exclusions. */
86
+ id: string;
87
+ /** One-sentence description shown in `registry.list()`. */
88
+ description: string;
89
+ /** Coarse classification stamped on every emitted finding (`failure-mode`, `knowledge-gap`, ...). */
90
+ area: string;
91
+ /** Bump on any breaking change to the actor prompt or output schema. */
92
+ version: string;
93
+ /** Actor system prompt. Must instruct the LLM to emit `findings` per the schema. */
94
+ actorDescription: string;
95
+ /** Responder system prompt; falls back to a minimal "format the findings" instruction. */
96
+ responderDescription?: string;
97
+ /** Tool functions the actor may call. Pick narrow subsets via `ANALYST_TOOL_GROUPS`. */
98
+ buildTools: (store: TraceAnalysisStore) => AxFunction[];
99
+ /** Recursion budget. `maxDepth: 0` disables subagents. */
100
+ recursion?: {
101
+ maxDepth: number;
102
+ maxParallelSubagents?: number;
103
+ };
104
+ /** Actor turn cap. Default 12. */
105
+ maxTurns?: number;
106
+ /** Runtime char cap. Default 6000. */
107
+ maxRuntimeChars?: number;
108
+ /** Cost classification surfaced in `registry.list()` and budget enforcement. */
109
+ cost: AnalystCost;
110
+ /** Per-finding-row hook — kinds may reject / rewrite before lifting. */
111
+ postProcess?: (row: RawAnalystFinding, ctx: AnalystContext) => RawAnalystFinding | null;
112
+ /** Optional optimizer hook — populated when a kind wants to fit its prompt against labeled examples. */
113
+ goldens?: TraceAnalystGolden[];
114
+ }
115
+ /**
116
+ * One labeled example consumed by Ax optimizers (MIPRO / GEPA / Bootstrap).
117
+ * Each input is the same `{question}` an analyst would receive; `expected`
118
+ * is the ground-truth finding set a fitted prompt should produce on this
119
+ * input. Metric: kind-specific (default: F1 on `finding_id` overlap).
120
+ */
121
+ interface TraceAnalystGolden {
122
+ question: string;
123
+ expected: ReadonlyArray<Omit<RawAnalystFinding, 'confidence'>>;
124
+ }
125
+ interface CreateTraceAnalystKindOpts {
126
+ /** AxAIService bound at registration time. */
127
+ ai: AxAIService;
128
+ /** Optional model override; falls back to the AI service's default. */
129
+ model?: string;
130
+ /** Override the spec's `version` (e.g. when an optimizer has fitted a new prompt). */
131
+ versionSuffix?: string;
132
+ /**
133
+ * Optional two-phase recovery: when the agentic harvest is empty but the
134
+ * actor produced a substantive free-form `report`, extract findings from that
135
+ * prose via a tolerant chat-completions pass (`structureFindings`) — no
136
+ * strict-emission contract, so it works on weak models. Omit to leave the
137
+ * actor's harvest as-is (the report is still surfaced fail-loud either way).
138
+ */
139
+ recovery?: {
140
+ baseUrl: string;
141
+ apiKey?: string;
142
+ model?: string;
143
+ fetchImpl?: typeof fetch;
144
+ };
145
+ }
146
+ /**
147
+ * Build an `Analyst<TraceAnalysisStore>` from a kind spec.
148
+ *
149
+ * Lifts the Ax pipeline once at registration time so the registry
150
+ * gets a stateless analyst. The Ax agent is freshly constructed per
151
+ * `analyze()` call (the agent carries chat-log + usage state we don't
152
+ * want shared across analyst runs).
153
+ */
154
+ declare function createTraceAnalystKind(spec: TraceAnalystKindSpec, opts: CreateTraceAnalystKindOpts): Analyst<TraceAnalysisStore>;
155
+ /**
156
+ * Render a compact prior-findings block the actor reads alongside its
157
+ * brief. Each row is one line so the actor can scan dozens cheaply.
158
+ * The kind's prompt instructs the actor to (a) check whether a new
159
+ * cluster matches a prior `finding_id` (carry the id forward via
160
+ * `id_basis` to keep diffs stable) and (b) raise severity / confidence
161
+ * when a prior finding has reappeared without remediation.
162
+ *
163
+ * Returns the empty string when there are no prior findings — most
164
+ * runs are "first-of-its-kind" and the prompt stays unchanged.
165
+ *
166
+ * Exported for tests + for consumers that build their own actor
167
+ * prompts (e.g. specialized analysts living outside the default kinds).
168
+ */
169
+ declare function renderPriorFindings(prior: AnalystContext['priorFindings']): string;
170
+
171
+ export { ANALYST_SEVERITIES as A, type CreateTraceAnalystKindOpts as C, RAW_FINDING_SCHEMA_PROMPT as R, type TraceAnalystKindSpec as T, type RawAnalystFinding as a, RawAnalystFindingSchema as b, type TraceAnalystGolden as c, createTraceAnalystKind as d, parseRawFinding as p, renderPriorFindings as r };
@@ -1,12 +1,12 @@
1
1
  export { C as CalibrationBin, a as CalibrationOptions, b as CalibrationPair, c as CalibrationReport, d as CorrelationResult, e as CorrelationStudyOptions, f as CorrelationStudyResult, E as EvalMetricSpec, O as OutcomePair, g as calibrationCurve, h as calibrationFromPairs, i as correlationStudy } from '../calibration-Dz8TQV4y.js';
2
2
  export { D as DeploymentOutcome, F as FileSystemOutcomeStore, a as FileSystemOutcomeStoreOptions, I as InMemoryOutcomeStore, O as OutcomeFilter, b as OutcomeStore } from '../outcome-store-rnXLEqSn.js';
3
- export { R as RubricOutcomePair, a as RubricPredictiveValidityInput, b as RubricPredictiveValidityReport, c as RubricRanking, r as rubricPredictiveValidity } from '../rubric-predictive-validity-DYTLjGWu.js';
3
+ export { R as RubricOutcomePair, a as RubricPredictiveValidityInput, b as RubricPredictiveValidityReport, c as RubricRanking, r as rubricPredictiveValidity } from '../rubric-predictive-validity-BIdf9h4R.js';
4
4
  import { C as ContinuousAgreement, a as CalibrationResult, b as ContinuousCalibrationResult, c as CandidateScore, G as GoldenItem } from '../judge-calibration-7C-IDmKr.js';
5
5
  import { S as SeriesConvergenceOptions, a as SeriesConvergenceResult } from '../series-convergence-D5OWMBg6.js';
6
6
  import { C as CorpusAgreementReport } from '../statistics-oUbOJe-S.js';
7
7
  import '../store-BsVi7ncX.js';
8
8
  import '../schema-SGWcK9wa.js';
9
- import '../run-record-B7RTi_ix.js';
9
+ import '../run-record-CZmcpWPo.js';
10
10
  import '@tangle-network/agent-interface';
11
11
  import '../errors-oeQrLqXC.js';
12
12
  import '../types-C7DGg5ex.js';
@@ -1,9 +1,13 @@
1
- import { J as JudgeScore } from '../types-C5gJrOVT.js';
1
+ import { J as JudgeScore } from '../types-Ca_63YSD.js';
2
2
  import { AgentProfile } from '@tangle-network/agent-interface';
3
3
  import { M as MatrixResult } from '../types-BUxNaJ8c.js';
4
- import '../run-record-B7RTi_ix.js';
4
+ import '../policy-edit-Clb2v6Oa.js';
5
+ import '../run-record-CZmcpWPo.js';
5
6
  import '../errors-oeQrLqXC.js';
6
7
  import '../schema-SGWcK9wa.js';
8
+ import '../store-9cAScOcb.js';
9
+ import '../types-C7DGg5ex.js';
10
+ import '@tangle-network/tcloud';
7
11
  import '../verdict-C9MlYujm.js';
8
12
 
9
13
  interface MultishotMessage {
package/dist/openapi.json CHANGED
@@ -2,7 +2,7 @@
2
2
  "openapi": "3.1.0",
3
3
  "info": {
4
4
  "title": "@tangle-network/agent-eval — wire protocol",
5
- "version": "0.115.3",
5
+ "version": "0.116.0",
6
6
  "description": "HTTP and stdio RPC interface to agent-eval. The TypeScript runtime is the source of truth; this spec is the contract that cross-language clients (Python, Rust, Go) generate from.\n\nWire-protocol version: 1.0.0. Bumps on breaking changes to request/response schemas.",
7
7
  "contact": {
8
8
  "name": "Tangle Network",