@tangle-network/agent-eval 0.133.3 → 0.134.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (73) hide show
  1. package/CHANGELOG.md +70 -0
  2. package/dist/analyst/index.d.ts +11 -35
  3. package/dist/analyst/index.d.ts.map +1 -1
  4. package/dist/analyst/index.js +4 -53
  5. package/dist/analyst/index.js.map +1 -1
  6. package/dist/{analyze-runs-BClW9OSe.d.ts → analyze-runs-DMo3Lb_y.d.ts} +4 -4
  7. package/dist/{analyze-runs-BClW9OSe.d.ts.map → analyze-runs-DMo3Lb_y.d.ts.map} +1 -1
  8. package/dist/benchmarks/index.d.ts +1 -1
  9. package/dist/benchmarks/index.js +1 -1
  10. package/dist/{benchmarks-BP9sgMia.js → benchmarks-v5piCeDl.js} +3 -3
  11. package/dist/{benchmarks-BP9sgMia.js.map → benchmarks-v5piCeDl.js.map} +1 -1
  12. package/dist/campaign/index.d.ts +5 -4
  13. package/dist/campaign/index.js +4 -3
  14. package/dist/{campaign--V4ffEKR.js → campaign-DEC_7DLn.js} +2 -2
  15. package/dist/{campaign--V4ffEKR.js.map → campaign-DEC_7DLn.js.map} +1 -1
  16. package/dist/{client-Du7B81wW.d.ts → client-BIyh1RCr.d.ts} +3 -3
  17. package/dist/{client-Du7B81wW.d.ts.map → client-BIyh1RCr.d.ts.map} +1 -1
  18. package/dist/contract/index.d.ts +12 -11
  19. package/dist/contract/index.d.ts.map +1 -1
  20. package/dist/contract/index.js +4 -11
  21. package/dist/contract/index.js.map +1 -1
  22. package/dist/{default-registry-Cl3pHo4n.d.ts → default-registry-Brxr728w.d.ts} +4 -268
  23. package/dist/default-registry-Brxr728w.d.ts.map +1 -0
  24. package/dist/{default-registry-D3T9XbuY.js → default-registry-IjYs7T8l.js} +4 -61
  25. package/dist/default-registry-IjYs7T8l.js.map +1 -0
  26. package/dist/hosted/index.d.ts +2 -2
  27. package/dist/{index-DOqvIJ8I.d.ts → index-BoJNQR6n.d.ts} +4 -3
  28. package/dist/index-BoJNQR6n.d.ts.map +1 -0
  29. package/dist/{index-B5MNN1f1.d.ts → index-C21xKtxu.d.ts} +4 -4
  30. package/dist/{index-B5MNN1f1.d.ts.map → index-C21xKtxu.d.ts.map} +1 -1
  31. package/dist/index.d.ts +13 -12
  32. package/dist/index.d.ts.map +1 -1
  33. package/dist/index.js +190 -20
  34. package/dist/index.js.map +1 -1
  35. package/dist/multishot/index.d.ts +1 -1
  36. package/dist/openapi.json +1 -1
  37. package/dist/proposal-findings-DCawte-y.js +164 -0
  38. package/dist/proposal-findings-DCawte-y.js.map +1 -0
  39. package/dist/{release-report-DKBtegGt.d.ts → release-report-CuULWKyk.d.ts} +2 -2
  40. package/dist/{release-report-DKBtegGt.d.ts.map → release-report-CuULWKyk.d.ts.map} +1 -1
  41. package/dist/reporting.d.ts +2 -2
  42. package/dist/{researcher-BtD5U1Up.d.ts → researcher-DVtruQ9U.d.ts} +2 -2
  43. package/dist/{researcher-BtD5U1Up.d.ts.map → researcher-DVtruQ9U.d.ts.map} +1 -1
  44. package/dist/rl.d.ts +2 -2
  45. package/dist/rl.d.ts.map +1 -1
  46. package/dist/rl.js +14 -1
  47. package/dist/rl.js.map +1 -1
  48. package/dist/{semantic-concept-judge-BypLt6Fw.js → semantic-concept-judge-C0P1VTXD.js} +2 -3
  49. package/dist/{semantic-concept-judge-BypLt6Fw.js.map → semantic-concept-judge-C0P1VTXD.js.map} +1 -1
  50. package/dist/{skill-usage-BaaxFSJR.d.ts → skill-usage-BDQVPIG1.d.ts} +3 -2
  51. package/dist/skill-usage-BDQVPIG1.d.ts.map +1 -0
  52. package/dist/{skillopt-optimization-method-vvJ4bMNI.js → skillopt-optimization-method-BY6vKLJB.js} +51 -30
  53. package/dist/skillopt-optimization-method-BY6vKLJB.js.map +1 -0
  54. package/dist/{skillopt-optimization-method-Dxr8pdZd.d.ts → skillopt-optimization-method-DJ3l4w8W.d.ts} +10 -13
  55. package/dist/skillopt-optimization-method-DJ3l4w8W.d.ts.map +1 -0
  56. package/dist/{summary-report-DyOhItws.d.ts → summary-report-DGp0-_XO.d.ts} +59 -3
  57. package/dist/summary-report-DGp0-_XO.d.ts.map +1 -0
  58. package/dist/types-DVjczBM9.d.ts +276 -0
  59. package/dist/types-DVjczBM9.d.ts.map +1 -0
  60. package/dist/{types-BokuXvOG.d.ts → types-DiWLru6Z.d.ts} +20 -37
  61. package/dist/types-DiWLru6Z.d.ts.map +1 -0
  62. package/docs/campaign-proposers.md +5 -0
  63. package/package.json +1 -1
  64. package/dist/default-registry-Cl3pHo4n.d.ts.map +0 -1
  65. package/dist/default-registry-D3T9XbuY.js.map +0 -1
  66. package/dist/index-DOqvIJ8I.d.ts.map +0 -1
  67. package/dist/run-score-iEEAWiBY.js +0 -41
  68. package/dist/run-score-iEEAWiBY.js.map +0 -1
  69. package/dist/skill-usage-BaaxFSJR.d.ts.map +0 -1
  70. package/dist/skillopt-optimization-method-Dxr8pdZd.d.ts.map +0 -1
  71. package/dist/skillopt-optimization-method-vvJ4bMNI.js.map +0 -1
  72. package/dist/summary-report-DyOhItws.d.ts.map +0 -1
  73. package/dist/types-BokuXvOG.d.ts.map +0 -1
@@ -1,273 +1,9 @@
1
1
  import { c as CostLedgerHandle } from "./cost-ledger-fGS_u_O1.js";
2
- import { a as RunRecord, n as RunCostProvenance, u as RunTokenUsage } from "./run-record-DcObtIGh.js";
3
- import { A as ChatClient, m as JudgeInput } from "./types-Cc3qbqzj.js";
2
+ import { A as ChatClient } from "./types-Cc3qbqzj.js";
4
3
  import { t as TraceAnalysisStore } from "./store-CxJry_cs.js";
4
+ import { c as AnalystRunInputs, i as AnalystFinding, l as AnalystRunResult, n as AnalystContext, p as EvidenceRef, r as AnalystCost, s as AnalystRunEvent, t as Analyst, u as AnalystRunSummary } from "./types-DVjczBM9.js";
5
5
  import { AxAIService, AxFunction } from "@ax-llm/ax";
6
6
  import { z } from "zod";
7
- //#region src/analyst/types.d.ts
8
- /**
9
- * Unified envelope every analyst emits. Schema-versioned so renderers
10
- * and time-series diffs survive future field additions.
11
- */
12
- interface AnalystFinding {
13
- schema_version: '1.0.0';
14
- /**
15
- * Stable hash over identity-defining fields (analyst_id + canonical
16
- * claim + area + optional subject). Two findings from two runs that
17
- * "are the same finding" share this id — that's what `diffFindings`
18
- * uses to compute appeared/disappeared sets across runs.
19
- */
20
- finding_id: string;
21
- analyst_id: string;
22
- produced_at: string;
23
- severity: AnalystSeverity;
24
- /**
25
- * Coarse classification. Renderers group by this. Free-form so
26
- * domain-specific analysts can introduce categories without a
27
- * schema change ('agent-reasoning', 'verification', 'cost',
28
- * 'tool-use', 'safety', 'latency', 'data-quality', ...).
29
- */
30
- area: string;
31
- claim: string;
32
- rationale?: string;
33
- evidence_refs: EvidenceRef[];
34
- recommended_action?: string;
35
- validation_plan?: string;
36
- /** 0..1 — the analyst's own confidence. Not calibrated across analysts. */
37
- confidence: number;
38
- /**
39
- * Optional subject the finding is about — leaf id, agent id, request
40
- * id. Included in finding_id when present so per-subject findings
41
- * diff cleanly across runs.
42
- */
43
- subject?: string;
44
- /** FIREWALL provenance (docs/learning-flywheel.md): true iff this finding was
45
- * lifted from a JUDGE verdict (an acceptance score), not OBSERVED from the
46
- * agent's behavior. A judge-derived finding must NEVER be admitted as a
47
- * steering input — that is the held-out judge leaking into the loop. Set at
48
- * the lift site (createJudgeAdapter); checked by `assertNoJudgeVerdict`.
49
- * Provenance, not evidence presence, is the correct discriminator: an
50
- * evidence-less trace-analyst observation legitimately steers, while a judge
51
- * verdict that happens to cite an artifact must not. */
52
- derived_from_judge?: boolean;
53
- /** Analyst-private extras; renderers ignore unless they know the analyst. */
54
- metadata?: Record<string, unknown>;
55
- }
56
- type AnalystSeverity = 'critical' | 'high' | 'medium' | 'low' | 'info';
57
- interface EvidenceRef {
58
- /**
59
- * Where the evidence lives. `span` and `event` refer to OTLP trace
60
- * elements; `artifact` to a file inside the run's artifact tree;
61
- * `finding` to another AnalystFinding (cross-analyst chaining);
62
- * `metric` to a named scalar reading the renderer knows how to read.
63
- */
64
- kind: 'span' | 'event' | 'artifact' | 'finding' | 'metric';
65
- uri: string;
66
- excerpt?: string;
67
- }
68
- /**
69
- * The discriminator the registry uses to pass the right input.
70
- * `custom` is the escape hatch — analysts that need something else
71
- * (e.g. an embedding cache, a partner SDK handle) read it from
72
- * `AnalystRunInputs.custom[<analyst id>]`.
73
- */
74
- type AnalystInputKind = 'trace-store' | 'artifact-dir' | 'run-record' | 'judge-input' | 'custom';
75
- interface AnalystCost {
76
- /** `deterministic` analysts MUST NOT call the LLM. */
77
- kind: 'deterministic' | 'llm';
78
- /** Optional declared upper bound; the registry can enforce a budget. */
79
- est_usd_per_run?: number;
80
- /** Models the analyst expects to use (informational). */
81
- models?: string[];
82
- /** Maximum post-cancellation wait for provider usage. Model analysts default to 5 seconds. */
83
- settlement_timeout_ms?: number;
84
- }
85
- interface AnalystRequirements {
86
- /** Min number of shots / samples the analyst needs to produce signal. */
87
- min_shots?: number;
88
- /** Capabilities the runtime must supply (e.g. ['network', 'gpu']). */
89
- capabilities?: string[];
90
- }
91
- /**
92
- * What's passed to every analyst call. The registry resolves which
93
- * field the analyst's `inputKind` selects and asserts it's present.
94
- */
95
- interface AnalystRunInputs {
96
- traceStore?: TraceAnalysisStore;
97
- artifactDir?: string;
98
- runRecord?: RunRecord;
99
- judgeInput?: JudgeInput;
100
- /** Keyed by analyst id; populated by callers that registered custom analysts. */
101
- custom?: Record<string, unknown>;
102
- }
103
- interface AnalystContext {
104
- runId: string;
105
- /** Stable correlation id so logs from a single registry.run() share a tag. */
106
- correlationId: string;
107
- /** Enforced wall-clock deadline (epoch ms). */
108
- deadlineMs?: number;
109
- /** Per-analyst USD budget. Analysts MAY check before issuing LLM calls. */
110
- budgetUsd?: number;
111
- /** Shared paid-call account when the analyst runs inside a larger campaign. */
112
- costLedger?: CostLedgerHandle;
113
- /** Attribution phase used when writing to the shared paid-call account. */
114
- costPhase?: string;
115
- /**
116
- * Shared chat client. Analysts that call an LLM go through this so
117
- * the operator picks transport (sandbox-sdk | router | cli-bridge |
118
- * direct-provider | mock) at the registry boundary without touching
119
- * analyst code.
120
- */
121
- chat?: ChatClient;
122
- /**
123
- * Findings from a prior run the operator wants the analyst to see as
124
- * retrieval context. Kinds that take advantage of cross-run memory
125
- * (failure-mode "I saw this cluster last run", knowledge-gap "the wiki
126
- * page I asked for is still missing") render these into the actor's
127
- * working set. Filtering is the operator's job: pass the slice that
128
- * matches the analyst's id, or pass everything and let the kind
129
- * filter. Empty / absent means no cross-run context.
130
- */
131
- priorFindings?: ReadonlyArray<AnalystFinding>;
132
- /**
133
- * Findings emitted by analysts that completed earlier in this registry run.
134
- * This is separate from `priorFindings`: upstream findings are dependency
135
- * context for the current pass, while prior findings are cross-run memory.
136
- * The registry populates this only when `RegistryRunOpts.chainFindings` is on.
137
- */
138
- upstreamFindings?: ReadonlyArray<AnalystFinding>;
139
- /**
140
- * Report metered work independently of findings. This keeps an empty finding
141
- * set from erasing token/cost telemetry. Multiple receipts are accumulated.
142
- */
143
- recordUsage?: (receipt: AnalystUsageReceipt) => void;
144
- /** Free-form runtime tags (env, host, op). Findings can echo these into metadata. */
145
- tags?: Record<string, string>;
146
- /** Logger callback — analysts SHOULD prefer this over console.* for testability. */
147
- log?: (msg: string, fields?: Record<string, unknown>) => void;
148
- /** Optional abort signal. Analysts SHOULD pass it through to LLM calls. */
149
- signal?: AbortSignal;
150
- }
151
- /**
152
- * The minimal contract. Concrete analysts can refine `TInput` so
153
- * implementations stay type-safe (e.g. a trace analyst's `TInput` is
154
- * `TraceAnalysisStore`); the registry passes the right field from
155
- * `AnalystRunInputs` based on `inputKind`.
156
- */
157
- interface Analyst<TInput = unknown> {
158
- /** Stable identifier — appears in finding_id, telemetry, and registry exclusion lists. */
159
- readonly id: string;
160
- /** Human-readable. One sentence. */
161
- readonly description: string;
162
- readonly inputKind: AnalystInputKind;
163
- readonly cost: AnalystCost;
164
- readonly requires?: AnalystRequirements;
165
- /** Bump on breaking changes to claim wording or area so old finding_ids don't collide. */
166
- readonly version: string;
167
- analyze(input: TInput, ctx: AnalystContext): Promise<AnalystFinding[]>;
168
- }
169
- /** Metered work performed by one analyst call. */
170
- interface AnalystUsageReceipt {
171
- /** Number of model-usage records observed at the provider boundary. */
172
- calls: number | null;
173
- /** Null when the provider did not return token accounting. */
174
- tokens: RunTokenUsage | null;
175
- /** Observed, estimated, or explicitly uncaptured dollar cost. */
176
- cost: RunCostProvenance;
177
- /** Known lower bound when one or more calls have uncaptured cost. */
178
- knownCostUsd?: number;
179
- }
180
- /**
181
- * Compute the stable finding_id from the identity-defining fields.
182
- * Default implementation hashes {analyst_id, area, subject, normalized claim}.
183
- * Analysts that emit findings whose claim text varies per run (timestamps,
184
- * counts) SHOULD either: (a) pass an explicit `id_basis` to fix the hash,
185
- * or (b) move the variable part into `rationale`/`metadata` and keep the
186
- * `claim` static.
187
- */
188
- declare function computeFindingId(input: {
189
- analyst_id: string;
190
- area: string;
191
- subject?: string;
192
- claim: string;
193
- /** Override the claim for hashing — use when the displayed claim has run-specific bits. */
194
- id_basis?: string;
195
- }): string;
196
- /**
197
- * Convenience factory: produce a fully-formed AnalystFinding with the
198
- * id computed automatically. Analyst code stays terse.
199
- */
200
- declare function makeFinding(init: Omit<AnalystFinding, 'schema_version' | 'finding_id' | 'produced_at'> & {
201
- id_basis?: string;
202
- produced_at?: string;
203
- }): AnalystFinding;
204
- interface AnalystRunSummary {
205
- analyst_id: string;
206
- status: 'ok' | 'skipped' | 'failed';
207
- /** Why skipped — missing input, budget exceeded, capability unmet. */
208
- reason?: string;
209
- findings_count: number;
210
- latency_ms: number;
211
- /** Additive model usage and cost provenance for this analyst. */
212
- usage: AnalystUsageReceipt;
213
- /** When `status='failed'`: the error class + message, never the full stack. */
214
- error?: {
215
- class: string;
216
- message: string;
217
- };
218
- }
219
- interface AnalystRunResult {
220
- run_id: string;
221
- correlation_id: string;
222
- started_at: string;
223
- ended_at: string;
224
- findings: AnalystFinding[];
225
- per_analyst: AnalystRunSummary[];
226
- /** Total LLM cost in USD across all analysts in this registry.run(). */
227
- total_cost_usd: number;
228
- /**
229
- * Provenance for `total_cost_usd`. When uncaptured, the numeric field is only
230
- * the known subtotal and must not be treated as the run's total spend.
231
- */
232
- total_cost_provenance?: RunCostProvenance;
233
- }
234
- /**
235
- * Events emitted by `AnalystRegistry.runStream(...)` in real time as
236
- * the registry executes. UIs subscribe via `for await (const ev of
237
- * registry.runStream(...))`; `registry.run(...)` is a thin collector
238
- * over the same stream, so the two surfaces share their invariants.
239
- *
240
- * Per-finding events are intentionally omitted — analyzers are batch
241
- * operations (an Ax actor returns the full `findings:json[]` at the
242
- * end of the responder), so streaming inside one analyst would only
243
- * emit partial JSON consumers can't render. The kind-completion event
244
- * is the right granularity; subscribers wanting per-finding rendering
245
- * iterate `event.findings` themselves.
246
- */
247
- type AnalystRunEvent = {
248
- type: 'run-started';
249
- run_id: string;
250
- correlation_id: string;
251
- started_at: string;
252
- /** The ordered list of analyst ids the registry will run. */
253
- analyst_ids: ReadonlyArray<string>;
254
- } | {
255
- type: 'analyst-skipped';
256
- summary: AnalystRunSummary;
257
- } | {
258
- type: 'analyst-started';
259
- analyst_id: string;
260
- started_at: string;
261
- } | {
262
- type: 'analyst-completed';
263
- /** `summary.status` is `'ok'` for clean completion or `'failed'` for thrown analysts. */
264
- summary: AnalystRunSummary;
265
- findings: ReadonlyArray<AnalystFinding>;
266
- } | {
267
- type: 'run-completed';
268
- result: AnalystRunResult;
269
- };
270
- //#endregion
271
7
  //#region src/analyst/finding-signature.d.ts
272
8
  declare const ANALYST_SEVERITIES: readonly ['critical', 'high', 'medium', 'low', 'info'];
273
9
  declare const RawAnalystEvidenceSchema: z.ZodObject<{
@@ -536,5 +272,5 @@ interface DefaultAnalystRegistryOptions {
536
272
  }
537
273
  declare function buildDefaultAnalystRegistry(opts?: DefaultAnalystRegistryOptions): AnalystRegistry;
538
274
  //#endregion
539
- export { AnalystRunResult as A, AnalystContext as C, AnalystRequirements as D, AnalystInputKind as E, computeFindingId as F, makeFinding as I, AnalystSeverity as M, AnalystUsageReceipt as N, AnalystRunEvent as O, EvidenceRef as P, Analyst as S, AnalystFinding as T, RawAnalystEvidenceSchema as _, AnalystRegistryOptions as a, evidenceRefsFromRawFinding as b, CreateTraceAnalystKindOpts as c, createTraceAnalystKind as d, renderPriorFindings as f, RawAnalystEvidence as g, RAW_FINDING_SCHEMA_PROMPT as h, AnalystRegistry as i, AnalystRunSummary as j, AnalystRunInputs as k, TraceAnalystGolden as l, ANALYST_SEVERITIES as m, buildDefaultAnalystRegistry as n, BudgetPolicy as o, renderUpstreamFindings as p, AnalystHooks as r, RegistryRunOpts as s, DefaultAnalystRegistryOptions as t, TraceAnalystKindSpec as u, RawAnalystFinding as v, AnalystCost as w, parseRawFinding as x, RawAnalystFindingSchema as y };
540
- //# sourceMappingURL=default-registry-Cl3pHo4n.d.ts.map
275
+ export { RawAnalystEvidenceSchema as _, AnalystRegistryOptions as a, evidenceRefsFromRawFinding as b, CreateTraceAnalystKindOpts as c, createTraceAnalystKind as d, renderPriorFindings as f, RawAnalystEvidence as g, RAW_FINDING_SCHEMA_PROMPT as h, AnalystRegistry as i, TraceAnalystGolden as l, ANALYST_SEVERITIES as m, buildDefaultAnalystRegistry as n, BudgetPolicy as o, renderUpstreamFindings as p, AnalystHooks as r, RegistryRunOpts as s, DefaultAnalystRegistryOptions as t, TraceAnalystKindSpec as u, RawAnalystFinding as v, parseRawFinding as x, RawAnalystFindingSchema as y };
276
+ //# sourceMappingURL=default-registry-Brxr728w.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"default-registry-Brxr728w.d.ts","names":[],"sources":["../src/analyst/finding-signature.ts","../src/analyst/kind-factory.ts","../src/analyst/registry.ts","../src/analyst/default-registry.ts"],"mappings":";;;;;;;cAmBa;cAEA,0BAAwB,EAAA;;;GAK1B,EAAA,KAAA;KAEC,qBAAqB,EAAE,aAAa;cAiBnC,yBAAuB,EAAA;;;;;;;;;;;;;;;;;GAKzB,EAAA,KAAA;KAEC,oBAAoB,EAAE,aAAa;;;;;;cAOlC;;iBAYG,2BAA2B,SAAS,oBAAoB;iBAQxD,gBACd,cACA,OAAO,aAAa,SAAS,mCAC5B;;;;;;;UCnCc;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA,aAAa,OAAO,uBAAuB;;EAE3C;IAAe;IAAkB;;;EAEjC;;EAEA;;EAEA;;EAEA,MAAM;;EAEN,eAAe,KAAK,mBAAmB,KAAK,mBAAmB;;EAE/D;;EAEA,UAAU;;;;;;;;UASK;EACf;EACA,UAAU,cAAc,KAAK;;UAGd;;EAEf,IAAI;;EAEJ;;EAEA;;;;;;;;EAQA;IAAa;IAAiB;IAAiB;IAAgB,mBAAmB;;;EAElF;;;;;;;;;;iBAWc,uBACd,MAAM,sBACN,MAAM,6BACL,QAAQ;;;;;;;;;;;;;;;iBAgRK,oBAAoB,OAAO;;iBAyB3B,uBAAuB,UAAU;;;UC3XhC;;EAEf,iBAAiB;IACf,SAAS;IACT,KAAK;IACL;aACS;;EAEX,gBAAgB;IACd,SAAS;IACT,SAAS;IACT,UAAU;IACV;aACS;;;;;;EAMX,SAAS;IACP,SAAS;IACT,OAAO;IACP;MACE,+BAA+B,QAAQ;;EAE3C,YAAY;IAAQ,QAAQ;aAA4B;;UAGzC;;EAEf;;EAEA,UAAU;;;;;;;EAOV,YAAY;IACV,SAAS;IACT;IACA;IACA;;;UAIa;;EAEf,OAAO;;EAEP,OAAO,aAAa,SAAS;;EAE7B,QAAQ;;EAER,gBAAgB;;UAGD;;EAEf;;EAEA;;EAEA,SAAS;;EAET;;EAEA,SAAS;;EAET,aAAa;;EAEb;;EAEA,OAAO;;;;;;;;;EASP,gBAAgB,cAAc,kBAAkB,eAAe,cAAc;;;;;;EAM7E;;cAGW;mBACM;mBACA;EAEjB,YAAY,UAAS;EAIrB,SAAS,SAAS;EAmBlB,QAAQ;IACN;IACA;IACA;IACA,MAAM;;EAUF,IACJ,eACA,QAAQ,kBACR,UAAS,kBACR,QAAQ;;;;;;;;;;;;EAoBJ,UACL,eACA,QAAQ,kBACR,UAAS,kBACR,eAAe;UA0OV;UAaA;;;;UC5aO;;EAEf,KAAK;;EAEL;;EAEA,iBAAiB;;EAEjB;;EAEA,WAAW;;iBAGG,4BACd,OAAM,gCACL"}
@@ -3,10 +3,10 @@ import { f as maximumChargeForLlmRequest, l as costReceiptFromLlm, n as LlmClien
3
3
  import { LLM_CONTEXT_TOKENS, LLM_INPUT_TOKEN_ATTR_KEYS, LLM_OUTPUT_TOKEN_ATTR_KEYS, TOOL_NAME_ATTR_KEYS } from "./trace-attributes.js";
4
4
  import { t as executionTrackByLane } from "./execution-tracks-CpgFPpS5.js";
5
5
  import { _ as spanEpochMillis, r as runTraceAnalysisLoop, t as buildTraceAnalystTools } from "./tools-BmuN627J.js";
6
- import { i as combineAbortSignals } from "./run-score-iEEAWiBY.js";
6
+ import { c as makeFinding, o as combineAbortSignals } from "./proposal-findings-DCawte-y.js";
7
7
  import { ai } from "@ax-llm/ax";
8
8
  import { z } from "zod";
9
- import { createHash, randomUUID } from "node:crypto";
9
+ import { randomUUID } from "node:crypto";
10
10
  //#region src/analyst/ax-service.ts
11
11
  const configuredModels = /* @__PURE__ */ new WeakMap();
12
12
  /**
@@ -470,63 +470,6 @@ function everyAdjacent(values, predicate) {
470
470
  return values.slice(1).every((current, index) => predicate(values[index], current));
471
471
  }
472
472
  //#endregion
473
- //#region src/analyst/types.ts
474
- /**
475
- * Analyst contract — the missing orchestration layer over agent-eval's
476
- * existing analyzers (analyzeTraces, MultiLayerVerifier, RunCritic,
477
- * SemanticConceptJudge, JudgeFn, ...).
478
- *
479
- * Each existing primitive returns its own output shape. The Analyst
480
- * contract is the single envelope every primitive lifts into, so a
481
- * registry can run N analysts against a run and a single renderer can
482
- * compose findings without knowing which analyzer produced them.
483
- *
484
- * The contract is intentionally domain-agnostic: nothing here knows
485
- * about code, voice, RAG, or any particular agent stack. Analysts
486
- * declare what INPUT KIND they need (a trace store, an artifact dir,
487
- * a RunRecord, a JudgeInput, or `custom`), and the registry routes
488
- * the matching input from `AnalystRunInputs`.
489
- */
490
- /**
491
- * Compute the stable finding_id from the identity-defining fields.
492
- * Default implementation hashes {analyst_id, area, subject, normalized claim}.
493
- * Analysts that emit findings whose claim text varies per run (timestamps,
494
- * counts) SHOULD either: (a) pass an explicit `id_basis` to fix the hash,
495
- * or (b) move the variable part into `rationale`/`metadata` and keep the
496
- * `claim` static.
497
- */
498
- function computeFindingId(input) {
499
- const basis = JSON.stringify({
500
- a: input.analyst_id,
501
- r: input.area,
502
- s: input.subject ?? "",
503
- c: normalizeClaim(input.id_basis ?? input.claim)
504
- });
505
- return `f_${createHash("sha256").update(basis).digest("hex").slice(0, 20)}`;
506
- }
507
- function normalizeClaim(c) {
508
- return c.toLowerCase().replace(/\s+/g, " ").replace(/[.!?;:,]+$/g, "").trim();
509
- }
510
- /**
511
- * Convenience factory: produce a fully-formed AnalystFinding with the
512
- * id computed automatically. Analyst code stays terse.
513
- */
514
- function makeFinding(init) {
515
- const { id_basis, produced_at, ...rest } = init;
516
- return {
517
- schema_version: "1.0.0",
518
- finding_id: computeFindingId({
519
- analyst_id: rest.analyst_id,
520
- area: rest.area,
521
- subject: rest.subject,
522
- claim: rest.claim,
523
- id_basis
524
- }),
525
- produced_at: produced_at ?? (/* @__PURE__ */ new Date()).toISOString(),
526
- ...rest
527
- };
528
- }
529
- //#endregion
530
473
  //#region src/analyst/behavioral-analyst.ts
531
474
  /**
532
475
  * `behavioralAnalyst` — a DETERMINISTIC analyst (cost.kind = 'deterministic',
@@ -2565,6 +2508,6 @@ function buildDefaultAnalystRegistry(opts = {}) {
2565
2508
  return registry;
2566
2509
  }
2567
2510
  //#endregion
2568
- export { parseFindingSubject as A, stripCodeFences as C, FindingSubjectStringSchema as D, FINDING_SUBJECT_SYNTAX as E, makeFinding as F, computeTraceMetrics as I, createChatClient as L, behavioralAnalyst as M, deriveEfficiencyFindings as N, KIND_EXPECTED_SUBJECTS as O, computeFindingId as P, createAnalystAi as R, coerceToFindingRows as S, FINDING_SUBJECT_KINDS as T, RawAnalystEvidenceSchema as _, KNOWLEDGE_GAP_KIND_SPEC as a, parseRawFinding as b, buildTraceToolsForGroup as c, renderUpstreamFindings as d, settleUsageReceiptFromCostLedger as f, RAW_FINDING_SCHEMA_PROMPT as g, ANALYST_SEVERITIES as h, KNOWLEDGE_POISONING_KIND_SPEC as i, renderFindingSubject as j, findingSubjectGrammarPromptFor as k, createTraceAnalystKind as l, structureFindings as m, AnalystRegistry as n, IMPROVEMENT_KIND_SPEC as o, validateUsageSettlementTimeout as p, DEFAULT_TRACE_ANALYST_KINDS as r, FAILURE_MODE_KIND_SPEC as s, buildDefaultAnalystRegistry as t, renderPriorFindings as u, RawAnalystFindingSchema as v, FINDING_SUBJECT_GRAMMAR_PROMPT as w, coerceJson as x, evidenceRefsFromRawFinding as y };
2511
+ export { parseFindingSubject as A, stripCodeFences as C, FindingSubjectStringSchema as D, FINDING_SUBJECT_SYNTAX as E, createChatClient as F, createAnalystAi as I, behavioralAnalyst as M, deriveEfficiencyFindings as N, KIND_EXPECTED_SUBJECTS as O, computeTraceMetrics as P, coerceToFindingRows as S, FINDING_SUBJECT_KINDS as T, RawAnalystEvidenceSchema as _, KNOWLEDGE_GAP_KIND_SPEC as a, parseRawFinding as b, buildTraceToolsForGroup as c, renderUpstreamFindings as d, settleUsageReceiptFromCostLedger as f, RAW_FINDING_SCHEMA_PROMPT as g, ANALYST_SEVERITIES as h, KNOWLEDGE_POISONING_KIND_SPEC as i, renderFindingSubject as j, findingSubjectGrammarPromptFor as k, createTraceAnalystKind as l, structureFindings as m, AnalystRegistry as n, IMPROVEMENT_KIND_SPEC as o, validateUsageSettlementTimeout as p, DEFAULT_TRACE_ANALYST_KINDS as r, FAILURE_MODE_KIND_SPEC as s, buildDefaultAnalystRegistry as t, renderPriorFindings as u, RawAnalystFindingSchema as v, FINDING_SUBJECT_GRAMMAR_PROMPT as w, coerceJson as x, evidenceRefsFromRawFinding as y };
2569
2512
 
2570
- //# sourceMappingURL=default-registry-D3T9XbuY.js.map
2513
+ //# sourceMappingURL=default-registry-IjYs7T8l.js.map