@tangle-network/agent-eval 0.125.0 → 0.126.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (82) hide show
  1. package/CHANGELOG.md +62 -35
  2. package/README.md +270 -189
  3. package/dist/analyst/index.d.ts +15 -145
  4. package/dist/analyst/index.js +33 -47
  5. package/dist/analyst/index.js.map +1 -1
  6. package/dist/benchmarks/index.d.ts +45 -162
  7. package/dist/benchmarks/index.js +8 -9
  8. package/dist/campaign/index.d.ts +3674 -5393
  9. package/dist/campaign/index.js +21 -95
  10. package/dist/{chunk-R226UZOI.js → chunk-474LBSOX.js} +2 -2
  11. package/dist/{chunk-HM6V7F3M.js → chunk-FO7HEH76.js} +3 -3
  12. package/dist/chunk-IILEIWGW.js +635 -0
  13. package/dist/chunk-IILEIWGW.js.map +1 -0
  14. package/dist/{chunk-EQUK3RFS.js → chunk-J5SQWP6Y.js} +8 -5
  15. package/dist/chunk-J5SQWP6Y.js.map +1 -0
  16. package/dist/{chunk-W5B3ZGP3.js → chunk-KE2VWPZX.js} +8 -6
  17. package/dist/{chunk-W5B3ZGP3.js.map → chunk-KE2VWPZX.js.map} +1 -1
  18. package/dist/{chunk-DT7OXY3C.js → chunk-LUNF2SEL.js} +538 -851
  19. package/dist/chunk-LUNF2SEL.js.map +1 -0
  20. package/dist/chunk-NGUYT5CI.js +4637 -0
  21. package/dist/chunk-NGUYT5CI.js.map +1 -0
  22. package/dist/{chunk-QFQZ3U3X.js → chunk-OCFJACJU.js} +2 -2
  23. package/dist/{chunk-GID26AN4.js → chunk-P22LJ3Y2.js} +4 -6
  24. package/dist/{chunk-GID26AN4.js.map → chunk-P22LJ3Y2.js.map} +1 -1
  25. package/dist/{chunk-SJT4OBVL.js → chunk-SDPM6554.js} +3 -3
  26. package/dist/{chunk-D5JZ7UDZ.js → chunk-UCLVDLCH.js} +136 -50
  27. package/dist/chunk-UCLVDLCH.js.map +1 -0
  28. package/dist/chunk-VMUENW6F.js +7274 -0
  29. package/dist/chunk-VMUENW6F.js.map +1 -0
  30. package/dist/{chunk-JKDNAOF5.js → chunk-W4L6C2XT.js} +2 -2
  31. package/dist/chunk-WGXIEX7P.js +116 -0
  32. package/dist/chunk-WGXIEX7P.js.map +1 -0
  33. package/dist/{chunk-GRCDRKII.js → chunk-WS3NZZQQ.js} +58 -20
  34. package/dist/chunk-WS3NZZQQ.js.map +1 -0
  35. package/dist/cli.js +3 -3
  36. package/dist/contract/index.d.ts +3220 -3094
  37. package/dist/contract/index.js +173 -42
  38. package/dist/contract/index.js.map +1 -1
  39. package/dist/control.js +2 -3
  40. package/dist/fuzz.d.ts +14 -1
  41. package/dist/fuzz.js +1 -1
  42. package/dist/hosted/index.d.ts +8 -1
  43. package/dist/index.d.ts +71 -687
  44. package/dist/index.js +178 -497
  45. package/dist/index.js.map +1 -1
  46. package/dist/openapi.json +1 -1
  47. package/dist/rl.d.ts +5 -100
  48. package/dist/rl.js +4 -5
  49. package/dist/rl.js.map +1 -1
  50. package/dist/{run-campaign-I3JXKVAK.js → run-campaign-LVFKZCEU.js} +3 -3
  51. package/dist/traces.js +2 -3
  52. package/dist/wire/index.d.ts +14 -1
  53. package/dist/wire/index.js +3 -3
  54. package/docs/campaign-proposers.md +363 -168
  55. package/docs/design/loop-taxonomy.md +142 -190
  56. package/docs/design.md +1 -1
  57. package/docs/distributed-driver.md +8 -11
  58. package/docs/feature-guide.md +20 -19
  59. package/docs/knowledge-readiness.md +2 -5
  60. package/docs/multi-shot-optimization.md +35 -27
  61. package/docs/rollout.md +5 -5
  62. package/package.json +4 -4
  63. package/dist/chunk-A62YMFWA.js +0 -9269
  64. package/dist/chunk-A62YMFWA.js.map +0 -1
  65. package/dist/chunk-A6GT67HT.js +0 -550
  66. package/dist/chunk-A6GT67HT.js.map +0 -1
  67. package/dist/chunk-D5JZ7UDZ.js.map +0 -1
  68. package/dist/chunk-DT7OXY3C.js.map +0 -1
  69. package/dist/chunk-EQUK3RFS.js.map +0 -1
  70. package/dist/chunk-GC4ATIKK.js +0 -317
  71. package/dist/chunk-GC4ATIKK.js.map +0 -1
  72. package/dist/chunk-GRCDRKII.js.map +0 -1
  73. package/dist/chunk-LOW3U7JZ.js +0 -328
  74. package/dist/chunk-LOW3U7JZ.js.map +0 -1
  75. package/dist/chunk-PMITBABE.js +0 -3841
  76. package/dist/chunk-PMITBABE.js.map +0 -1
  77. /package/dist/{chunk-R226UZOI.js.map → chunk-474LBSOX.js.map} +0 -0
  78. /package/dist/{chunk-HM6V7F3M.js.map → chunk-FO7HEH76.js.map} +0 -0
  79. /package/dist/{chunk-QFQZ3U3X.js.map → chunk-OCFJACJU.js.map} +0 -0
  80. /package/dist/{chunk-SJT4OBVL.js.map → chunk-SDPM6554.js.map} +0 -0
  81. /package/dist/{chunk-JKDNAOF5.js.map → chunk-W4L6C2XT.js.map} +0 -0
  82. /package/dist/{run-campaign-I3JXKVAK.js.map → run-campaign-LVFKZCEU.js.map} +0 -0
@@ -1,25 +1,30 @@
1
1
  import {
2
+ executionTrackByLane
3
+ } from "./chunk-HHWE3POT.js";
4
+ import {
5
+ combineAbortSignals
6
+ } from "./chunk-WGXIEX7P.js";
7
+ import {
8
+ LlmClient,
2
9
  callLlm,
3
10
  costReceiptFromLlm,
4
11
  costReceiptFromLlmError,
5
12
  maximumChargeForLlmRequest
6
- } from "./chunk-EQUK3RFS.js";
13
+ } from "./chunk-J5SQWP6Y.js";
7
14
  import {
8
15
  CostLedger
9
- } from "./chunk-GRCDRKII.js";
16
+ } from "./chunk-WS3NZZQQ.js";
10
17
  import {
11
18
  buildTraceAnalystTools,
12
- runTraceAnalysisLoop
19
+ runTraceAnalysisLoop,
20
+ spanEpochMillis
13
21
  } from "./chunk-IR3KBHOY.js";
14
22
  import {
15
- validateAgentProfileCell
16
- } from "./chunk-GC4ATIKK.js";
17
- import {
18
- canonicalize
19
- } from "./chunk-VSMTAMNK.js";
20
- import {
21
- ValidationError
22
- } from "./chunk-ONWEPEDO.js";
23
+ LLM_CONTEXT_TOKENS,
24
+ LLM_INPUT_TOKEN_ATTR_KEYS,
25
+ LLM_OUTPUT_TOKEN_ATTR_KEYS,
26
+ TOOL_NAME_ATTR_KEYS
27
+ } from "./chunk-K4DBDHLK.js";
23
28
 
24
29
  // src/analyst/ax-service.ts
25
30
  import { ai } from "@ax-llm/ax";
@@ -56,6 +61,405 @@ function resolveAnalystModel(service, override) {
56
61
  return model;
57
62
  }
58
63
 
64
+ // src/analyst/chat-client.ts
65
+ function createChatClient(opts) {
66
+ switch (opts.transport) {
67
+ case "router":
68
+ return wrapLlmClient(
69
+ opts.transport,
70
+ opts.defaultModel,
71
+ new LlmClient({
72
+ baseUrl: opts.baseUrl ?? "https://router.tangle.tools/v1",
73
+ apiKey: opts.apiKey,
74
+ maxRetries: opts.maximumAttempts
75
+ })
76
+ );
77
+ case "cli-bridge":
78
+ return wrapLlmClient(
79
+ opts.transport,
80
+ opts.defaultModel,
81
+ new LlmClient({
82
+ baseUrl: opts.baseUrl ?? "http://127.0.0.1:3344/v1",
83
+ apiKey: opts.bearer ?? "",
84
+ maxRetries: opts.maximumAttempts
85
+ })
86
+ );
87
+ case "direct-provider":
88
+ return wrapLlmClient(
89
+ opts.transport,
90
+ opts.defaultModel,
91
+ new LlmClient({
92
+ baseUrl: opts.baseUrl,
93
+ apiKey: opts.apiKey,
94
+ maxRetries: opts.maximumAttempts
95
+ })
96
+ );
97
+ case "sandbox-sdk":
98
+ return {
99
+ transport: "sandbox-sdk",
100
+ defaultModel: opts.defaultModel,
101
+ maximumAttempts: opts.maximumAttempts,
102
+ chat: async (req, callOpts) => opts.chat(resolveModel(req, opts.defaultModel), callOpts)
103
+ };
104
+ case "mock":
105
+ return {
106
+ transport: "mock",
107
+ defaultModel: opts.defaultModel,
108
+ maximumAttempts: 1,
109
+ chat: async (req, callOpts) => opts.handler(resolveModel(req, opts.defaultModel), callOpts)
110
+ };
111
+ }
112
+ }
113
+ function wrapLlmClient(transport, defaultModel, inner) {
114
+ return {
115
+ transport,
116
+ defaultModel,
117
+ maximumAttempts: inner.maximumAttempts,
118
+ chat: (req, callOpts) => {
119
+ const resolved = resolveModel(req, defaultModel);
120
+ const request = {
121
+ model: resolved.model,
122
+ messages: req.messages,
123
+ jsonMode: req.jsonMode,
124
+ jsonSchema: req.jsonSchema,
125
+ temperature: req.temperature,
126
+ maxTokens: req.maxTokens,
127
+ timeoutMs: req.timeoutMs
128
+ };
129
+ return inner.call(request, {
130
+ signal: callOpts?.signal,
131
+ idempotencyKey: callOpts?.idempotencyKey
132
+ });
133
+ }
134
+ };
135
+ }
136
+ function resolveModel(req, defaultModel) {
137
+ if (req.model) return req;
138
+ if (!defaultModel) {
139
+ throw new Error(
140
+ "ChatClient.chat: no model on request and no defaultModel on the client. Either pass req.model or bind defaultModel at createChatClient()."
141
+ );
142
+ }
143
+ return { ...req, model: defaultModel };
144
+ }
145
+
146
+ // src/trace-analyst/behavioral-metrics.ts
147
+ var INPUT_GROWTH_FACTOR = 3;
148
+ var MIN_TOOL_CALLS = 3;
149
+ var VERIFY_RE = /verif|eval|inspect|check|assert|validat|review|confirm|read|grep|glob|search|view|\blist\b|\bls\b|\bcat\b|\bfind\b|diff|status|\btest|lint|typecheck/i;
150
+ function num(v) {
151
+ return typeof v === "number" && Number.isFinite(v) ? v : null;
152
+ }
153
+ function numAttr(attrs, keys) {
154
+ for (const key of keys) {
155
+ const value = num(attrs[key]);
156
+ if (value !== null) return value;
157
+ }
158
+ return null;
159
+ }
160
+ function inputTokensOf(s) {
161
+ const exactContext = num(s.attributes[LLM_CONTEXT_TOKENS]);
162
+ if (exactContext !== null) return exactContext;
163
+ return numAttr(s.attributes, LLM_INPUT_TOKEN_ATTR_KEYS) ?? num(s.attributes["llm.usage.input_tokens"]);
164
+ }
165
+ function outputTokensOf(s) {
166
+ return numAttr(s.attributes, LLM_OUTPUT_TOKEN_ATTR_KEYS) ?? num(s.attributes["llm.usage.output_tokens"]);
167
+ }
168
+ function stepOf(s) {
169
+ return num(s.attributes.step);
170
+ }
171
+ function toolNameOf(s) {
172
+ if (s.tool_name) return s.tool_name;
173
+ for (const key of TOOL_NAME_ATTR_KEYS) {
174
+ const t = s.attributes[key];
175
+ if (typeof t === "string" && t.length > 0) return t;
176
+ }
177
+ return null;
178
+ }
179
+ function computeTraceMetrics(spans) {
180
+ const traceIds = new Set(spans.map((span) => span.trace_id));
181
+ if (traceIds.size > 1) {
182
+ throw new Error(
183
+ `computeTraceMetrics: expected spans from one trace, received ${traceIds.size} traces`
184
+ );
185
+ }
186
+ const traceId = traceIds.values().next().value ?? null;
187
+ const samples = spans.map((span) => ({
188
+ span,
189
+ input: inputTokensOf(span),
190
+ output: outputTokensOf(span),
191
+ step: stepOf(span)
192
+ }));
193
+ const llmSamples = samples.filter((sample) => sample.span.kind === "LLM");
194
+ const tokenSamples = llmSamples.length > 0 ? llmSamples : samples.filter((sample) => sample.input !== null || sample.output !== null);
195
+ const tokenSequences = buildTokenSequences(tokenSamples, spans);
196
+ const primarySequence = tokenSequences[0];
197
+ const inputTokenTrajectory = primarySequence?.inputTokenTrajectory.filter((value) => value !== null) ?? [];
198
+ const outputTokenTrajectory = primarySequence?.outputTokenTrajectory.filter((value) => value !== null) ?? [];
199
+ const toolHistogram = {};
200
+ let hasSelfVerification = false;
201
+ for (const s of spans) {
202
+ const tool = toolNameOf(s);
203
+ if (tool) {
204
+ toolHistogram[tool] = (toolHistogram[tool] ?? 0) + 1;
205
+ if (VERIFY_RE.test(tool)) hasSelfVerification = true;
206
+ }
207
+ }
208
+ const totalToolCalls = Object.values(toolHistogram).reduce((a, b) => a + b, 0);
209
+ const distinctTools = Object.keys(toolHistogram).length;
210
+ const toolDiversityRatio = totalToolCalls === 0 ? 1 : distinctTools / totalToolCalls;
211
+ const signals = [];
212
+ const seenTokenSignals = /* @__PURE__ */ new Set();
213
+ for (const sequence of tokenSequences) {
214
+ for (const signal of tokenSignals(sequence)) {
215
+ if (seenTokenSignals.has(signal.code)) continue;
216
+ seenTokenSignals.add(signal.code);
217
+ signals.push(signal);
218
+ }
219
+ }
220
+ if (totalToolCalls >= MIN_TOOL_CALLS && distinctTools === 1) {
221
+ const only = Object.keys(toolHistogram)[0];
222
+ signals.push({
223
+ code: "single-tool-dependency",
224
+ severity: "medium",
225
+ detail: `All ${totalToolCalls} observed tool calls are \`${only}\`; no alternate tool call was observed.`,
226
+ evidence: { tool: only, calls: totalToolCalls, distinct_tools: 1 }
227
+ });
228
+ }
229
+ if (totalToolCalls >= MIN_TOOL_CALLS && !hasSelfVerification) {
230
+ signals.push({
231
+ code: "no-self-verification",
232
+ severity: "medium",
233
+ detail: `${totalToolCalls} tool calls were observed without a verification-named tool call.`,
234
+ evidence: { tool_calls: totalToolCalls, verification_calls: 0 }
235
+ });
236
+ }
237
+ return {
238
+ traceId,
239
+ llmCallCount: tokenSamples.length,
240
+ tokenSequences,
241
+ inputTokenTrajectory,
242
+ outputTokenTrajectory,
243
+ toolHistogram,
244
+ totalToolCalls,
245
+ distinctTools,
246
+ toolDiversityRatio,
247
+ hasSelfVerification,
248
+ signals
249
+ };
250
+ }
251
+ function buildTokenSequences(samples, spans) {
252
+ const spansById = new Map(spans.map((span) => [span.span_id, span]));
253
+ const executionScopeFor = createTokenExecutionScopeResolver(spansById);
254
+ const scopedSamples = samples.map((sample) => ({
255
+ sample,
256
+ ...executionScopeFor(sample.span)
257
+ }));
258
+ const trackByLane = executionTrackByLane(scopedSamples);
259
+ const byTrack = /* @__PURE__ */ new Map();
260
+ for (const scoped of scopedSamples) {
261
+ const trackId = trackByLane.get(scoped.key);
262
+ const track = byTrack.get(trackId) ?? { scopeId: scoped.scopeId, samples: [] };
263
+ track.samples.push(scoped.sample);
264
+ byTrack.set(trackId, track);
265
+ }
266
+ const sequences = [];
267
+ for (const { scopeId, samples: tracked } of byTrack.values()) {
268
+ const runs = serialTokenRuns([...tracked].sort(compareTokenSamples));
269
+ runs.forEach((run, index) => {
270
+ sequences.push({
271
+ scopeId: runs.length === 1 ? scopeId : `${scopeId}#${index + 1}`,
272
+ spanIds: run.map((sample) => sample.span.span_id),
273
+ inputTokenTrajectory: run.map((sample) => sample.input),
274
+ outputTokenTrajectory: run.map((sample) => sample.output)
275
+ });
276
+ });
277
+ }
278
+ return sequences.sort(
279
+ (a, b) => b.spanIds.length - a.spanIds.length || a.scopeId.localeCompare(b.scopeId) || a.spanIds[0].localeCompare(b.spanIds[0])
280
+ );
281
+ }
282
+ function createTokenExecutionScopeResolver(spansById) {
283
+ const cache = /* @__PURE__ */ new Map();
284
+ return (span) => {
285
+ const ancestry = resolveAncestorScope(span.parent_span_id, spansById, cache);
286
+ const rootId = ancestry.rootId;
287
+ const scopeId = ancestry.agentId ? `span:${ancestry.agentId}` : ancestry.missingParentId ? `parent:${ancestry.missingParentId}` : rootId ? `root:${rootId}` : span.agent_name ? `agent:${span.agent_name}` : `trace:${span.trace_id}`;
288
+ const scopeSpanId = ancestry.agentId ?? ancestry.missingParentId ?? rootId;
289
+ const laneSpan = ancestry.laneSpanId ? spansById.get(ancestry.laneSpanId) : void 0;
290
+ const direct = scopeSpanId === null || ancestry.laneSpanId === scopeSpanId;
291
+ const timedSpan = direct ? span : laneSpan;
292
+ return {
293
+ key: JSON.stringify([scopeId, direct ? span.span_id : ancestry.laneSpanId]),
294
+ scopeKey: scopeId,
295
+ scopeId,
296
+ start: timedSpan ? spanEpochMillis(timedSpan.start_time) : null,
297
+ end: timedSpan ? spanEpochMillis(timedSpan.end_time) : null
298
+ };
299
+ };
300
+ }
301
+ function resolveAncestorScope(startId, spansById, cache) {
302
+ const empty = {
303
+ agentId: null,
304
+ rootId: null,
305
+ missingParentId: null,
306
+ laneSpanId: null
307
+ };
308
+ if (!startId) return empty;
309
+ const path = [];
310
+ const pathIndex = /* @__PURE__ */ new Map();
311
+ let currentId = startId;
312
+ let resolved = empty;
313
+ while (currentId) {
314
+ const cached = cache.get(currentId);
315
+ if (cached) {
316
+ resolved = cached;
317
+ break;
318
+ }
319
+ const cycleStart = pathIndex.get(currentId);
320
+ if (cycleStart !== void 0) {
321
+ resolved = {
322
+ agentId: null,
323
+ rootId: [...path.slice(cycleStart)].sort()[0],
324
+ missingParentId: null,
325
+ laneSpanId: [...path.slice(cycleStart)].sort()[0]
326
+ };
327
+ break;
328
+ }
329
+ const current = spansById.get(currentId);
330
+ if (!current) {
331
+ resolved = {
332
+ agentId: null,
333
+ rootId: null,
334
+ missingParentId: currentId,
335
+ laneSpanId: currentId
336
+ };
337
+ break;
338
+ }
339
+ if (current.kind === "AGENT") {
340
+ resolved = {
341
+ agentId: current.span_id,
342
+ rootId: null,
343
+ missingParentId: null,
344
+ laneSpanId: current.span_id
345
+ };
346
+ break;
347
+ }
348
+ pathIndex.set(currentId, path.length);
349
+ path.push(currentId);
350
+ currentId = current.parent_span_id;
351
+ }
352
+ for (let index = path.length - 1; index >= 0; index -= 1) {
353
+ if (resolved.agentId === null && resolved.rootId === null) {
354
+ resolved = { ...resolved, rootId: path[index], laneSpanId: path[index] };
355
+ } else if (resolved.laneSpanId === (resolved.agentId ?? resolved.missingParentId ?? resolved.rootId)) {
356
+ resolved = { ...resolved, laneSpanId: path[index] };
357
+ }
358
+ cache.set(path[index], resolved);
359
+ }
360
+ return resolved;
361
+ }
362
+ function compareTokenSamples(a, b) {
363
+ const aStart = spanEpochMillis(a.span.start_time);
364
+ const bStart = spanEpochMillis(b.span.start_time);
365
+ if (aStart === null && bStart !== null) return 1;
366
+ if (aStart !== null && bStart === null) return -1;
367
+ if (aStart !== null && bStart !== null && aStart !== bStart) return aStart - bStart;
368
+ if (a.step !== null && b.step !== null && a.step !== b.step) return a.step - b.step;
369
+ return a.span.span_id.localeCompare(b.span.span_id);
370
+ }
371
+ function serialTokenRuns(ordered) {
372
+ const runs = [];
373
+ let serial = [];
374
+ let overlap = [];
375
+ let overlapEnd = Number.NEGATIVE_INFINITY;
376
+ const flushSerial = () => {
377
+ if (serial.length > 0) runs.push(serial);
378
+ serial = [];
379
+ };
380
+ const flushOverlap = () => {
381
+ if (overlap.length === 1) {
382
+ serial.push(overlap[0]);
383
+ } else if (overlap.length > 1) {
384
+ flushSerial();
385
+ for (const sample of overlap) runs.push([sample]);
386
+ }
387
+ overlap = [];
388
+ overlapEnd = Number.NEGATIVE_INFINITY;
389
+ };
390
+ for (const sample of ordered) {
391
+ const start = spanEpochMillis(sample.span.start_time);
392
+ const end = spanEpochMillis(sample.span.end_time);
393
+ if (start === null || end === null || sample.span.duration_ms <= 0 || end < start) {
394
+ flushOverlap();
395
+ flushSerial();
396
+ runs.push([sample]);
397
+ continue;
398
+ }
399
+ if (overlap.length > 0 && start >= overlapEnd) flushOverlap();
400
+ overlap.push(sample);
401
+ overlapEnd = Math.max(overlapEnd, end);
402
+ }
403
+ flushOverlap();
404
+ flushSerial();
405
+ return runs;
406
+ }
407
+ function tokenSignals(sequence) {
408
+ const signals = [];
409
+ const inputs = sequence.inputTokenTrajectory;
410
+ const outputs = sequence.outputTokenTrajectory;
411
+ if (inputs.length >= 3 && inputs.every((value) => value !== null)) {
412
+ const first = inputs[0];
413
+ const last = inputs[inputs.length - 1];
414
+ const isMonotonic = everyAdjacent(inputs, (previous, current) => current >= previous);
415
+ const growthFromZero = first === 0 && last > 0;
416
+ const growth = growthFromZero ? Infinity : first > 0 ? last / first : 0;
417
+ if (isMonotonic && last > first && growth >= INPUT_GROWTH_FACTOR) {
418
+ const growthLabel = growthFromZero ? "0\u2192nonzero (unbounded)" : `${growth.toFixed(1)}x`;
419
+ signals.push({
420
+ code: "monotonic-input-growth",
421
+ severity: "high",
422
+ detail: `LLM input tokens grew ${growthLabel} (${first}\u2192${last}) across ${inputs.length} serial calls without an intervening decrease.`,
423
+ evidence: {
424
+ first,
425
+ last,
426
+ growth_x: growthFromZero ? "unbounded" : Number(growth.toFixed(2)),
427
+ calls: inputs.length,
428
+ scope: sequence.scopeId,
429
+ first_span_id: sequence.spanIds[0],
430
+ last_span_id: sequence.spanIds[sequence.spanIds.length - 1]
431
+ }
432
+ });
433
+ }
434
+ }
435
+ if (inputs.length >= 3 && inputs.length === outputs.length && inputs.every((value) => value !== null) && outputs.every((value) => value !== null)) {
436
+ const first = outputs[0];
437
+ const last = outputs[outputs.length - 1];
438
+ const inputIsMonotonic = everyAdjacent(inputs, (previous, current) => current >= previous);
439
+ const outputIsMonotonic = everyAdjacent(outputs, (previous, current) => current <= previous);
440
+ const inputGrew = inputs[inputs.length - 1] > inputs[0];
441
+ if (inputIsMonotonic && inputGrew && outputIsMonotonic && last < first) {
442
+ signals.push({
443
+ code: "output-length-decay",
444
+ severity: "medium",
445
+ detail: `LLM output tokens shrank ${first}\u2192${last} over ${outputs.length} serial calls while input tokens increased monotonically.`,
446
+ evidence: {
447
+ first,
448
+ last,
449
+ calls: outputs.length,
450
+ scope: sequence.scopeId,
451
+ first_span_id: sequence.spanIds[0],
452
+ last_span_id: sequence.spanIds[sequence.spanIds.length - 1]
453
+ }
454
+ });
455
+ }
456
+ }
457
+ return signals;
458
+ }
459
+ function everyAdjacent(values, predicate) {
460
+ return values.slice(1).every((current, index) => predicate(values[index], current));
461
+ }
462
+
59
463
  // src/analyst/types.ts
60
464
  import { createHash } from "crypto";
61
465
  function computeFindingId(input) {
@@ -86,6 +490,112 @@ function makeFinding(init) {
86
490
  };
87
491
  }
88
492
 
493
+ // src/analyst/behavioral-analyst.ts
494
+ var RECOMMENDED_ACTION = {
495
+ "monotonic-input-growth": "Inspect context assembly; if prior history is repeatedly included, summarize completed work before the next model call.",
496
+ "output-length-decay": "Check late-step completeness; if shorter responses omit required work, add explicit completion criteria to the agent instructions.",
497
+ "single-tool-dependency": "Test whether an inspect or verification tool improves outcomes after the repeated call fails or returns no progress.",
498
+ "no-self-verification": "After state-changing actions, require an observable check before the agent proceeds."
499
+ };
500
+ var ANALYST_ID = "efficiency-behavioral";
501
+ var AGGREGATE_CLAIM = {
502
+ "monotonic-input-growth": (observed, analyzed) => `${observed}/${analyzed} analyzed traces showed input tokens grow from zero to nonzero or to at least 3x their initial value across at least 3 serial model calls without a decrease.`,
503
+ "output-length-decay": (observed, analyzed) => `${observed}/${analyzed} analyzed traces showed output tokens decrease while input tokens increased monotonically across at least 3 serial model calls.`,
504
+ "single-tool-dependency": (observed, analyzed) => `${observed}/${analyzed} analyzed traces used only one named tool across at least 3 tool calls.`,
505
+ "no-self-verification": (observed, analyzed) => `${observed}/${analyzed} analyzed traces had at least 3 tool calls without a verification-named tool call.`
506
+ };
507
+ function deriveEfficiencyFindings(metrics, opts = {}) {
508
+ const analystId = opts.analystId ?? ANALYST_ID;
509
+ const traceId = metrics.traceId;
510
+ return metrics.signals.map(
511
+ (sig) => makeFinding({
512
+ analyst_id: analystId,
513
+ area: "efficiency",
514
+ subject: sig.code,
515
+ claim: sig.detail,
516
+ severity: sig.severity,
517
+ // Deterministic arithmetic over spans, not a model judgment → certain.
518
+ confidence: 1,
519
+ evidence_refs: [
520
+ {
521
+ kind: "metric",
522
+ uri: traceId ? `metric://trace/${encodeURIComponent(traceId)}/efficiency/${sig.code}` : `metric://efficiency/${sig.code}`,
523
+ excerpt: JSON.stringify(sig.evidence)
524
+ }
525
+ ],
526
+ recommended_action: RECOMMENDED_ACTION[sig.code],
527
+ metadata: {
528
+ deterministic: true,
529
+ evidence: sig.evidence,
530
+ ...traceId ? { trace_id: traceId } : {}
531
+ },
532
+ id_basis: sig.code,
533
+ ...opts.producedAt ? { produced_at: opts.producedAt } : {}
534
+ })
535
+ );
536
+ }
537
+ function behavioralAnalyst() {
538
+ return {
539
+ id: ANALYST_ID,
540
+ description: "Deterministic behavioral/efficiency findings over OTLP spans \u2014 token-growth, output-decay, tool-monoculture, missing self-verification. Zero LLM; model-agnostic by construction.",
541
+ inputKind: "trace-store",
542
+ cost: { kind: "deterministic" },
543
+ version: "2.0.0",
544
+ async analyze(store) {
545
+ const overview = await store.getOverview();
546
+ const analyzedTraceIds = [...new Set(overview.sample_trace_ids)].sort();
547
+ const findingsById = /* @__PURE__ */ new Map();
548
+ for (const traceId of analyzedTraceIds) {
549
+ const viewed = await store.viewTrace({ trace_id: traceId });
550
+ if (viewed.trace_id !== traceId) {
551
+ throw new Error(
552
+ `behavioralAnalyst: requested trace '${traceId}', received '${viewed.trace_id}'`
553
+ );
554
+ }
555
+ if (!viewed.spans) {
556
+ throw new Error(
557
+ `behavioralAnalyst: trace '${traceId}' is oversized; complete spans are required`
558
+ );
559
+ }
560
+ const metrics = computeTraceMetrics(viewed.spans);
561
+ if (metrics.traceId !== null && metrics.traceId !== traceId) {
562
+ throw new Error(
563
+ `behavioralAnalyst: requested trace '${traceId}', received '${metrics.traceId}'`
564
+ );
565
+ }
566
+ for (const finding of deriveEfficiencyFindings(metrics)) {
567
+ const current = findingsById.get(finding.finding_id);
568
+ if (!current) {
569
+ findingsById.set(finding.finding_id, {
570
+ finding,
571
+ traceIds: [traceId],
572
+ evidence: [...finding.evidence_refs]
573
+ });
574
+ continue;
575
+ }
576
+ current.traceIds.push(traceId);
577
+ current.evidence.push(...finding.evidence_refs);
578
+ }
579
+ }
580
+ return [...findingsById.values()].map(({ finding, traceIds, evidence }) => ({
581
+ ...finding,
582
+ claim: AGGREGATE_CLAIM[finding.subject](
583
+ traceIds.length,
584
+ analyzedTraceIds.length
585
+ ),
586
+ rationale: `${traceIds.length}/${analyzedTraceIds.length} analyzed traces exhibited this pattern.`,
587
+ evidence_refs: evidence,
588
+ metadata: {
589
+ deterministic: true,
590
+ trace_ids: traceIds,
591
+ observed_trace_count: traceIds.length,
592
+ analyzed_trace_count: analyzedTraceIds.length
593
+ }
594
+ }));
595
+ }
596
+ };
597
+ }
598
+
89
599
  // src/analyst/finding-subject.ts
90
600
  import { z } from "zod";
91
601
  var FINDING_SUBJECT_KINDS = [
@@ -1601,11 +2111,6 @@ function validateTimeout(timeoutMs) {
1601
2111
  }
1602
2112
  return timeoutMs;
1603
2113
  }
1604
- function combineAbortSignals(caller, timeout) {
1605
- if (!caller) return timeout;
1606
- if (!timeout) return caller;
1607
- return AbortSignal.any([caller, timeout]);
1608
- }
1609
2114
  async function waitForOperation(operation, signal, abortGraceMs) {
1610
2115
  if (!signal) return operation;
1611
2116
  if (signal.aborted) {
@@ -1863,825 +2368,29 @@ function selectPriorFindings(source, analystId) {
1863
2368
  return merged.length > 0 ? merged : void 0;
1864
2369
  }
1865
2370
 
1866
- // src/concurrency.ts
1867
- var Mutex = class {
1868
- locked = false;
1869
- waiters = [];
1870
- async acquire() {
1871
- if (!this.locked) {
1872
- this.locked = true;
1873
- return () => this.release();
1874
- }
1875
- return new Promise((resolve) => {
1876
- this.waiters.push(() => {
1877
- resolve(() => this.release());
1878
- });
1879
- });
1880
- }
1881
- release() {
1882
- const next = this.waiters.shift();
1883
- if (next) {
1884
- next();
1885
- } else {
1886
- this.locked = false;
1887
- }
1888
- }
1889
- async runExclusive(fn) {
1890
- const release = await this.acquire();
1891
- try {
1892
- return await fn();
1893
- } finally {
1894
- release();
1895
- }
1896
- }
1897
- /** True iff someone holds the lock right now. Diagnostics only. */
1898
- get isLocked() {
1899
- return this.locked;
1900
- }
1901
- /** Pending waiter count. Diagnostics only. */
1902
- get pending() {
1903
- return this.waiters.length;
1904
- }
1905
- };
1906
- async function mapConcurrent(items, concurrency, map) {
1907
- if (!Number.isInteger(concurrency) || concurrency < 1) {
1908
- throw new Error(`mapConcurrent: concurrency must be a positive integer, got ${concurrency}`);
1909
- }
1910
- if (items.length === 0) return [];
1911
- const results = new Array(items.length);
1912
- let nextIndex = 0;
1913
- let stopped = false;
1914
- let failed = false;
1915
- let failure;
1916
- const worker = async () => {
1917
- while (!stopped) {
1918
- const index = nextIndex;
1919
- nextIndex += 1;
1920
- if (index >= items.length) return;
1921
- try {
1922
- results[index] = await map(items[index], index);
1923
- } catch (error) {
1924
- stopped = true;
1925
- if (!failed) {
1926
- failed = true;
1927
- failure = error;
1928
- }
1929
- }
1930
- }
1931
- };
1932
- await Promise.all(Array.from({ length: Math.min(concurrency, items.length) }, () => worker()));
1933
- if (failed) throw failure;
1934
- return results;
1935
- }
1936
-
1937
- // src/analyst/steer-firewall.ts
1938
- var OBSERVABLE_KINDS = /* @__PURE__ */ new Set([
1939
- "span",
1940
- "event",
1941
- "artifact"
1942
- ]);
1943
- function isTraceObservable(finding) {
1944
- return finding.evidence_refs.some((ref) => OBSERVABLE_KINDS.has(ref.kind));
1945
- }
1946
- function isJudgeVerdict(finding) {
1947
- return finding.derived_from_judge === true;
1948
- }
1949
- function assertNoJudgeVerdict(findings, context = "steer") {
1950
- const leaks = findings.filter(isJudgeVerdict);
1951
- if (leaks.length > 0) {
1952
- throw new Error(
1953
- `${context}: a judge verdict cannot be admitted as steering input \u2014 that is the held-out judge leaking into the loop. Offending judge-derived findings: [${leaks.map((f) => f.finding_id).join(", ")}]. Steering consumes observations of behavior, never acceptance verdicts.`
1954
- );
1955
- }
1956
- return findings;
1957
- }
1958
-
1959
- // src/analyst/policy-edit.ts
1960
- import { createHash as createHash2 } from "crypto";
1961
- var POLICY_EDIT_AXES = [
1962
- "carrier",
1963
- "representation",
1964
- "budget",
1965
- "sampling",
1966
- "output_contract",
1967
- "tool_contract",
1968
- "routing",
1969
- "memory",
1970
- "agent_profile",
1971
- "deployment_target"
1972
- ];
1973
- var POLICY_EDIT_TARGET_SURFACES = [
1974
- "prompt",
1975
- "tool-contract",
1976
- "runtime-config",
1977
- "memory",
1978
- "agent-profile",
1979
- "code",
1980
- "deployment"
1981
- ];
1982
- var POLICY_EDIT_CANDIDATE_RECORD_SCHEMA = "tangle.policy-edit-candidate.v1";
1983
- var PolicyEditValidationError = class extends ValidationError {
1984
- path;
1985
- constructor(message, path = "") {
1986
- super(path ? `${message} (at ${path})` : message);
1987
- this.path = path;
1988
- }
1989
- };
1990
- var DEFAULT_MIN_SCORE = 0.7;
1991
- var DEFAULT_MIN_EXPECTED_GAIN = 0.01;
1992
- var POLICY_EDIT_ID = /^policy-edit:sha256:[0-9a-f]{64}$/;
1993
- function makePolicyEdit(init) {
1994
- const normalized = normalizePolicyEdit({
1995
- schemaVersion: "policy-edit/v1",
1996
- ...init,
1997
- source: normalizeSource(init.source)
1998
- });
1999
- const edit = {
2000
- ...normalized,
2001
- editId: init.editId ?? computePolicyEditId(normalized)
2002
- };
2003
- return validatePolicyEdit(edit);
2004
- }
2005
- function computePolicyEditId(edit) {
2006
- const { editId: _editId, schemaVersion, ...material } = edit;
2007
- void _editId;
2008
- const canonical = JSON.stringify(canonicalize({ schemaVersion, ...material }));
2009
- return `policy-edit:sha256:${createHash2("sha256").update(canonical).digest("hex")}`;
2010
- }
2011
- function validatePolicyEdit(input) {
2012
- if (input === null || typeof input !== "object") {
2013
- throw new PolicyEditValidationError("expected object");
2014
- }
2015
- const obj = input;
2016
- expectLiteral(obj.schemaVersion, "policy-edit/v1", "schemaVersion");
2017
- expectString(obj.editId, "editId");
2018
- if (!POLICY_EDIT_ID.test(obj.editId)) {
2019
- throw new PolicyEditValidationError(
2020
- "editId must match policy-edit:sha256:<64 lowercase hex chars>",
2021
- "editId"
2022
- );
2023
- }
2024
- expectOneOf(obj.axis, POLICY_EDIT_AXES, "axis");
2025
- validateTarget(obj.target);
2026
- validateChange(obj.change);
2027
- expectString(obj.claim, "claim");
2028
- validateExpectedGain(obj.expectedGain);
2029
- expectConfidence(obj.confidence, "confidence");
2030
- expectOneOf(obj.risk, ["low", "medium", "high", "unknown"], "risk");
2031
- validateSource(obj.source);
2032
- if (obj.rationale !== void 0) expectString(obj.rationale, "rationale");
2033
- if (obj.validationPlan !== void 0) expectString(obj.validationPlan, "validationPlan");
2034
- if (obj.metadata !== void 0 && (obj.metadata === null || typeof obj.metadata !== "object")) {
2035
- throw new PolicyEditValidationError("expected object", "metadata");
2036
- }
2037
- const expectedId = computePolicyEditId(obj);
2038
- if (obj.editId !== expectedId) {
2039
- throw new PolicyEditValidationError("editId does not match policy edit content", "editId");
2040
- }
2041
- return obj;
2042
- }
2043
- function makePolicyEditCandidateRecord(edit) {
2044
- return validatePolicyEditCandidateRecord({
2045
- schema: POLICY_EDIT_CANDIDATE_RECORD_SCHEMA,
2046
- policyEdit: edit
2047
- });
2048
- }
2049
- function validatePolicyEditCandidateRecord(input) {
2050
- if (input === null || typeof input !== "object" || Array.isArray(input)) {
2051
- throw new PolicyEditValidationError("expected object", "candidateRecord");
2052
- }
2053
- const obj = input;
2054
- const keys = Object.keys(obj).sort();
2055
- if (keys.length !== 2 || keys[0] !== "policyEdit" || keys[1] !== "schema") {
2056
- throw new PolicyEditValidationError("expected exactly schema and policyEdit", "candidateRecord");
2057
- }
2058
- expectLiteral(obj.schema, POLICY_EDIT_CANDIDATE_RECORD_SCHEMA, "candidateRecord.schema");
2059
- const policyEdit = validatePolicyEdit(obj.policyEdit);
2060
- assertJsonSafe(policyEdit, "candidateRecord.policyEdit");
2061
- const snapshot = JSON.parse(JSON.stringify(policyEdit));
2062
- return {
2063
- schema: POLICY_EDIT_CANDIDATE_RECORD_SCHEMA,
2064
- policyEdit: validatePolicyEdit(snapshot)
2065
- };
2066
- }
2067
- function isPolicyEdit(input) {
2068
- try {
2069
- validatePolicyEdit(input);
2070
- return true;
2071
- } catch {
2072
- return false;
2073
- }
2074
- }
2075
- function policyEditsFromFindings(findings, opts = {}) {
2076
- assertNoJudgeVerdict(findings, "policyEditsFromFindings");
2077
- const edits = [];
2078
- for (const finding of findings) {
2079
- const edit = policyEditFromFinding(finding, opts);
2080
- if (edit) edits.push(edit);
2081
- }
2082
- return edits;
2083
- }
2084
- function policyEditFromFinding(finding, opts = {}) {
2085
- assertNoJudgeVerdict([finding], "policyEditFromFinding");
2086
- if (!finding.recommended_action?.trim()) return null;
2087
- const expectedGain = resolveExpectedGain(finding, opts);
2088
- if (!expectedGain) return null;
2089
- const routed = routeFindingSubject(finding.subject, opts);
2090
- const risk = resolveRisk(finding, opts);
2091
- return makePolicyEdit({
2092
- axis: routed.axis,
2093
- target: routed.target,
2094
- change: { kind: "text", mode: "append", value: finding.recommended_action.trim() },
2095
- claim: finding.claim,
2096
- rationale: finding.rationale,
2097
- expectedGain,
2098
- confidence: finding.confidence,
2099
- risk,
2100
- validationPlan: finding.validation_plan,
2101
- source: {
2102
- findingIds: [finding.finding_id],
2103
- analystIds: [finding.analyst_id],
2104
- evidenceRefs: finding.evidence_refs,
2105
- derivedFromJudge: finding.derived_from_judge
2371
+ // src/analyst/default-registry.ts
2372
+ function buildDefaultAnalystRegistry(opts = {}) {
2373
+ const registry = new AnalystRegistry(opts.registry);
2374
+ if (opts.includeBehavioral !== false) {
2375
+ registry.register(behavioralAnalyst());
2376
+ }
2377
+ if (opts.ai) {
2378
+ const kinds = opts.kinds ?? DEFAULT_TRACE_ANALYST_KINDS;
2379
+ for (const spec of kinds) {
2380
+ registry.register(createTraceAnalystKind(spec, { ai: opts.ai, model: opts.model }));
2106
2381
  }
2107
- });
2108
- }
2109
- function scorePolicyEditReadiness(edit, opts = {}) {
2110
- validatePolicyEdit(edit);
2111
- const minExpectedGain = opts.minExpectedGain ?? DEFAULT_MIN_EXPECTED_GAIN;
2112
- const evidenceScore = Math.min(1, edit.source.evidenceRefs.length / 2);
2113
- const confidenceScore = clamp01(edit.confidence);
2114
- const gainScore = clamp01(
2115
- Math.abs(edit.expectedGain.amount) / Math.max(minExpectedGain * 5, 1e-3)
2116
- );
2117
- const targetScore = targetSpecificityScore(edit);
2118
- const riskPenalty = edit.risk === "high" && opts.allowHighRisk !== true ? 0.35 : edit.risk === "unknown" ? 0.2 : 0;
2119
- return clamp01(
2120
- 0.3 * evidenceScore + 0.25 * confidenceScore + 0.25 * gainScore + 0.2 * targetScore - riskPenalty
2121
- );
2122
- }
2123
- function admitPolicyEdit(edit, opts = {}) {
2124
- const validated = validatePolicyEdit(edit);
2125
- const score = scorePolicyEditReadiness(validated, opts);
2126
- const reasons = [];
2127
- const minExpectedGain = opts.minExpectedGain ?? DEFAULT_MIN_EXPECTED_GAIN;
2128
- const requireEvidence = opts.requireEvidence ?? true;
2129
- if (validated.source.derivedFromJudge) {
2130
- reasons.push("source is judge-derived; judge verdicts cannot steer policy edits");
2131
- }
2132
- if (requireEvidence && validated.source.evidenceRefs.length === 0) {
2133
- reasons.push("missing evidence refs");
2134
- }
2135
- if (Math.abs(validated.expectedGain.amount) < minExpectedGain) {
2136
- reasons.push(`expected gain below ${minExpectedGain}`);
2137
- }
2138
- if (validated.risk === "high" && opts.allowHighRisk !== true) {
2139
- reasons.push("high-risk edit requires explicit allowHighRisk");
2140
2382
  }
2141
- if (score < (opts.minScore ?? DEFAULT_MIN_SCORE)) {
2142
- reasons.push(
2143
- `readiness score ${score.toFixed(3)} below ${(opts.minScore ?? DEFAULT_MIN_SCORE).toFixed(3)}`
2144
- );
2145
- }
2146
- return {
2147
- edit: validated,
2148
- decision: reasons.length === 0 ? "admit" : "reject",
2149
- score,
2150
- reasons
2151
- };
2152
- }
2153
- function applyPolicyEditToSurface(surface, edit) {
2154
- const validated = validatePolicyEdit(edit);
2155
- if (validated.change.kind === "text") return applyTextChange(surface, validated.change);
2156
- return applyJsonChange(surface, validated.change);
2157
- }
2158
- function routeFindingSubject(subject, opts) {
2159
- const parsed = parseFindingSubject(subject);
2160
- if (!parsed) {
2161
- return {
2162
- axis: opts.defaultAxis ?? "representation",
2163
- target: { surface: opts.defaultTargetSurface ?? "prompt" }
2164
- };
2165
- }
2166
- return routeParsedSubject(parsed);
2167
- }
2168
- function routeParsedSubject(subject) {
2169
- switch (subject.kind) {
2170
- case "system-prompt":
2171
- return {
2172
- axis: "representation",
2173
- target: { surface: "prompt", path: `system-prompt:${subject.section}` }
2174
- };
2175
- case "skill":
2176
- return {
2177
- axis: "agent_profile",
2178
- target: { surface: "agent-profile", path: `skill:${subject.name}` }
2179
- };
2180
- case "tool-doc":
2181
- return {
2182
- axis: "tool_contract",
2183
- target: {
2184
- surface: "tool-contract",
2185
- path: subject.aspect ? `tool-doc:${subject.tool}:${subject.aspect}` : `tool-doc:${subject.tool}`
2186
- }
2187
- };
2188
- case "new-tool":
2189
- return {
2190
- axis: "tool_contract",
2191
- target: { surface: "tool-contract", path: `new-tool:${subject.name}` }
2192
- };
2193
- case "mcp":
2194
- return {
2195
- axis: "tool_contract",
2196
- target: {
2197
- surface: "agent-profile",
2198
- path: subject.tool ? `mcp:${subject.server}:${subject.tool}` : `mcp:${subject.server}`
2199
- }
2200
- };
2201
- case "hook":
2202
- return {
2203
- axis: "agent_profile",
2204
- target: { surface: "agent-profile", path: `hook:${subject.name}` }
2205
- };
2206
- case "subagent":
2207
- return {
2208
- axis: "routing",
2209
- target: { surface: "agent-profile", path: `subagent:${subject.name}` }
2210
- };
2211
- case "workflow":
2212
- return {
2213
- axis: "routing",
2214
- target: { surface: "runtime-config", path: `workflow:${subject.name}` }
2215
- };
2216
- case "rollout-policy":
2217
- return {
2218
- axis: rolloutPolicyAxis(subject.field),
2219
- target: { surface: "runtime-config", path: `rollout-policy:${subject.field}` }
2220
- };
2221
- case "agent-profile":
2222
- return {
2223
- axis: "agent_profile",
2224
- target: { surface: "agent-profile", path: `agent-profile:${subject.field}` }
2225
- };
2226
- case "code":
2227
- return {
2228
- axis: "representation",
2229
- target: { surface: "code", path: `code:${subject.path}` }
2230
- };
2231
- case "rag":
2232
- return {
2233
- axis: "memory",
2234
- target: { surface: "memory", path: `rag:${subject.corpus}:${subject.docId}` }
2235
- };
2236
- case "memory":
2237
- return { axis: "memory", target: { surface: "memory", path: `memory:${subject.key}` } };
2238
- case "scaffolding":
2239
- return {
2240
- axis: "routing",
2241
- target: { surface: "runtime-config", path: `scaffolding:${subject.concern}` }
2242
- };
2243
- case "output-schema":
2244
- return {
2245
- axis: "output_contract",
2246
- target: { surface: "runtime-config", path: `output-schema:${subject.field}` }
2247
- };
2248
- case "knowledge.wiki":
2249
- return {
2250
- axis: "memory",
2251
- target: {
2252
- surface: "memory",
2253
- path: `agent-knowledge:wiki:${subject.slug}${subject.heading ? `#${subject.heading}` : ""}`
2254
- }
2255
- };
2256
- case "knowledge.claim":
2257
- return {
2258
- axis: "memory",
2259
- target: { surface: "memory", path: `agent-knowledge:claim:${subject.topic}` }
2260
- };
2261
- case "knowledge.raw":
2262
- return {
2263
- axis: "memory",
2264
- target: { surface: "memory", path: `agent-knowledge:raw:${subject.sourceId}` }
2265
- };
2266
- case "knowledge.stale":
2267
- return {
2268
- axis: "memory",
2269
- target: { surface: "memory", path: `agent-knowledge:stale:${subject.slug}` }
2270
- };
2271
- case "websearch.outdated":
2272
- return {
2273
- axis: "memory",
2274
- target: { surface: "memory", path: `websearch:outdated:${subject.topic}` }
2275
- };
2276
- case "prior-run-summary":
2277
- return {
2278
- axis: "memory",
2279
- target: { surface: "memory", path: `prior-run-summary:${subject.topic}` }
2280
- };
2281
- case "cluster":
2282
- return { axis: "representation", target: { surface: "prompt", path: subject.label } };
2283
- }
2284
- }
2285
- function rolloutPolicyAxis(field) {
2286
- const normalized = field.toLowerCase();
2287
- if (/budget|max(?:imum)?[-_. ]?(?:turns?|tokens?|cost)|timeout|deadline/.test(normalized)) {
2288
- return "budget";
2289
- }
2290
- if (/temperature|top[-_. ]?p|sampling|seed|shots?|parallel|concurrency/.test(normalized)) {
2291
- return "sampling";
2292
- }
2293
- if (/output|schema|format/.test(normalized)) return "output_contract";
2294
- return "routing";
2295
- }
2296
- function resolveExpectedGain(finding, opts) {
2297
- if (typeof opts.expectedGain === "function") return opts.expectedGain(finding) ?? null;
2298
- if (opts.expectedGain) return opts.expectedGain;
2299
- return readExpectedGainFromMetadata(finding.metadata);
2300
- }
2301
- function readExpectedGainFromMetadata(metadata) {
2302
- const raw = readPolicyEditMetadata(metadata)?.expectedGain ?? readPolicyEditMetadata(metadata)?.expected_gain;
2303
- if (!raw || typeof raw !== "object") return null;
2304
- const obj = raw;
2305
- if (typeof obj.metric !== "string" || obj.direction !== "increase" && obj.direction !== "decrease" || typeof obj.amount !== "number") {
2306
- return null;
2307
- }
2308
- const out = {
2309
- metric: obj.metric,
2310
- direction: obj.direction,
2311
- amount: obj.amount
2312
- };
2313
- if (obj.unit === "absolute" || obj.unit === "relative" || obj.unit === "percent" || obj.unit === "score") {
2314
- out.unit = obj.unit;
2315
- }
2316
- if (typeof obj.rationale === "string") out.rationale = obj.rationale;
2317
- return out;
2318
- }
2319
- function readPolicyEditMetadata(metadata) {
2320
- const raw = metadata?.policyEdit ?? metadata?.policy_edit;
2321
- return raw && typeof raw === "object" ? raw : null;
2322
- }
2323
- function resolveRisk(finding, opts) {
2324
- if (typeof opts.risk === "function") return opts.risk(finding);
2325
- if (opts.risk) return opts.risk;
2326
- const raw = readPolicyEditMetadata(finding.metadata)?.risk;
2327
- if (raw === "low" || raw === "medium" || raw === "high" || raw === "unknown") return raw;
2328
- if (finding.severity === "critical" || finding.severity === "high") return "medium";
2329
- return "low";
2330
- }
2331
- function applyTextChange(surface, change) {
2332
- if (typeof surface !== "string") {
2333
- throw new PolicyEditValidationError("text policy edits require a string surface", "change");
2334
- }
2335
- if (change.mode === "append") {
2336
- if (hasExactTextBlock(surface, change.value)) return surface;
2337
- return `${surface.trimEnd()}
2338
-
2339
- ${change.value}`.trimStart();
2340
- }
2341
- if (change.mode === "prepend") {
2342
- if (hasExactTextBlock(surface, change.value)) return surface;
2343
- return `${change.value}
2344
-
2345
- ${surface.trimStart()}`.trimEnd();
2346
- }
2347
- const find = expectNonEmpty(change.find, "change.find");
2348
- if (!surface.includes(find)) {
2349
- throw new PolicyEditValidationError("replace target not found in surface", "change.find");
2350
- }
2351
- return surface.replace(find, change.value);
2352
- }
2353
- function applyJsonChange(surface, change) {
2354
- const root = parseJsonSurface(surface);
2355
- const path = splitPath(change.path);
2356
- if (change.mode === "remove") return setJsonAtPath(root, path, void 0, "remove");
2357
- if (change.mode === "set") return setJsonAtPath(root, path, change.value ?? null, "set");
2358
- const prior = readJsonAtPath(root, path);
2359
- const merged = prior && typeof prior === "object" && !Array.isArray(prior) && change.value && typeof change.value === "object" && !Array.isArray(change.value) ? { ...prior, ...change.value } : change.value ?? null;
2360
- return setJsonAtPath(root, path, merged, "set");
2361
- }
2362
- function parseJsonSurface(surface) {
2363
- if (typeof surface === "string") {
2364
- try {
2365
- return JSON.parse(surface);
2366
- } catch {
2367
- throw new PolicyEditValidationError(
2368
- "json policy edits require a JSON string surface",
2369
- "change"
2370
- );
2371
- }
2372
- }
2373
- assertJson(surface, "surface");
2374
- return surface;
2375
- }
2376
- function readJsonAtPath(root, path) {
2377
- let cursor = root;
2378
- for (const part of path) {
2379
- if (!cursor || typeof cursor !== "object" || Array.isArray(cursor)) return void 0;
2380
- cursor = cursor[part];
2381
- }
2382
- return cursor;
2383
- }
2384
- function setJsonAtPath(root, path, value, mode) {
2385
- if (path.length === 0) {
2386
- if (mode === "remove") return null;
2387
- return value ?? null;
2388
- }
2389
- if (root === null || typeof root !== "object" || Array.isArray(root)) {
2390
- throw new PolicyEditValidationError("json edit root must be an object", "change.path");
2391
- }
2392
- const out = { ...root };
2393
- let cursor = out;
2394
- for (let i = 0; i < path.length - 1; i++) {
2395
- const key = path[i];
2396
- const existing = cursor[key];
2397
- if (mode === "remove" && (!existing || typeof existing !== "object" || Array.isArray(existing))) {
2398
- return out;
2399
- }
2400
- const next = existing && typeof existing === "object" && !Array.isArray(existing) ? { ...existing } : {};
2401
- cursor[key] = next;
2402
- cursor = next;
2403
- }
2404
- const leaf = path[path.length - 1];
2405
- if (mode === "remove") delete cursor[leaf];
2406
- else cursor[leaf] = value ?? null;
2407
- return out;
2408
- }
2409
- function normalizePolicyEdit(input) {
2410
- const out = {
2411
- schemaVersion: "policy-edit/v1",
2412
- axis: input.axis,
2413
- target: normalizeTarget(input.target),
2414
- change: normalizeChange(input.change),
2415
- claim: input.claim.trim(),
2416
- expectedGain: normalizeExpectedGain(input.expectedGain),
2417
- confidence: input.confidence,
2418
- risk: input.risk,
2419
- source: normalizeSource(input.source)
2420
- };
2421
- if (input.rationale?.trim()) out.rationale = input.rationale.trim();
2422
- if (input.validationPlan?.trim()) out.validationPlan = input.validationPlan.trim();
2423
- if (input.metadata) out.metadata = input.metadata;
2424
- return out;
2425
- }
2426
- function assertJsonSafe(value, path, ancestors = /* @__PURE__ */ new WeakSet()) {
2427
- if (value === null || typeof value === "string" || typeof value === "boolean") return;
2428
- if (typeof value === "number") {
2429
- if (Number.isFinite(value)) return;
2430
- throw new PolicyEditValidationError("expected finite JSON number", path);
2431
- }
2432
- if (typeof value !== "object") {
2433
- throw new PolicyEditValidationError("expected JSON-safe value", path);
2434
- }
2435
- if (ancestors.has(value)) {
2436
- throw new PolicyEditValidationError("cyclic value is not JSON-safe", path);
2437
- }
2438
- ancestors.add(value);
2439
- if (Array.isArray(value)) {
2440
- for (let i = 0; i < value.length; i++) {
2441
- if (!(i in value)) {
2442
- throw new PolicyEditValidationError("sparse array is not JSON-safe", `${path}.${i}`);
2443
- }
2444
- assertJsonSafe(value[i], `${path}.${i}`, ancestors);
2445
- }
2446
- } else {
2447
- const prototype = Object.getPrototypeOf(value);
2448
- if (prototype !== Object.prototype && prototype !== null) {
2449
- throw new PolicyEditValidationError("expected plain JSON object", path);
2450
- }
2451
- if (Object.getOwnPropertySymbols(value).length > 0) {
2452
- throw new PolicyEditValidationError("symbol keys are not JSON-safe", path);
2453
- }
2454
- for (const [key, child] of Object.entries(value)) {
2455
- assertJsonSafe(child, `${path}.${key}`, ancestors);
2456
- }
2457
- }
2458
- ancestors.delete(value);
2459
- }
2460
- function normalizeTarget(target) {
2461
- const out = { surface: target.surface };
2462
- if (target.path?.trim()) out.path = target.path.trim();
2463
- if (target.agentProfileCell)
2464
- out.agentProfileCell = validateAgentProfileCell(target.agentProfileCell);
2465
- if (target.label?.trim()) out.label = target.label.trim();
2466
- return out;
2467
- }
2468
- function normalizeChange(change) {
2469
- if (change.kind === "text") {
2470
- const out2 = {
2471
- kind: "text",
2472
- mode: change.mode,
2473
- value: change.value.trim()
2474
- };
2475
- if (change.find?.trim()) out2.find = change.find.trim();
2476
- return out2;
2477
- }
2478
- const out = {
2479
- kind: "json",
2480
- mode: change.mode,
2481
- path: change.path.trim()
2482
- };
2483
- if (change.value !== void 0) out.value = change.value;
2484
- return out;
2485
- }
2486
- function normalizeExpectedGain(gain) {
2487
- const out = {
2488
- metric: gain.metric.trim(),
2489
- direction: gain.direction,
2490
- amount: gain.amount
2491
- };
2492
- if (gain.unit) out.unit = gain.unit;
2493
- if (gain.rationale?.trim()) out.rationale = gain.rationale.trim();
2494
- return out;
2495
- }
2496
- function normalizeSource(source) {
2497
- const out = {
2498
- findingIds: uniqueSorted(source.findingIds.map((s) => s.trim()).filter(Boolean)),
2499
- analystIds: uniqueSorted(source.analystIds.map((s) => s.trim()).filter(Boolean)),
2500
- evidenceRefs: source.evidenceRefs
2501
- };
2502
- if (source.derivedFromJudge) out.derivedFromJudge = true;
2503
- return out;
2504
- }
2505
- function validateTarget(target) {
2506
- if (!target || typeof target !== "object")
2507
- throw new PolicyEditValidationError("expected object", "target");
2508
- const obj = target;
2509
- expectOneOf(obj.surface, POLICY_EDIT_TARGET_SURFACES, "target.surface");
2510
- if (obj.path !== void 0) expectString(obj.path, "target.path");
2511
- if (obj.label !== void 0) expectString(obj.label, "target.label");
2512
- if (obj.agentProfileCell !== void 0) validateAgentProfileCell(obj.agentProfileCell);
2513
- }
2514
- function validateChange(change) {
2515
- if (!change || typeof change !== "object")
2516
- throw new PolicyEditValidationError("expected object", "change");
2517
- const obj = change;
2518
- if (obj.kind !== "text" && obj.kind !== "json") {
2519
- throw new PolicyEditValidationError("kind must be text or json", "change.kind");
2520
- }
2521
- if (obj.kind === "text") {
2522
- expectOneOf(obj.mode, ["append", "prepend", "replace"], "change.mode");
2523
- expectString(obj.value, "change.value");
2524
- if (obj.mode === "replace") expectString(obj.find, "change.find");
2525
- return;
2526
- }
2527
- expectOneOf(obj.mode, ["set", "merge", "remove"], "change.mode");
2528
- expectString(obj.path, "change.path");
2529
- if (obj.value !== void 0) assertJson(obj.value, "change.value");
2530
- }
2531
- function validateExpectedGain(gain) {
2532
- if (!gain || typeof gain !== "object")
2533
- throw new PolicyEditValidationError("expected object", "expectedGain");
2534
- const obj = gain;
2535
- expectString(obj.metric, "expectedGain.metric");
2536
- expectOneOf(obj.direction, ["increase", "decrease"], "expectedGain.direction");
2537
- if (!Number.isFinite(obj.amount) || obj.amount <= 0) {
2538
- throw new PolicyEditValidationError(
2539
- "amount must be a positive finite number",
2540
- "expectedGain.amount"
2541
- );
2542
- }
2543
- if (obj.unit !== void 0) {
2544
- expectOneOf(
2545
- obj.unit,
2546
- ["absolute", "relative", "percent", "score"],
2547
- "expectedGain.unit"
2548
- );
2549
- }
2550
- if (obj.rationale !== void 0) expectString(obj.rationale, "expectedGain.rationale");
2551
- }
2552
- function validateSource(source) {
2553
- if (!source || typeof source !== "object")
2554
- throw new PolicyEditValidationError("expected object", "source");
2555
- const obj = source;
2556
- expectNonEmptyStringArray(obj.findingIds, "source.findingIds");
2557
- expectNonEmptyStringArray(obj.analystIds, "source.analystIds");
2558
- if (!Array.isArray(obj.evidenceRefs)) {
2559
- throw new PolicyEditValidationError("expected array", "source.evidenceRefs");
2560
- }
2561
- for (const [i, ref] of obj.evidenceRefs.entries())
2562
- validateEvidenceRef(ref, `source.evidenceRefs.${i}`);
2563
- if (obj.derivedFromJudge !== void 0 && typeof obj.derivedFromJudge !== "boolean") {
2564
- throw new PolicyEditValidationError("expected boolean", "source.derivedFromJudge");
2565
- }
2566
- }
2567
- function validateEvidenceRef(ref, path) {
2568
- if (!ref || typeof ref !== "object") throw new PolicyEditValidationError("expected object", path);
2569
- const obj = ref;
2570
- expectOneOf(obj.kind, ["span", "event", "artifact", "finding", "metric"], `${path}.kind`);
2571
- expectString(obj.uri, `${path}.uri`);
2572
- if (obj.excerpt !== void 0) expectString(obj.excerpt, `${path}.excerpt`);
2573
- }
2574
- function assertJson(value, path) {
2575
- if (value === null || typeof value === "string" || typeof value === "boolean" || typeof value === "number" && Number.isFinite(value)) {
2576
- return;
2577
- }
2578
- if (Array.isArray(value)) {
2579
- for (const [i, item] of value.entries()) assertJson(item, `${path}.${i}`);
2580
- return;
2581
- }
2582
- if (typeof value === "object") {
2583
- for (const [key, item] of Object.entries(value)) {
2584
- if (!key) throw new PolicyEditValidationError("empty object key", path);
2585
- assertJson(item, `${path}.${key}`);
2586
- }
2587
- return;
2588
- }
2589
- throw new PolicyEditValidationError("expected JSON-compatible value", path);
2590
- }
2591
- function targetSpecificityScore(edit) {
2592
- let score = 0.4;
2593
- if (edit.target.path) score += 0.25;
2594
- if (edit.target.agentProfileCell) score += 0.15;
2595
- if (edit.change.kind === "json" || edit.change.mode === "replace") score += 0.2;
2596
- else if (edit.change.value.length > 0) score += 0.1;
2597
- return clamp01(score);
2598
- }
2599
- function splitPath(path) {
2600
- const parts = path.split(".").map((p) => p.trim()).filter(Boolean);
2601
- if (parts.length === 0)
2602
- throw new PolicyEditValidationError("path must not be empty", "change.path");
2603
- return parts;
2604
- }
2605
- function expectLiteral(value, expected, path) {
2606
- if (value !== expected) throw new PolicyEditValidationError(`expected ${expected}`, path);
2607
- }
2608
- function expectString(value, path) {
2609
- if (typeof value !== "string" || value.trim().length === 0) {
2610
- throw new PolicyEditValidationError("expected non-empty string", path);
2611
- }
2612
- }
2613
- function expectNonEmpty(value, path) {
2614
- expectString(value, path);
2615
- return value;
2616
- }
2617
- function expectConfidence(value, path) {
2618
- if (typeof value !== "number" || !Number.isFinite(value) || value < 0 || value > 1) {
2619
- throw new PolicyEditValidationError("expected finite number in [0,1]", path);
2620
- }
2621
- }
2622
- function hasExactTextBlock(surface, value) {
2623
- const needle = normalizeTextBlock(value);
2624
- const normalizedSurface = surface.replace(/\r\n/g, "\n");
2625
- return [...normalizedSurface.split(/\n{2,}/), ...normalizedSurface.split("\n")].some(
2626
- (block) => normalizeTextBlock(block) === needle
2627
- );
2628
- }
2629
- function normalizeTextBlock(value) {
2630
- return value.replace(/\r\n/g, "\n").trim();
2631
- }
2632
- function expectOneOf(value, allowed, path) {
2633
- if (typeof value !== "string" || !allowed.includes(value)) {
2634
- throw new PolicyEditValidationError(`expected one of ${allowed.join(", ")}`, path);
2635
- }
2636
- }
2637
- function expectStringArray(value, path) {
2638
- if (!Array.isArray(value)) throw new PolicyEditValidationError("expected array", path);
2639
- for (const [i, item] of value.entries()) expectString(item, `${path}.${i}`);
2640
- }
2641
- function expectNonEmptyStringArray(value, path) {
2642
- expectStringArray(value, path);
2643
- if (value.length === 0) throw new PolicyEditValidationError("expected non-empty array", path);
2644
- }
2645
- function uniqueSorted(values) {
2646
- return [...new Set(values)].sort();
2647
- }
2648
- function clamp01(n) {
2649
- if (!Number.isFinite(n)) return 0;
2650
- if (n < 0) return 0;
2651
- if (n > 1) return 1;
2652
- return n;
2653
- }
2654
-
2655
- // src/run-score.ts
2656
- var DEFAULT_RUN_SCORE_WEIGHTS = {
2657
- success: 4,
2658
- goalProgress: 2,
2659
- repoGroundedness: 1.5,
2660
- driftPenalty: -1.5,
2661
- toolUseQuality: 1,
2662
- patchQuality: 1.25,
2663
- testReality: 1.5,
2664
- finalGate: 3,
2665
- reviewerBlockers: -2,
2666
- costUsd: -0.2,
2667
- wallSeconds: -0.1
2668
- };
2669
- function aggregateRunScore(score, weights = {}) {
2670
- const w = { ...DEFAULT_RUN_SCORE_WEIGHTS, ...weights };
2671
- return w.success * clamp012(score.success) + w.goalProgress * clamp012(score.goalProgress) + w.repoGroundedness * clamp012(score.repoGroundedness) + w.driftPenalty * clamp012(score.driftPenalty) + w.toolUseQuality * clamp012(score.toolUseQuality) + w.patchQuality * clamp012(score.patchQuality) + w.testReality * clamp012(score.testReality) + w.finalGate * clamp012(score.finalGate) + w.reviewerBlockers * clamp012(score.reviewerBlockers) + w.costUsd * Math.max(0, finiteOrZero(score.costUsd)) + w.wallSeconds * Math.max(0, finiteOrZero(score.wallSeconds) / 60);
2672
- }
2673
- function clamp012(value) {
2674
- if (!Number.isFinite(value)) return 0;
2675
- return Math.max(0, Math.min(1, value));
2676
- }
2677
- function finiteOrZero(value) {
2678
- return Number.isFinite(value) ? value : 0;
2383
+ return registry;
2679
2384
  }
2680
2385
 
2681
2386
  export {
2682
2387
  createAnalystAi,
2388
+ createChatClient,
2389
+ computeTraceMetrics,
2683
2390
  computeFindingId,
2684
2391
  makeFinding,
2392
+ deriveEfficiencyFindings,
2393
+ behavioralAnalyst,
2685
2394
  FINDING_SUBJECT_KINDS,
2686
2395
  parseFindingSubject,
2687
2396
  renderFindingSubject,
@@ -2714,28 +2423,6 @@ export {
2714
2423
  KNOWLEDGE_POISONING_KIND_SPEC,
2715
2424
  DEFAULT_TRACE_ANALYST_KINDS,
2716
2425
  AnalystRegistry,
2717
- Mutex,
2718
- mapConcurrent,
2719
- isTraceObservable,
2720
- isJudgeVerdict,
2721
- assertNoJudgeVerdict,
2722
- POLICY_EDIT_AXES,
2723
- POLICY_EDIT_TARGET_SURFACES,
2724
- POLICY_EDIT_CANDIDATE_RECORD_SCHEMA,
2725
- PolicyEditValidationError,
2726
- makePolicyEdit,
2727
- computePolicyEditId,
2728
- validatePolicyEdit,
2729
- makePolicyEditCandidateRecord,
2730
- validatePolicyEditCandidateRecord,
2731
- isPolicyEdit,
2732
- policyEditsFromFindings,
2733
- policyEditFromFinding,
2734
- scorePolicyEditReadiness,
2735
- admitPolicyEdit,
2736
- applyPolicyEditToSurface,
2737
- DEFAULT_RUN_SCORE_WEIGHTS,
2738
- aggregateRunScore,
2739
- clamp012 as clamp01
2426
+ buildDefaultAnalystRegistry
2740
2427
  };
2741
- //# sourceMappingURL=chunk-DT7OXY3C.js.map
2428
+ //# sourceMappingURL=chunk-LUNF2SEL.js.map