@tangle-network/agent-eval 0.120.0 → 0.120.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (171) hide show
  1. package/CHANGELOG.md +12 -0
  2. package/package.json +1 -1
  3. package/dist/analyst/index.d.ts +0 -3111
  4. package/dist/analyst/index.js +0 -403
  5. package/dist/analyst/index.js.map +0 -1
  6. package/dist/authenticity/index.d.ts +0 -161
  7. package/dist/authenticity/index.js +0 -215
  8. package/dist/authenticity/index.js.map +0 -1
  9. package/dist/belief-state/index.d.ts +0 -1301
  10. package/dist/belief-state/index.js +0 -2152
  11. package/dist/belief-state/index.js.map +0 -1
  12. package/dist/benchmarks/index.d.ts +0 -974
  13. package/dist/benchmarks/index.js +0 -60
  14. package/dist/benchmarks/index.js.map +0 -1
  15. package/dist/builder-eval/index.d.ts +0 -695
  16. package/dist/builder-eval/index.js +0 -366
  17. package/dist/builder-eval/index.js.map +0 -1
  18. package/dist/campaign/index.d.ts +0 -7454
  19. package/dist/campaign/index.js +0 -272
  20. package/dist/campaign/index.js.map +0 -1
  21. package/dist/chunk-32BZXMSO.js +0 -3878
  22. package/dist/chunk-32BZXMSO.js.map +0 -1
  23. package/dist/chunk-3A246TSA.js +0 -998
  24. package/dist/chunk-3A246TSA.js.map +0 -1
  25. package/dist/chunk-3RF76KTD.js +0 -84
  26. package/dist/chunk-3RF76KTD.js.map +0 -1
  27. package/dist/chunk-3YYRZDON.js +0 -45
  28. package/dist/chunk-3YYRZDON.js.map +0 -1
  29. package/dist/chunk-4I2E3LLO.js +0 -1030
  30. package/dist/chunk-4I2E3LLO.js.map +0 -1
  31. package/dist/chunk-ARU2PZFM.js +0 -312
  32. package/dist/chunk-ARU2PZFM.js.map +0 -1
  33. package/dist/chunk-BOD4O7OF.js +0 -40
  34. package/dist/chunk-BOD4O7OF.js.map +0 -1
  35. package/dist/chunk-DPZAEKA6.js +0 -880
  36. package/dist/chunk-DPZAEKA6.js.map +0 -1
  37. package/dist/chunk-DTJ6QUQB.js +0 -131
  38. package/dist/chunk-DTJ6QUQB.js.map +0 -1
  39. package/dist/chunk-GGE4NNQT.js +0 -65
  40. package/dist/chunk-GGE4NNQT.js.map +0 -1
  41. package/dist/chunk-H5UD2323.js +0 -286
  42. package/dist/chunk-H5UD2323.js.map +0 -1
  43. package/dist/chunk-HHWE3POT.js +0 -94
  44. package/dist/chunk-HHWE3POT.js.map +0 -1
  45. package/dist/chunk-HKUCJ437.js +0 -787
  46. package/dist/chunk-HKUCJ437.js.map +0 -1
  47. package/dist/chunk-JHCHEVET.js +0 -274
  48. package/dist/chunk-JHCHEVET.js.map +0 -1
  49. package/dist/chunk-JHOJHHU7.js +0 -867
  50. package/dist/chunk-JHOJHHU7.js.map +0 -1
  51. package/dist/chunk-JM2SKQMS.js +0 -750
  52. package/dist/chunk-JM2SKQMS.js.map +0 -1
  53. package/dist/chunk-JN2FCO5W.js +0 -7958
  54. package/dist/chunk-JN2FCO5W.js.map +0 -1
  55. package/dist/chunk-K4DBDHLK.js +0 -158
  56. package/dist/chunk-K4DBDHLK.js.map +0 -1
  57. package/dist/chunk-K6N6XJJX.js +0 -306
  58. package/dist/chunk-K6N6XJJX.js.map +0 -1
  59. package/dist/chunk-MA6HLL3S.js +0 -65
  60. package/dist/chunk-MA6HLL3S.js.map +0 -1
  61. package/dist/chunk-MAZ26DC7.js +0 -99
  62. package/dist/chunk-MAZ26DC7.js.map +0 -1
  63. package/dist/chunk-MOXWMGPC.js +0 -577
  64. package/dist/chunk-MOXWMGPC.js.map +0 -1
  65. package/dist/chunk-NJC7U437.js +0 -626
  66. package/dist/chunk-NJC7U437.js.map +0 -1
  67. package/dist/chunk-NPCTHQIO.js +0 -91
  68. package/dist/chunk-NPCTHQIO.js.map +0 -1
  69. package/dist/chunk-ONWEPEDO.js +0 -57
  70. package/dist/chunk-ONWEPEDO.js.map +0 -1
  71. package/dist/chunk-OYZAPX5G.js +0 -1526
  72. package/dist/chunk-OYZAPX5G.js.map +0 -1
  73. package/dist/chunk-PC4UYEBM.js +0 -166
  74. package/dist/chunk-PC4UYEBM.js.map +0 -1
  75. package/dist/chunk-PICTDURQ.js +0 -766
  76. package/dist/chunk-PICTDURQ.js.map +0 -1
  77. package/dist/chunk-PJQFMIOX.js +0 -1182
  78. package/dist/chunk-PJQFMIOX.js.map +0 -1
  79. package/dist/chunk-PXD6ZFNY.js +0 -1107
  80. package/dist/chunk-PXD6ZFNY.js.map +0 -1
  81. package/dist/chunk-PXE2VKMX.js +0 -140
  82. package/dist/chunk-PXE2VKMX.js.map +0 -1
  83. package/dist/chunk-PZ5AY32C.js +0 -10
  84. package/dist/chunk-PZ5AY32C.js.map +0 -1
  85. package/dist/chunk-QBRSJK47.js +0 -622
  86. package/dist/chunk-QBRSJK47.js.map +0 -1
  87. package/dist/chunk-QWMPPZ3X.js +0 -550
  88. package/dist/chunk-QWMPPZ3X.js.map +0 -1
  89. package/dist/chunk-S3UZOQ5Y.js +0 -328
  90. package/dist/chunk-S3UZOQ5Y.js.map +0 -1
  91. package/dist/chunk-S5TT5R3L.js +0 -2668
  92. package/dist/chunk-S5TT5R3L.js.map +0 -1
  93. package/dist/chunk-T4SQEITX.js +0 -95
  94. package/dist/chunk-T4SQEITX.js.map +0 -1
  95. package/dist/chunk-TT4KNT67.js +0 -124
  96. package/dist/chunk-TT4KNT67.js.map +0 -1
  97. package/dist/chunk-U5CHZ5M3.js +0 -357
  98. package/dist/chunk-U5CHZ5M3.js.map +0 -1
  99. package/dist/chunk-ULOKLHIQ.js +0 -1937
  100. package/dist/chunk-ULOKLHIQ.js.map +0 -1
  101. package/dist/chunk-VI2UW6B6.js +0 -162
  102. package/dist/chunk-VI2UW6B6.js.map +0 -1
  103. package/dist/chunk-VQMK5FMP.js +0 -247
  104. package/dist/chunk-VQMK5FMP.js.map +0 -1
  105. package/dist/chunk-VSMTAMNK.js +0 -53
  106. package/dist/chunk-VSMTAMNK.js.map +0 -1
  107. package/dist/chunk-VZSRQ272.js +0 -149
  108. package/dist/chunk-VZSRQ272.js.map +0 -1
  109. package/dist/chunk-WW2A73HW.js +0 -159
  110. package/dist/chunk-WW2A73HW.js.map +0 -1
  111. package/dist/chunk-X4UCIOTZ.js +0 -136
  112. package/dist/chunk-X4UCIOTZ.js.map +0 -1
  113. package/dist/chunk-XDIRG3TO.js +0 -1266
  114. package/dist/chunk-XDIRG3TO.js.map +0 -1
  115. package/dist/chunk-XJYR7XFV.js +0 -317
  116. package/dist/chunk-XJYR7XFV.js.map +0 -1
  117. package/dist/chunk-ZET2UAYW.js +0 -89
  118. package/dist/chunk-ZET2UAYW.js.map +0 -1
  119. package/dist/chunk-ZZUXHH3R.js +0 -99
  120. package/dist/chunk-ZZUXHH3R.js.map +0 -1
  121. package/dist/cli.d.ts +0 -1
  122. package/dist/cli.js +0 -112
  123. package/dist/cli.js.map +0 -1
  124. package/dist/contract/index.d.ts +0 -4969
  125. package/dist/contract/index.js +0 -1653
  126. package/dist/contract/index.js.map +0 -1
  127. package/dist/control.d.ts +0 -1013
  128. package/dist/control.js +0 -34
  129. package/dist/control.js.map +0 -1
  130. package/dist/fuzz.d.ts +0 -759
  131. package/dist/fuzz.js +0 -714
  132. package/dist/fuzz.js.map +0 -1
  133. package/dist/hosted/index.d.ts +0 -730
  134. package/dist/hosted/index.js +0 -14
  135. package/dist/hosted/index.js.map +0 -1
  136. package/dist/index.d.ts +0 -16780
  137. package/dist/index.js +0 -12168
  138. package/dist/index.js.map +0 -1
  139. package/dist/matrix/index.d.ts +0 -155
  140. package/dist/matrix/index.js +0 -8
  141. package/dist/matrix/index.js.map +0 -1
  142. package/dist/meta-eval/index.d.ts +0 -1030
  143. package/dist/meta-eval/index.js +0 -417
  144. package/dist/meta-eval/index.js.map +0 -1
  145. package/dist/multishot/index.d.ts +0 -579
  146. package/dist/multishot/index.js +0 -589
  147. package/dist/multishot/index.js.map +0 -1
  148. package/dist/openapi.json +0 -992
  149. package/dist/pipelines/index.d.ts +0 -567
  150. package/dist/pipelines/index.js +0 -515
  151. package/dist/pipelines/index.js.map +0 -1
  152. package/dist/reporting.d.ts +0 -1277
  153. package/dist/reporting.js +0 -48
  154. package/dist/reporting.js.map +0 -1
  155. package/dist/rl.d.ts +0 -4092
  156. package/dist/rl.js +0 -1724
  157. package/dist/rl.js.map +0 -1
  158. package/dist/run-campaign-HNFPJET4.js +0 -14
  159. package/dist/run-campaign-HNFPJET4.js.map +0 -1
  160. package/dist/storyboard/index.d.ts +0 -279
  161. package/dist/storyboard/index.js +0 -767
  162. package/dist/storyboard/index.js.map +0 -1
  163. package/dist/trace-attributes.d.ts +0 -52
  164. package/dist/trace-attributes.js +0 -62
  165. package/dist/trace-attributes.js.map +0 -1
  166. package/dist/traces.d.ts +0 -2343
  167. package/dist/traces.js +0 -249
  168. package/dist/traces.js.map +0 -1
  169. package/dist/wire/index.d.ts +0 -1252
  170. package/dist/wire/index.js +0 -81
  171. package/dist/wire/index.js.map +0 -1
@@ -1,1937 +0,0 @@
1
- import {
2
- recordAggregateMeasurements,
3
- summarizeExecutionMeasurements
4
- } from "./chunk-H5UD2323.js";
5
- import {
6
- extractUsage,
7
- extractUsageFromSse
8
- } from "./chunk-PXE2VKMX.js";
9
- import {
10
- analyzeTraces
11
- } from "./chunk-WW2A73HW.js";
12
- import {
13
- applyToolSpanOtlpAttributes,
14
- compareSpanTime,
15
- firstStringAttr,
16
- projectOtlpFlatLine,
17
- spanEpochMillis,
18
- traceSpanKindToOpenInferenceKind
19
- } from "./chunk-PXD6ZFNY.js";
20
- import {
21
- defaultProviderRedactor,
22
- providerFromBaseUrl
23
- } from "./chunk-PC4UYEBM.js";
24
- import {
25
- validateRunRecord
26
- } from "./chunk-S3UZOQ5Y.js";
27
- import {
28
- canonicalize,
29
- hashJson
30
- } from "./chunk-VSMTAMNK.js";
31
- import {
32
- ReplayError
33
- } from "./chunk-ONWEPEDO.js";
34
- import {
35
- LLM_INPUT_TOKENS,
36
- LLM_MODEL_ATTR_KEYS,
37
- LLM_MODEL_NAME,
38
- LLM_OUTPUT_TOKENS,
39
- OPENINFERENCE_SPAN_KIND,
40
- TOOL_NAME,
41
- applyLlmSpanOtlpAttributes
42
- } from "./chunk-K4DBDHLK.js";
43
-
44
- // src/trace-analyst/hook.ts
45
- var DEFAULT_QUESTION = "Summarise what happened in this run. Surface any failure modes, surprising findings, or evidence that the run's verdict is wrong.";
46
- function traceAnalystOnRunComplete(opts) {
47
- return async (ctx) => {
48
- if (opts.shouldRun && !opts.shouldRun(ctx)) return;
49
- const source = opts.analyze.source;
50
- if (source === void 0) {
51
- await ctx.store.appendEvent({
52
- eventId: `analyst-skip-${ctx.runId}`,
53
- runId: ctx.runId,
54
- kind: "log",
55
- timestamp: Date.now(),
56
- payload: { source: "trace_analyst_hook", reason: "no source configured" }
57
- });
58
- return;
59
- }
60
- const result = await analyzeTraces({ question: opts.question ?? DEFAULT_QUESTION }, {
61
- ...opts.analyze,
62
- source
63
- });
64
- if (opts.save) await opts.save(result, ctx);
65
- if (opts.gateOn && !opts.gateOn(result, ctx)) {
66
- await ctx.store.appendEvent({
67
- eventId: `analyst-gate-${ctx.runId}`,
68
- runId: ctx.runId,
69
- kind: "log",
70
- timestamp: Date.now(),
71
- payload: {
72
- source: "trace_analyst_hook",
73
- reason: "analyst_gate_failed",
74
- findings: result.findings
75
- }
76
- });
77
- }
78
- };
79
- }
80
-
81
- // src/trace-analyst/insights.ts
82
- var DOMAIN_STOP_WORDS = /* @__PURE__ */ new Set([
83
- "and",
84
- "advanced",
85
- "app",
86
- "build",
87
- "create",
88
- "easy",
89
- "expert",
90
- "extreme",
91
- "for",
92
- "from",
93
- "hard",
94
- "implementation",
95
- "integrate",
96
- "medium",
97
- "project",
98
- "task",
99
- "the",
100
- "this",
101
- "with",
102
- "workflow"
103
- ]);
104
- function tokenizeDomainWords(value) {
105
- return [...value.matchAll(/[A-Za-z][A-Za-z0-9.+#-]{2,}/g)].map((match) => match[0].toLowerCase()).filter((word) => !DOMAIN_STOP_WORDS.has(word));
106
- }
107
- function inferDomainKeywords(suite) {
108
- const suiteWords = new Set(tokenizeDomainWords(`${suite.name} ${suite.collectionId ?? ""}`));
109
- const source = [
110
- suite.name,
111
- suite.collectionId ?? "",
112
- ...suite.tasks.flatMap((task) => [
113
- task.id,
114
- task.name,
115
- task.prompt ?? "",
116
- task.difficulty ?? "",
117
- ...task.tags ?? [],
118
- ...task.gaps ?? []
119
- ])
120
- ].join(" ");
121
- const counts = /* @__PURE__ */ new Map();
122
- for (const word of tokenizeDomainWords(source)) counts.set(word, (counts.get(word) ?? 0) + 1);
123
- return [...counts.entries()].filter(([word, count]) => count >= 2 || suiteWords.has(word)).sort((a, b) => b[1] - a[1] || a[0].localeCompare(b[0])).map(([word]) => word).slice(0, 18);
124
- }
125
- function domainEvidencePattern(keywords) {
126
- const escaped = keywords.filter((keyword) => keyword.length >= 3).map((keyword) => keyword.replace(/[.*+?^${}()|[\]\\]/g, "\\$&"));
127
- return escaped.length > 0 ? new RegExp(`(?<![A-Za-z0-9])(?:${escaped.join("|")})(?![A-Za-z0-9])`, "i") : /(?<![A-Za-z0-9])(?:sdk|api|css|dns|xml|provider|client|service|integration|webhook|transaction|auth|oauth|graphql|rest)(?![A-Za-z0-9])/i;
128
- }
129
- function describeTraceInsightScope(suite) {
130
- const taskLabel = suite.tasks.length === 1 ? "1 implementation task" : `${suite.tasks.length} implementation tasks`;
131
- const tags = /* @__PURE__ */ new Map();
132
- for (const task of suite.tasks) {
133
- for (const tag of task.tags ?? []) tags.set(tag, (tags.get(tag) ?? 0) + 1);
134
- }
135
- const topTags = [...tags.entries()].sort((a, b) => b[1] - a[1] || a[0].localeCompare(b[0])).slice(0, 8).map(([tag]) => tag);
136
- if (topTags.length > 0) return `${taskLabel} across ${topTags.join(", ")}.`;
137
- const difficulties = [
138
- ...new Set(
139
- suite.tasks.map((task) => task.difficulty).filter((value) => Boolean(value))
140
- )
141
- ].join(", ");
142
- return `${taskLabel} across ${difficulties || "the selected benchmark scope"}.`;
143
- }
144
- function planTraceInsightQuestions(input) {
145
- const hasFailures = input.suite.tasks.some((task) => task.outcome && task.outcome !== "satisfied");
146
- const hasMultipleShots = input.suite.tasks.some(
147
- (task) => (task.gaps ?? []).some((gap) => /shot|review|retry|continue/i.test(gap))
148
- );
149
- const questions = [
150
- {
151
- id: "execution-path",
152
- question: "What did the worker actually do before the first meaningful implementation edit?",
153
- why: "Separates grounded execution from polished but shallow output."
154
- },
155
- {
156
- id: "research-grounding",
157
- question: "Did the worker inspect docs, source, examples, or package references before committing to an implementation path?",
158
- why: "Identifies whether failures came from weak retrieval, weak examples, or premature coding."
159
- },
160
- {
161
- id: "domain-proof",
162
- question: "Which tasks produced executable domain proof versus UI copy, placeholders, or inferred behavior?",
163
- why: "Keeps product-quality claims tied to concrete evidence."
164
- },
165
- {
166
- id: "root-cause",
167
- question: "For each major failure cluster, is the likely root cause prompt/scaffold, docs/examples, SDK/API ergonomics, evaluator, runtime, or model behavior?",
168
- why: "Turns trace observations into actionable ownership."
169
- },
170
- {
171
- id: "evidence-quality",
172
- question: "Which external-facing claims are directly supported by trace ids, span ids, verifier findings, reviewer notes, or generated code?",
173
- why: "Prevents unsupported customer-report conclusions."
174
- }
175
- ];
176
- if (hasMultipleShots) {
177
- questions.push({
178
- id: "reviewer-lift",
179
- question: "Where did reviewer feedback improve score, stall, or regress across shots?",
180
- why: "Shows whether the driver loop is learning or merely repeating work."
181
- });
182
- }
183
- if (hasFailures) {
184
- questions.push({
185
- id: "optimization-targets",
186
- question: "Which prompt, evaluator, scaffold, or workflow changes should feed the next GEPA/autoresearch optimization run?",
187
- why: "Connects benchmark evidence to the optimization loop."
188
- });
189
- }
190
- return questions;
191
- }
192
- function buildTraceInsightContext(input) {
193
- return {
194
- suite: input.suite,
195
- scope: describeTraceInsightScope(input.suite),
196
- keywords: inferDomainKeywords(input.suite),
197
- questions: planTraceInsightQuestions(input),
198
- panel: defaultTraceInsightPanel(),
199
- findings: input.findings ?? [],
200
- agent: input.agent ?? null,
201
- totals: input.totals ?? null
202
- };
203
- }
204
- function scoreTraceInsightReadiness(context) {
205
- const failedTasks = context.suite.tasks.filter(
206
- (task) => task.outcome && task.outcome !== "satisfied"
207
- );
208
- const findingTaskIds = new Set(context.findings.flatMap((finding) => finding.taskIds));
209
- const failedTasksWithFindings = failedTasks.filter((task) => findingTaskIds.has(task.id));
210
- const tasksWithGaps = context.suite.tasks.filter((task) => (task.gaps ?? []).length > 0);
211
- const gates = [
212
- {
213
- id: "domain-context",
214
- label: "Domain context inferred",
215
- passed: context.keywords.length > 0,
216
- severity: "high",
217
- detail: context.keywords.length > 0 ? `${context.keywords.length} domain terms inferred: ${context.keywords.slice(0, 8).join(", ")}` : "No domain terms were inferred from suite, tasks, prompts, tags, or gaps."
218
- },
219
- {
220
- id: "panel-coverage",
221
- label: "Analyst panel planned",
222
- passed: context.panel.length >= 4 && context.questions.length >= 5,
223
- severity: "high",
224
- detail: `${context.panel.length} panel roles and ${context.questions.length} investigation questions planned.`
225
- },
226
- {
227
- id: "failure-coverage",
228
- label: "Failures mapped to findings",
229
- passed: failedTasks.length === 0 || failedTasksWithFindings.length / failedTasks.length >= 0.5,
230
- severity: "critical",
231
- detail: failedTasks.length === 0 ? "No failed tasks in suite." : `${failedTasksWithFindings.length}/${failedTasks.length} failed tasks appear in finding clusters.`
232
- },
233
- {
234
- id: "gap-evidence",
235
- label: "Task gaps captured",
236
- passed: failedTasks.length === 0 || tasksWithGaps.length / failedTasks.length >= 0.5,
237
- severity: "medium",
238
- detail: `${tasksWithGaps.length} tasks include explicit evaluator or analyst gaps.`
239
- }
240
- ];
241
- const penalty = gates.reduce((sum, gate) => {
242
- if (gate.passed) return sum;
243
- if (gate.severity === "critical") return sum + 35;
244
- if (gate.severity === "high") return sum + 20;
245
- if (gate.severity === "medium") return sum + 10;
246
- return sum + 5;
247
- }, 0);
248
- const score = Math.max(0, Math.min(1, 1 - penalty / 100));
249
- return {
250
- score,
251
- grade: score >= 0.9 ? "external-ready" : score >= 0.7 ? "internal-review" : "raw-analysis",
252
- gates
253
- };
254
- }
255
- function defaultTraceInsightPanel() {
256
- return [
257
- {
258
- id: "trace-forensics",
259
- name: "Trace Forensics",
260
- responsibility: "Reconstruct what the worker did in order, including research, edits, reviewer interventions, verifier feedback, and stop reason."
261
- },
262
- {
263
- id: "root-cause",
264
- name: "Root Cause",
265
- responsibility: "Map failures to prompt/scaffold, docs/examples, SDK/API/product ergonomics, evaluator, runtime, or model behavior."
266
- },
267
- {
268
- id: "optimization",
269
- name: "Optimization",
270
- responsibility: "Identify prompt, reviewer, evaluator, scaffold, and GEPA/autoresearch changes that should be tested next."
271
- },
272
- {
273
- id: "external-evidence",
274
- name: "External Evidence",
275
- responsibility: "Separate customer-safe claims from internal harness findings and reject conclusions without task, trace, span, code, reviewer, or verifier evidence."
276
- }
277
- ];
278
- }
279
- function buildTraceInsightPrompt(input) {
280
- const context = buildTraceInsightContext(input);
281
- const maxRepresentativeTraces = input.maxRepresentativeTraces ?? 6;
282
- return `Analyze this benchmark run and produce evidence-backed trace intelligence.
283
-
284
- Audience:
285
- - internal AI/product leadership
286
- - possible customer-facing report for ${input.suite.name}
287
-
288
- Investigation plan:
289
- ${context.questions.map((item, index) => `${index + 1}. ${item.question} (${item.why})`).join("\n")}
290
-
291
- Analyst panel:
292
- ${context.panel.map((role) => `- ${role.name}: ${role.responsibility}`).join("\n")}
293
-
294
- If the task branches are independent, use subagents for the panel roles above and aggregate their findings. Do not run a panel role unless its answer will change the final report.
295
-
296
- Required output:
297
- 1. Executive verdict: what this run proves and does not prove.
298
- 2. The investigation questions you answered and the evidence used.
299
- 3. Failure taxonomy: agent prompting, evaluator/harness, docs/examples, SDK/API/product integration, infra.
300
- 4. Evidence-backed examples with trace ids/task ids and concrete verifier findings.
301
- 5. Highest-ROI fixes for the benchmark harness, prompt/GEPA optimization, and customer-facing product/docs surface.
302
- 6. What is safe for an external report versus what must stay internal.
303
- 7. One rerun plan that would validate lift after optimization.
304
-
305
- Budget:
306
- - Inspect the dataset overview, the failure summary, and at most ${maxRepresentativeTraces} representative traces.
307
- - Prefer traces named in the failure summary over broad exploration.
308
- - Do not do exhaustive trace sweeps.
309
- - Return the final report as soon as the taxonomy and examples are supported.
310
-
311
- Run summary:
312
- ${JSON.stringify(
313
- {
314
- suite: input.suite.name,
315
- scope: context.scope,
316
- inferredKeywords: context.keywords,
317
- agent: context.agent,
318
- totals: context.totals,
319
- findings: context.findings.map((finding) => ({
320
- kind: finding.kind,
321
- severity: finding.severity,
322
- taskCount: finding.taskIds.length,
323
- proposedFixClass: finding.proposedFixClass
324
- })),
325
- failures: input.suite.tasks.filter((task) => task.outcome && task.outcome !== "satisfied").map((task) => ({
326
- task: task.id,
327
- difficulty: task.difficulty,
328
- outcome: task.outcome,
329
- score: task.score,
330
- gaps: task.gaps ?? []
331
- }))
332
- },
333
- null,
334
- 2
335
- )}
336
-
337
- Use the trace tools. Do not invent facts. Cite task ids. Separate customer-facing claims from internal harness/model findings.`;
338
- }
339
-
340
- // src/trace-analyst/otlp-flatten.ts
341
- var DEFAULT_KIND_MAP = {
342
- 0: "SPAN_KIND_UNSPECIFIED",
343
- 1: "SPAN_KIND_INTERNAL",
344
- 2: "SPAN_KIND_SERVER",
345
- 3: "SPAN_KIND_CLIENT",
346
- 4: "SPAN_KIND_PRODUCER",
347
- 5: "SPAN_KIND_CONSUMER"
348
- };
349
- var STATUS_MAP = {
350
- 0: "STATUS_CODE_UNSET",
351
- 1: "STATUS_CODE_OK",
352
- 2: "STATUS_CODE_ERROR"
353
- };
354
- function attrValue(v) {
355
- if (v.stringValue !== void 0) return v.stringValue;
356
- if (v.intValue !== void 0) return Number(v.intValue);
357
- if (v.doubleValue !== void 0) return v.doubleValue;
358
- if (v.boolValue !== void 0) return v.boolValue;
359
- return "";
360
- }
361
- function attrsToRecord(attrs) {
362
- const out = {};
363
- for (const a of attrs) out[a.key] = attrValue(a.value);
364
- return out;
365
- }
366
- function nanoToIso(nano) {
367
- const ms = Number(nano) / 1e6;
368
- return Number.isFinite(ms) ? new Date(ms).toISOString() : (/* @__PURE__ */ new Date(0)).toISOString();
369
- }
370
- function applyOpenInference(attrs) {
371
- if ("llm.model" in attrs && !(LLM_MODEL_NAME in attrs)) {
372
- attrs[LLM_MODEL_NAME] = attrs["llm.model"];
373
- }
374
- if ("llm.input_tokens" in attrs && !(LLM_INPUT_TOKENS in attrs)) {
375
- attrs[LLM_INPUT_TOKENS] = attrs["llm.input_tokens"];
376
- }
377
- if ("inference.llm.input_tokens" in attrs && !(LLM_INPUT_TOKENS in attrs)) {
378
- attrs[LLM_INPUT_TOKENS] = attrs["inference.llm.input_tokens"];
379
- }
380
- if ("llm.output_tokens" in attrs && !(LLM_OUTPUT_TOKENS in attrs)) {
381
- attrs[LLM_OUTPUT_TOKENS] = attrs["llm.output_tokens"];
382
- }
383
- if ("inference.llm.output_tokens" in attrs && !(LLM_OUTPUT_TOKENS in attrs)) {
384
- attrs[LLM_OUTPUT_TOKENS] = attrs["inference.llm.output_tokens"];
385
- }
386
- if (TOOL_NAME in attrs && !("inference.tool.name" in attrs)) {
387
- attrs["inference.tool.name"] = attrs[TOOL_NAME];
388
- }
389
- if ("span.kind" in attrs && !(OPENINFERENCE_SPAN_KIND in attrs)) {
390
- attrs[OPENINFERENCE_SPAN_KIND] = String(attrs["span.kind"]).toUpperCase();
391
- }
392
- }
393
- function flattenOtlpExportToNdjson(otlpExport, opts = {}) {
394
- const vocab = opts.attributeVocabulary ?? "openinference";
395
- const kindMap = { ...DEFAULT_KIND_MAP, ...opts.kindMap };
396
- const lines = [];
397
- for (const rs of otlpExport.resourceSpans ?? []) {
398
- const resource = { attributes: attrsToRecord(rs.resource?.attributes ?? []) };
399
- for (const scope of rs.scopeSpans ?? []) {
400
- for (const span of scope.spans ?? []) {
401
- const attributes = attrsToRecord(span.attributes ?? []);
402
- if (vocab === "openinference") applyOpenInference(attributes);
403
- const line = {
404
- trace_id: span.traceId,
405
- span_id: span.spanId,
406
- parent_span_id: span.parentSpanId ?? null,
407
- name: span.name,
408
- kind: kindMap[span.kind] ?? "SPAN_KIND_UNSPECIFIED",
409
- start_time: nanoToIso(span.startTimeUnixNano),
410
- end_time: nanoToIso(span.endTimeUnixNano),
411
- status: {
412
- code: STATUS_MAP[span.status?.code ?? 0] ?? "STATUS_CODE_UNSET",
413
- ...span.status?.message ? { message: span.status.message } : {}
414
- },
415
- resource,
416
- attributes
417
- };
418
- if (span.events && span.events.length > 0) {
419
- line.events = span.events.map((e) => ({
420
- name: e.name,
421
- timeUnixNano: e.timeUnixNano,
422
- ...e.attributes ? { attributes: attrsToRecord(e.attributes) } : {}
423
- }));
424
- }
425
- lines.push(line);
426
- }
427
- }
428
- }
429
- return lines;
430
- }
431
-
432
- // src/trace-analyst/otlp-to-run-records.ts
433
- function otlpToRunRecords(otlpJsonl, opts) {
434
- return otlpToTraceRunRecords(otlpJsonl, opts).map((r) => r.record);
435
- }
436
- function otlpRowsToRunRecords(rows, opts) {
437
- return otlpRowsToTraceRunRecords(rows, opts).map((row) => row.record);
438
- }
439
- function otlpToTraceRunRecords(otlpJsonl, opts) {
440
- return traceRunRecordsFromSpans(
441
- groupSpansByLogicalRun(groupJsonlSpansByTrace(otlpJsonl), opts.logicalRunIdForTrace),
442
- opts
443
- );
444
- }
445
- function otlpRowsToTraceRunRecords(rows, opts) {
446
- return traceRunRecordsFromSpans(
447
- groupSpansByLogicalRun(groupRowsByTrace(rows), opts.logicalRunIdForTrace),
448
- opts
449
- );
450
- }
451
- function traceRunRecordsFromSpans(byTrace, opts) {
452
- const splitTag = opts.splitTag ?? "holdout";
453
- const commitSha = opts.commitSha ?? "unknown";
454
- const promptHash = opts.promptHash ?? "unknown";
455
- const configHash = opts.configHash ?? "unknown";
456
- const seed = opts.seed ?? 0;
457
- const fallbackModel = opts.fallbackModel ?? "unknown@otlp";
458
- if (byTrace.size === 0) {
459
- throw new Error(
460
- "otlpToRunRecords: OTLP input produced zero valid spans \u2014 every row was empty, malformed, or missing trace_id/span_id"
461
- );
462
- }
463
- const traceIds = [...byTrace.keys()].sort();
464
- const out = [];
465
- for (const traceId of traceIds) {
466
- const spans = byTrace.get(traceId);
467
- const agg = aggregateTrace(traceId, spans, fallbackModel);
468
- const score = resolveScore(opts, traceId, agg);
469
- const { costUsd, costProvenance } = resolveCost(opts, agg);
470
- const raw = {
471
- source_trace_count: agg.sourceTraceCount,
472
- span_count: agg.spanCount,
473
- llm_span_count: agg.llmSpanCount,
474
- tool_span_count: agg.toolSpanCount,
475
- agent_span_count: agg.agentSpanCount,
476
- error_span_count: agg.errorSpanCount,
477
- prompt_tokens: agg.tokenUsage.input,
478
- completion_tokens: agg.tokenUsage.output
479
- };
480
- if (agg.tokenUsage.reasoning !== void 0) raw.reasoning_tokens = agg.tokenUsage.reasoning;
481
- if (agg.tokenUsage.cached !== void 0) raw.cached_tokens = agg.tokenUsage.cached;
482
- if (agg.tokenUsage.cacheWrite !== void 0) {
483
- raw.cache_write_tokens = agg.tokenUsage.cacheWrite;
484
- }
485
- if (agg.costMeasurement.value !== void 0 && !agg.costMeasurement.complete) {
486
- raw.partial_observed_cost_usd = agg.costMeasurement.value;
487
- }
488
- recordAggregateMeasurements(raw, agg.aggregateMeasurement);
489
- if (costProvenance.kind === "uncaptured") raw.cost_unpriced = 1;
490
- const outcome = splitTag === "holdout" ? { holdoutScore: score, raw } : { searchScore: score, raw };
491
- const { promptText, completionText } = extractPromptCompletion(spans, agg.callSpanIds);
492
- const judgeMetadata = opts.judgeMetadataForTrace?.(traceId);
493
- const record = validateRunRecord({
494
- runId: `otlp:${opts.experimentId}:${opts.candidateId}:${traceId}`,
495
- experimentId: opts.experimentId,
496
- candidateId: opts.candidateId,
497
- seed,
498
- model: ensureSnapshot(agg.model, fallbackModel),
499
- promptHash,
500
- configHash,
501
- commitSha,
502
- wallMs: agg.wallMs,
503
- costUsd,
504
- costProvenance,
505
- tokenUsage: agg.tokenUsage,
506
- ...judgeMetadata ? { judgeMetadata } : {},
507
- outcome,
508
- ...agg.firstErrorMessage ? { failureMode: agg.firstErrorMessage } : {},
509
- splitTag,
510
- scenarioId: traceId
511
- });
512
- out.push({
513
- record,
514
- ...promptText !== void 0 ? { promptText } : {},
515
- ...completionText !== void 0 ? { completionText } : {}
516
- });
517
- }
518
- return out;
519
- }
520
- function* yieldJsonlRows(otlpJsonl) {
521
- for (const line of otlpJsonl.split("\n")) {
522
- const trimmed = line.trim();
523
- if (trimmed.length === 0) continue;
524
- let parsed;
525
- try {
526
- parsed = JSON.parse(trimmed);
527
- } catch {
528
- continue;
529
- }
530
- if (parsed && typeof parsed === "object") yield parsed;
531
- }
532
- }
533
- function groupJsonlSpansByTrace(otlpJsonl) {
534
- return groupRowsByTrace(yieldJsonlRows(otlpJsonl));
535
- }
536
- function groupRowsByTrace(rows) {
537
- const byTrace = /* @__PURE__ */ new Map();
538
- for (const row of rows) {
539
- if (!row || typeof row !== "object") continue;
540
- const span = projectOtlpFlatLine(row);
541
- if (!span) continue;
542
- const arr = byTrace.get(span.trace_id);
543
- if (arr) arr.push(span);
544
- else byTrace.set(span.trace_id, [span]);
545
- }
546
- return byTrace;
547
- }
548
- function groupSpansByLogicalRun(byTrace, logicalRunIdForTrace) {
549
- if (!logicalRunIdForTrace) return byTrace;
550
- const byRun = /* @__PURE__ */ new Map();
551
- for (const [traceId, spans] of byTrace) {
552
- const suppliedRunId = logicalRunIdForTrace(traceId);
553
- if (typeof suppliedRunId !== "string" || suppliedRunId.trim().length === 0) {
554
- throw new Error(
555
- `otlpToRunRecords: logicalRunIdForTrace('${traceId}') returned an empty run id`
556
- );
557
- }
558
- const runId = suppliedRunId.trim();
559
- const target = byRun.get(runId) ?? [];
560
- for (const span of spans) {
561
- target.push({
562
- ...span,
563
- span_id: qualifySpanId(traceId, span.span_id),
564
- parent_span_id: span.parent_span_id ? qualifySpanId(traceId, span.parent_span_id) : null
565
- });
566
- }
567
- byRun.set(runId, target);
568
- }
569
- return byRun;
570
- }
571
- function qualifySpanId(traceId, spanId) {
572
- const prefix = `${traceId}:`;
573
- return spanId.startsWith(prefix) ? spanId : `${prefix}${spanId}`;
574
- }
575
- function aggregateTrace(traceId, spans, fallbackModel) {
576
- const ordered = [...spans].sort(
577
- (a2, b2) => compareSpanTime(a2.start_time, b2.start_time) || a2.span_id.localeCompare(b2.span_id)
578
- );
579
- const measurements = summarizeExecutionMeasurements(
580
- ordered.map((span) => ({
581
- id: span.span_id,
582
- ...span.parent_span_id ? { parentId: span.parent_span_id } : {},
583
- attributes: span.attributes,
584
- modelCall: span.kind === "LLM" || span.kind === "UNKNOWN" && (span.model_name !== null || typeof span.attributes["gen_ai.operation.name"] === "string"),
585
- aggregate: span.kind !== "LLM" && span.kind !== "UNKNOWN"
586
- }))
587
- );
588
- let toolSpanCount = 0;
589
- let agentSpanCount = 0;
590
- let errorSpanCount = 0;
591
- let firstErrorMessage;
592
- const modelVotes = /* @__PURE__ */ new Map();
593
- let earliest = ordered[0]?.start_time ?? "";
594
- let latest = ordered[0]?.end_time ?? "";
595
- for (const s of ordered) {
596
- if (s.start_time && (!earliest || compareSpanTime(s.start_time, earliest) < 0))
597
- earliest = s.start_time;
598
- if (s.end_time && (!latest || compareSpanTime(s.end_time, latest) > 0)) latest = s.end_time;
599
- if (s.kind === "TOOL") {
600
- toolSpanCount += 1;
601
- } else if (s.kind === "AGENT") {
602
- agentSpanCount += 1;
603
- }
604
- if (s.status === "ERROR") {
605
- errorSpanCount += 1;
606
- if (firstErrorMessage === void 0) {
607
- firstErrorMessage = (s.status_message ?? `${s.name} \u2014 STATUS_CODE_ERROR`).slice(0, 500);
608
- }
609
- }
610
- }
611
- const callSpanIds = new Set(measurements.callSpanIds);
612
- for (const span of ordered) {
613
- if (!callSpanIds.has(span.span_id)) continue;
614
- const model2 = firstStringAttr(span.attributes, LLM_MODEL_ATTR_KEYS) ?? span.model_name;
615
- if (model2) modelVotes.set(model2, (modelVotes.get(model2) ?? 0) + 1);
616
- }
617
- const model = topVote(modelVotes) ?? firstModelAttr(ordered) ?? fallbackModel;
618
- let wallMs = 0;
619
- const a = spanEpochMillis(earliest);
620
- const b = spanEpochMillis(latest);
621
- if (a !== null && b !== null) wallMs = Math.max(0, b - a);
622
- const sourceTraceIds = [...new Set(spans.map((span) => span.trace_id))].sort();
623
- return {
624
- traceId,
625
- sourceTraceCount: sourceTraceIds.length,
626
- sourceTraceIds,
627
- spanCount: spans.length,
628
- llmSpanCount: measurements.modelCallCount,
629
- toolSpanCount,
630
- agentSpanCount,
631
- errorSpanCount,
632
- tokenUsage: measurements.tokenUsage,
633
- firstErrorMessage,
634
- model,
635
- startTime: earliest,
636
- endTime: latest,
637
- wallMs,
638
- callSpanIds: measurements.callSpanIds,
639
- costMeasurement: measurements.cost,
640
- ...measurements.aggregate ? { aggregateMeasurement: measurements.aggregate } : {}
641
- };
642
- }
643
- function resolveScore(opts, traceId, agg) {
644
- const supplied = opts.scoreForTrace?.(traceId, agg);
645
- if (supplied !== void 0) {
646
- if (!Number.isFinite(supplied)) {
647
- throw new Error(
648
- `otlpToRunRecords: scoreForTrace('${traceId}') returned non-finite ${supplied}`
649
- );
650
- }
651
- return supplied;
652
- }
653
- return agg.errorSpanCount > 0 ? 0 : 1;
654
- }
655
- function resolveCost(opts, agg) {
656
- const observedCost = agg.costMeasurement;
657
- if (observedCost.complete && observedCost.value !== void 0) {
658
- return {
659
- costUsd: observedCost.value,
660
- costProvenance: { kind: "observed", usd: observedCost.value }
661
- };
662
- }
663
- if (agg.aggregateMeasurement?.costUsd !== void 0) {
664
- return {
665
- costUsd: agg.aggregateMeasurement.costUsd,
666
- costProvenance: { kind: "observed", usd: agg.aggregateMeasurement.costUsd }
667
- };
668
- }
669
- if (opts.priceUsdPerToken !== void 0) {
670
- const totalTokens = agg.tokenUsage.input + agg.tokenUsage.output;
671
- const costUsd = totalTokens * opts.priceUsdPerToken;
672
- return { costUsd, costProvenance: { kind: "estimated", usd: costUsd } };
673
- }
674
- return { costUsd: 0, costProvenance: { kind: "uncaptured", usd: null } };
675
- }
676
- function extractPromptCompletion(spans, callSpanIds) {
677
- const callIds = new Set(callSpanIds);
678
- const measuredCalls = spans.filter((span) => callIds.has(span.span_id));
679
- const llm = (measuredCalls.length > 0 ? measuredCalls : spans.filter((s) => s.kind === "LLM")).sort(
680
- (a, b) => compareSpanTime(a.start_time, b.start_time) || a.span_id.localeCompare(b.span_id)
681
- );
682
- if (llm.length === 0) return {};
683
- const promptText = firstStringAttr(llm[0].attributes, ["input.value", "llm.input_messages", "gen_ai.prompt"]) ?? void 0;
684
- const last = llm[llm.length - 1];
685
- const completionText = firstStringAttr(last.attributes, [
686
- "output.value",
687
- "llm.output_messages",
688
- "gen_ai.completion"
689
- ]) ?? void 0;
690
- return {
691
- ...promptText !== void 0 ? { promptText } : {},
692
- ...completionText !== void 0 ? { completionText } : {}
693
- };
694
- }
695
- function topVote(votes) {
696
- let best = null;
697
- let bestN = 0;
698
- for (const [k, n] of votes) {
699
- if (n > bestN || n === bestN && best !== null && k < best) {
700
- best = k;
701
- bestN = n;
702
- }
703
- }
704
- return best;
705
- }
706
- function firstModelAttr(spans) {
707
- for (const s of spans) {
708
- const m = firstStringAttr(s.attributes, LLM_MODEL_ATTR_KEYS) ?? s.model_name;
709
- if (m) return m;
710
- }
711
- return null;
712
- }
713
- function ensureSnapshot(model, fallbackModel) {
714
- if (modelHasSnapshot(model)) return model;
715
- const fallbackTag = fallbackModel.includes("@") ? fallbackModel.slice(fallbackModel.indexOf("@")) : "@otlp";
716
- return `${model}${fallbackTag}`;
717
- }
718
- function modelHasSnapshot(model) {
719
- if (model.includes("@")) return true;
720
- if (/-\d{8}$/.test(model)) return true;
721
- if (/-\d{4}-\d{2}-\d{2}$/.test(model)) return true;
722
- if (/:date-/.test(model)) return true;
723
- return false;
724
- }
725
-
726
- // src/trace/store.ts
727
- var InMemoryTraceStore = class {
728
- runs = /* @__PURE__ */ new Map();
729
- allSpans = [];
730
- allEvents = [];
731
- allArtifacts = [];
732
- allBudget = [];
733
- async appendRun(run) {
734
- if (this.runs.has(run.runId)) throw new Error(`run ${run.runId} already exists`);
735
- this.runs.set(run.runId, { ...run });
736
- }
737
- async updateRun(runId, patch) {
738
- const existing = this.runs.get(runId);
739
- if (!existing) throw new Error(`run ${runId} not found`);
740
- this.runs.set(runId, { ...existing, ...patch });
741
- }
742
- async appendSpan(span) {
743
- this.allSpans.push({ ...span });
744
- }
745
- async updateSpan(spanId, patch) {
746
- const idx = this.allSpans.findIndex((s) => s.spanId === spanId);
747
- if (idx < 0) throw new Error(`span ${spanId} not found`);
748
- this.allSpans[idx] = { ...this.allSpans[idx], ...patch };
749
- }
750
- async appendEvent(event) {
751
- this.allEvents.push({ ...event });
752
- }
753
- async appendArtifact(artifact) {
754
- this.allArtifacts.push({ ...artifact });
755
- }
756
- async appendBudgetEntry(entry) {
757
- this.allBudget.push({ ...entry });
758
- }
759
- async getRun(runId) {
760
- const r = this.runs.get(runId);
761
- return r ? { ...r } : void 0;
762
- }
763
- async listRuns(filter = {}) {
764
- return [...this.runs.values()].filter((r) => matchesRun(r, filter));
765
- }
766
- async spans(filter = {}) {
767
- return this.allSpans.filter((s) => matchesSpan(s, filter)).map((s) => ({ ...s }));
768
- }
769
- async events(filter = {}) {
770
- return this.allEvents.filter((e) => matchesEvent(e, filter)).map((e) => ({ ...e }));
771
- }
772
- async budget(runId) {
773
- return this.allBudget.filter((b) => b.runId === runId).map((b) => ({ ...b }));
774
- }
775
- async artifacts(runId) {
776
- return this.allArtifacts.filter((a) => a.runId === runId).map((a) => ({ ...a }));
777
- }
778
- };
779
- function matchesRun(r, f) {
780
- if (f.scenarioId && r.scenarioId !== f.scenarioId) return false;
781
- if (f.variantId && r.variantId !== f.variantId) return false;
782
- if (f.status && r.status !== f.status) return false;
783
- if (f.since !== void 0 && r.startedAt < f.since) return false;
784
- if (f.until !== void 0 && r.startedAt > f.until) return false;
785
- if (f.tag && r.tags?.[f.tag.key] !== f.tag.value) return false;
786
- if (f.parentRunId && r.parentRunId !== f.parentRunId) return false;
787
- if (f.projectId && r.projectId !== f.projectId) return false;
788
- if (f.chatId && r.chatId !== f.chatId) return false;
789
- if (f.layer && r.layer !== f.layer) return false;
790
- return true;
791
- }
792
- function matchesSpan(s, f) {
793
- if (f.runId && s.runId !== f.runId) return false;
794
- if (f.parentSpanId && s.parentSpanId !== f.parentSpanId) return false;
795
- if (f.kind && s.kind !== f.kind) return false;
796
- if (f.name && s.name !== f.name) return false;
797
- if (f.toolName && (s.kind !== "tool" || s.toolName !== f.toolName)) return false;
798
- if (f.judgeId && (s.kind !== "judge" || s.judgeId !== f.judgeId)) return false;
799
- if (f.since !== void 0 && s.startedAt < f.since) return false;
800
- if (f.until !== void 0 && s.startedAt > f.until) return false;
801
- return true;
802
- }
803
- function matchesEvent(e, f) {
804
- if (f.runId && e.runId !== f.runId) return false;
805
- if (f.spanId && e.spanId !== f.spanId) return false;
806
- if (f.kind && e.kind !== f.kind) return false;
807
- if (f.since !== void 0 && e.timestamp < f.since) return false;
808
- if (f.until !== void 0 && e.timestamp > f.until) return false;
809
- return true;
810
- }
811
- var FileSystemTraceStore = class {
812
- dir;
813
- maxBytes;
814
- /** Lazy in-memory index for queries — populated on first read. */
815
- index;
816
- /** Memoized index build — concurrent first reads share one build, and an
817
- * append racing an in-flight load awaits this so its row isn't lost. */
818
- indexPromise;
819
- /** Strictly-increasing rollover stamp. Date.now() alone collides when two
820
- * rollovers land in the same millisecond, overwriting a rolled file. */
821
- lastRolloverStamp = 0;
822
- /**
823
- * Per-file append serialization. stat → conditional-rename → appendFile is a
824
- * read-modify-write on the active file; without a lock two concurrent appends
825
- * to the same `name` can both pass the size check (exceeding maxBytes) or one
826
- * can append to a file the other just renamed away. Each entry chains the
827
- * next append behind the prior one for that file.
828
- */
829
- appendLocks = /* @__PURE__ */ new Map();
830
- constructor(options) {
831
- this.dir = options.dir;
832
- this.maxBytes = options.maxBytes ?? 32 * 1024 * 1024;
833
- }
834
- async ensureDir() {
835
- const fs = await import("fs/promises");
836
- await fs.mkdir(this.dir, { recursive: true });
837
- }
838
- async append(name, record) {
839
- if (this.indexPromise) await this.indexPromise;
840
- const prior = this.appendLocks.get(name) ?? Promise.resolve();
841
- const result = prior.then(() => this.appendLocked(name, record));
842
- const tail = result.then(
843
- () => {
844
- },
845
- () => {
846
- }
847
- );
848
- this.appendLocks.set(name, tail);
849
- try {
850
- await result;
851
- } finally {
852
- if (this.appendLocks.get(name) === tail) this.appendLocks.delete(name);
853
- }
854
- }
855
- async appendLocked(name, record) {
856
- await this.ensureDir();
857
- const fs = await import("fs/promises");
858
- const path = await import("path");
859
- const active = path.join(this.dir, `${name}.ndjson`);
860
- try {
861
- const stat = await fs.stat(active);
862
- if (stat.size >= this.maxBytes) {
863
- const stamp = Math.max(Date.now(), this.lastRolloverStamp + 1);
864
- this.lastRolloverStamp = stamp;
865
- const rolled = path.join(this.dir, `${name}.${stamp}.ndjson`);
866
- await fs.rename(active, rolled);
867
- }
868
- } catch {
869
- }
870
- await fs.appendFile(active, `${JSON.stringify(record)}
871
- `, "utf8");
872
- if (this.index && !record?._update) {
873
- await this.insertInto(name, record);
874
- }
875
- }
876
- async insertInto(name, record) {
877
- if (!this.index) return;
878
- switch (name) {
879
- case "runs":
880
- await this.index.appendRun(record);
881
- break;
882
- case "spans":
883
- await this.index.appendSpan(record);
884
- break;
885
- case "events":
886
- await this.index.appendEvent(record);
887
- break;
888
- case "artifacts":
889
- await this.index.appendArtifact(record);
890
- break;
891
- case "budget":
892
- await this.index.appendBudgetEntry(record);
893
- break;
894
- }
895
- }
896
- load() {
897
- this.indexPromise ??= this.buildIndex();
898
- return this.indexPromise;
899
- }
900
- async buildIndex() {
901
- const fs = await import("fs/promises");
902
- const path = await import("path");
903
- const store = new InMemoryTraceStore();
904
- let entries;
905
- try {
906
- entries = await fs.readdir(this.dir);
907
- } catch {
908
- this.index = store;
909
- return store;
910
- }
911
- entries.sort((a, b) => {
912
- const baseA = a.split(".")[0];
913
- const baseB = b.split(".")[0];
914
- if (baseA !== baseB) return baseA < baseB ? -1 : 1;
915
- const ta = a.match(/^[^.]+\.(\d+)\.ndjson$/);
916
- const tb = b.match(/^[^.]+\.(\d+)\.ndjson$/);
917
- const tsA = ta ? Number(ta[1]) : Number.POSITIVE_INFINITY;
918
- const tsB = tb ? Number(tb[1]) : Number.POSITIVE_INFINITY;
919
- return tsA - tsB;
920
- });
921
- {
922
- const runUpdates = [];
923
- const spanUpdates = [];
924
- for (const file of entries) {
925
- if (!file.endsWith(".ndjson")) continue;
926
- const full = path.join(this.dir, file);
927
- const content = await fs.readFile(full, "utf8");
928
- const base = file.split(".")[0];
929
- const lines = content.split("\n");
930
- for (let ln = 0; ln < lines.length; ln++) {
931
- const line = lines[ln];
932
- if (!line.trim()) continue;
933
- let record;
934
- try {
935
- record = JSON.parse(line);
936
- } catch (err) {
937
- throw new Error(
938
- `FileSystemTraceStore: corrupt NDJSON in ${file} line ${ln + 1}: ${err instanceof Error ? err.message : String(err)} \u2014 ${line.slice(0, 120)}`
939
- );
940
- }
941
- if (base === "runs") {
942
- if (record?._update) {
943
- runUpdates.push(record);
944
- continue;
945
- }
946
- try {
947
- await store.appendRun(record);
948
- } catch {
949
- await store.updateRun(record.runId, record);
950
- }
951
- } else if (base === "spans") {
952
- if (record?._update) {
953
- spanUpdates.push(record);
954
- continue;
955
- }
956
- await store.appendSpan(record);
957
- } else if (base === "events") {
958
- await store.appendEvent(record);
959
- } else if (base === "artifacts") {
960
- await store.appendArtifact(record);
961
- } else if (base === "budget") {
962
- await store.appendBudgetEntry(record);
963
- }
964
- }
965
- }
966
- for (const record of runUpdates) {
967
- await store.updateRun(record.runId, record);
968
- }
969
- for (const record of spanUpdates) {
970
- await store.updateSpan(record.spanId, record);
971
- }
972
- }
973
- this.index = store;
974
- return store;
975
- }
976
- async appendRun(run) {
977
- await this.append("runs", run);
978
- }
979
- async updateRun(runId, patch) {
980
- await this.append("runs", { runId, ...patch, _update: true });
981
- if (this.index) await this.index.updateRun(runId, patch);
982
- }
983
- async appendSpan(span) {
984
- await this.append("spans", span);
985
- }
986
- async updateSpan(spanId, patch) {
987
- await this.append("spans", { spanId, ...patch, _update: true });
988
- if (this.index) await this.index.updateSpan(spanId, patch);
989
- }
990
- async appendEvent(event) {
991
- await this.append("events", event);
992
- }
993
- async appendArtifact(artifact) {
994
- await this.append("artifacts", artifact);
995
- }
996
- async appendBudgetEntry(entry) {
997
- await this.append("budget", entry);
998
- }
999
- async getRun(runId) {
1000
- return (await this.load()).getRun(runId);
1001
- }
1002
- async listRuns(filter) {
1003
- return (await this.load()).listRuns(filter);
1004
- }
1005
- async spans(filter) {
1006
- return (await this.load()).spans(filter);
1007
- }
1008
- async events(filter) {
1009
- return (await this.load()).events(filter);
1010
- }
1011
- async budget(runId) {
1012
- return (await this.load()).budget(runId);
1013
- }
1014
- async artifacts(runId) {
1015
- return (await this.load()).artifacts(runId);
1016
- }
1017
- };
1018
-
1019
- // src/trace/capture-fetch.ts
1020
- var DEFAULT_BODY_CAP = 2 * 1024 * 1024;
1021
- function headersToRecord(headers) {
1022
- if (!headers) return void 0;
1023
- const out = {};
1024
- headers.forEach((value, key) => {
1025
- out[key.toLowerCase()] = value;
1026
- });
1027
- return Object.keys(out).length > 0 ? out : void 0;
1028
- }
1029
- function parseMaybeJson(text) {
1030
- if (text.length === 0) return void 0;
1031
- try {
1032
- return JSON.parse(text);
1033
- } catch {
1034
- return text;
1035
- }
1036
- }
1037
- async function readRequestBody(input, init) {
1038
- if (typeof init?.body === "string") return parseMaybeJson(init.body);
1039
- if (init?.body != null) return void 0;
1040
- if (input instanceof Request) {
1041
- try {
1042
- return parseMaybeJson(await input.clone().text());
1043
- } catch {
1044
- return void 0;
1045
- }
1046
- }
1047
- return void 0;
1048
- }
1049
- function endpointFromUrl(url, baseUrl) {
1050
- const normalisedBase = baseUrl.replace(/\/+$/, "");
1051
- if (url.startsWith(normalisedBase)) return url.slice(normalisedBase.length) || "/";
1052
- try {
1053
- return new URL(url).pathname;
1054
- } catch {
1055
- return url;
1056
- }
1057
- }
1058
- function captureFetchToRawSink(fetch2, sink, ctx, opts = {}) {
1059
- const provider = ctx.provider ?? providerFromBaseUrl(ctx.baseUrl);
1060
- const redactor = opts.redactor ?? defaultProviderRedactor;
1061
- const bodyCap = opts.responseBodyByteCap ?? DEFAULT_BODY_CAP;
1062
- let warned = false;
1063
- const baseEvent = (direction, endpoint) => ({
1064
- eventId: crypto.randomUUID(),
1065
- runId: ctx.runId,
1066
- spanId: ctx.spanId,
1067
- provider,
1068
- model: ctx.model,
1069
- endpoint,
1070
- baseUrl: ctx.baseUrl,
1071
- attemptIndex: 0,
1072
- // retries are re-invocations one layer up; documented in 0.x
1073
- direction,
1074
- timestamp: Date.now(),
1075
- redactedFields: []
1076
- });
1077
- const record = async (event) => {
1078
- try {
1079
- await sink.record(redactor(event));
1080
- } catch (err) {
1081
- if (opts.failClosed) throw err;
1082
- if (!warned) {
1083
- warned = true;
1084
- console.warn(
1085
- `captureFetchToRawSink: sink.record failed (capture is best-effort) \u2014 ${err instanceof Error ? err.message : String(err)}`
1086
- );
1087
- }
1088
- }
1089
- };
1090
- return async (input, init) => {
1091
- const url = typeof input === "string" ? input : input instanceof URL ? input.toString() : input.url;
1092
- const method = (init?.method ?? (input instanceof Request ? input.method : "GET")).toUpperCase();
1093
- const endpoint = endpointFromUrl(url, ctx.baseUrl);
1094
- const reqHeaders = new Headers(
1095
- init?.headers ?? (input instanceof Request ? input.headers : void 0)
1096
- );
1097
- await record({
1098
- ...baseEvent("request", endpoint),
1099
- requestHeaders: { ...headersToRecord(reqHeaders), "x-http-method": method },
1100
- requestBody: await readRequestBody(input, init)
1101
- });
1102
- const start = Date.now();
1103
- let response;
1104
- try {
1105
- response = await fetch2(input, init);
1106
- } catch (err) {
1107
- await record({
1108
- ...baseEvent("error", endpoint),
1109
- durationMs: Date.now() - start,
1110
- errorMessage: err instanceof Error ? err.message : String(err)
1111
- });
1112
- throw err;
1113
- }
1114
- let responseBody;
1115
- let rawText;
1116
- const redactedFields = [];
1117
- try {
1118
- rawText = await response.clone().text();
1119
- if (rawText.length > bodyCap) {
1120
- responseBody = rawText.slice(0, bodyCap);
1121
- redactedFields.push("body_truncated");
1122
- } else {
1123
- responseBody = parseMaybeJson(rawText);
1124
- }
1125
- } catch {
1126
- responseBody = void 0;
1127
- }
1128
- if (opts.onUsage && rawText !== void 0) {
1129
- try {
1130
- const parsedForUsage = parseMaybeJson(rawText);
1131
- const usage = extractUsage(parsedForUsage) ?? (typeof parsedForUsage === "string" ? extractUsageFromSse(rawText, { mode: opts.sseUsageMode }) : null);
1132
- if (usage) opts.onUsage(usage, ctx);
1133
- } catch (err) {
1134
- if (opts.failClosed) throw err;
1135
- }
1136
- }
1137
- await record({
1138
- ...baseEvent("response", endpoint),
1139
- durationMs: Date.now() - start,
1140
- statusCode: response.status,
1141
- responseHeaders: headersToRecord(response.headers),
1142
- responseBody,
1143
- redactedFields
1144
- });
1145
- return response;
1146
- };
1147
- }
1148
-
1149
- // src/trace/otel.ts
1150
- var OTEL_AGENT_EVAL_SCOPE = { name: "@tangle-network/agent-eval", version: "0.3.0" };
1151
- async function exportRunAsOtlp(store, runId, resourceAttrs = {}) {
1152
- const run = await store.getRun(runId);
1153
- if (!run) throw new Error(`run ${runId} not found`);
1154
- const spans = await store.spans({ runId });
1155
- const events = await store.events({ runId });
1156
- const eventsBySpan = /* @__PURE__ */ new Map();
1157
- for (const e of events) {
1158
- if (!e.spanId) continue;
1159
- const arr = eventsBySpan.get(e.spanId) ?? [];
1160
- arr.push(e);
1161
- eventsBySpan.set(e.spanId, arr);
1162
- }
1163
- const traceId = runToTraceId(run);
1164
- const otlpSpans = spans.map(
1165
- (s) => spanToOtlp(s, traceId, eventsBySpan.get(s.spanId) ?? [])
1166
- );
1167
- return {
1168
- resourceSpans: [
1169
- {
1170
- resource: {
1171
- attributes: toAttributes({
1172
- "service.name": "agent-eval",
1173
- "run.id": run.runId,
1174
- "run.scenario_id": run.scenarioId,
1175
- "run.variant_id": run.variantId ?? "",
1176
- "run.dataset_version": run.datasetVersion ?? "",
1177
- "run.code_sha": run.codeSha ?? "",
1178
- "run.model_fingerprint": run.modelFingerprint ?? "",
1179
- ...resourceAttrs
1180
- })
1181
- },
1182
- scopeSpans: [{ scope: OTEL_AGENT_EVAL_SCOPE, spans: otlpSpans }]
1183
- }
1184
- ]
1185
- };
1186
- }
1187
- function spanToOtlp(span, traceId, events) {
1188
- const endedAt = span.endedAt ?? span.startedAt;
1189
- return {
1190
- traceId,
1191
- spanId: padSpanId(span.spanId),
1192
- parentSpanId: span.parentSpanId ? padSpanId(span.parentSpanId) : void 0,
1193
- name: span.name,
1194
- kind: 1,
1195
- // SPAN_KIND_INTERNAL
1196
- startTimeUnixNano: msToNs(span.startedAt),
1197
- endTimeUnixNano: msToNs(endedAt),
1198
- attributes: toAttributes(flattenSpanAttributes(span)),
1199
- events: events.map((e) => ({
1200
- timeUnixNano: msToNs(e.timestamp),
1201
- name: e.kind,
1202
- attributes: toAttributes(flattenPayload(e.payload))
1203
- })),
1204
- status: span.status === "error" ? { code: 2, message: span.error } : { code: 1 }
1205
- };
1206
- }
1207
- function flattenSpanAttributes(span) {
1208
- const base = {};
1209
- if (span.attributes) {
1210
- for (const [k, v] of Object.entries(span.attributes)) {
1211
- if (typeof v === "string" || typeof v === "number" || typeof v === "boolean") base[k] = v;
1212
- }
1213
- }
1214
- base[OPENINFERENCE_SPAN_KIND] = traceSpanKindToOpenInferenceKind(span.kind);
1215
- if (span.kind === "llm") {
1216
- applyLlmSpanOtlpAttributes(base, span);
1217
- } else if (span.kind === "tool") {
1218
- applyToolSpanOtlpAttributes(base, span);
1219
- } else if (span.kind === "retrieval") {
1220
- base["retrieval.query"] = span.query;
1221
- base["retrieval.hits"] = span.hits.length;
1222
- } else if (span.kind === "judge") {
1223
- base["judge.id"] = span.judgeId;
1224
- base["judge.dimension"] = span.dimension;
1225
- base["judge.score"] = span.score;
1226
- base["judge.target_span_id"] = span.targetSpanId;
1227
- } else if (span.kind === "sandbox") {
1228
- if (span.image) base["sandbox.image"] = span.image;
1229
- if (span.exitCode !== void 0) base["sandbox.exit_code"] = span.exitCode;
1230
- if (span.testsPassed !== void 0) base["sandbox.tests_passed"] = span.testsPassed;
1231
- if (span.testsTotal !== void 0) base["sandbox.tests_total"] = span.testsTotal;
1232
- }
1233
- return base;
1234
- }
1235
- function flattenPayload(payload) {
1236
- const out = {};
1237
- for (const [k, v] of Object.entries(payload)) {
1238
- if (typeof v === "string" || typeof v === "number" || typeof v === "boolean") out[k] = v;
1239
- else out[k] = JSON.stringify(v);
1240
- }
1241
- return out;
1242
- }
1243
- function toAttributes(record) {
1244
- return Object.entries(record).map(([key, value]) => ({
1245
- key,
1246
- value: typeof value === "number" ? Number.isInteger(value) ? { intValue: value.toString() } : { doubleValue: value } : typeof value === "boolean" ? { boolValue: value } : { stringValue: value }
1247
- }));
1248
- }
1249
- function msToNs(ms) {
1250
- return (BigInt(Math.floor(ms)) * 1000000n).toString();
1251
- }
1252
- function padSpanId(id) {
1253
- const cleaned = id.replace(/-/g, "");
1254
- return cleaned.slice(0, 16).padEnd(16, "0");
1255
- }
1256
- function runToTraceId(run) {
1257
- const cleaned = run.runId.replace(/-/g, "");
1258
- return cleaned.slice(0, 32).padEnd(32, "0");
1259
- }
1260
-
1261
- // src/trace/otel-bridge.ts
1262
- function otelRunCompleteHook(exporter) {
1263
- return async (ctx) => {
1264
- const spans = await ctx.store.spans({ runId: ctx.runId });
1265
- for (const span of spans) {
1266
- if (span.endedAt) {
1267
- exporter.exportSpan(storeSpanToExportable(span, ctx.runId));
1268
- }
1269
- }
1270
- await exporter.flush();
1271
- };
1272
- }
1273
- function createOtelTracingStore(inner, exporter, traceId) {
1274
- return {
1275
- async appendRun(run) {
1276
- return inner.appendRun(run);
1277
- },
1278
- async updateRun(runId, patch) {
1279
- return inner.updateRun(runId, patch);
1280
- },
1281
- async appendSpan(span) {
1282
- if (span.endedAt) {
1283
- exporter.exportSpan(storeSpanToExportable(span, traceId));
1284
- }
1285
- return inner.appendSpan(span);
1286
- },
1287
- async updateSpan(spanId, patch) {
1288
- await inner.updateSpan(spanId, patch);
1289
- if (patch.endedAt) {
1290
- const spans = await inner.spans({ runId: traceId });
1291
- const found = spans.find((s) => s.spanId === spanId);
1292
- if (found) {
1293
- exporter.exportSpan(storeSpanToExportable(found, traceId));
1294
- }
1295
- }
1296
- },
1297
- async appendEvent(event) {
1298
- return inner.appendEvent(event);
1299
- },
1300
- async appendBudgetEntry(entry) {
1301
- return inner.appendBudgetEntry(entry);
1302
- },
1303
- async appendArtifact(artifact) {
1304
- return inner.appendArtifact(artifact);
1305
- },
1306
- getRun: inner.getRun.bind(inner),
1307
- listRuns: inner.listRuns.bind(inner),
1308
- spans: inner.spans.bind(inner),
1309
- events: inner.events.bind(inner),
1310
- budget: inner.budget.bind(inner),
1311
- artifacts: inner.artifacts.bind(inner)
1312
- };
1313
- }
1314
- function storeSpanToExportable(span, traceId) {
1315
- const llm = span.kind === "llm" ? span : void 0;
1316
- const tool = span.kind === "tool" ? span : void 0;
1317
- return {
1318
- traceId,
1319
- spanId: span.spanId,
1320
- parentSpanId: span.parentSpanId,
1321
- name: span.name,
1322
- kind: span.kind,
1323
- startedAt: span.startedAt,
1324
- endedAt: span.endedAt,
1325
- status: span.status,
1326
- error: span.error,
1327
- model: llm?.model,
1328
- inputTokens: llm?.inputTokens,
1329
- outputTokens: llm?.outputTokens,
1330
- reasoningTokens: llm?.reasoningTokens,
1331
- cachedTokens: llm?.cachedTokens,
1332
- cacheWriteTokens: llm?.cacheWriteTokens,
1333
- costUsd: llm?.costUsd,
1334
- tool: tool ? {
1335
- toolName: tool.toolName,
1336
- args: tool.args,
1337
- argsCaptured: tool.argsCaptured,
1338
- result: tool.result,
1339
- latencyMs: tool.latencyMs
1340
- } : void 0,
1341
- attributes: span.attributes
1342
- };
1343
- }
1344
-
1345
- // src/trace/otel-export.ts
1346
- function createOtelExporter(config) {
1347
- const resolvedEndpoint = config?.endpoint ?? (typeof process !== "undefined" ? process.env.OTEL_EXPORTER_OTLP_ENDPOINT : void 0);
1348
- if (!resolvedEndpoint) return void 0;
1349
- const endpoint = resolvedEndpoint;
1350
- const headers = config?.headers ?? parseHeadersFromEnv();
1351
- const batchSize = config?.batchSize ?? 64;
1352
- const flushIntervalMs = config?.flushIntervalMs ?? 5e3;
1353
- const serviceName = config?.serviceName ?? "agent-eval";
1354
- const resourceAttrs = config?.resourceAttributes ?? {};
1355
- const pending = [];
1356
- let timer;
1357
- let stopped = false;
1358
- const exporter = {
1359
- exportSpan(span) {
1360
- if (stopped) return;
1361
- pending.push(toOtlpSpan(span));
1362
- if (pending.length >= batchSize) {
1363
- void doFlush();
1364
- }
1365
- },
1366
- async flush() {
1367
- await doFlush();
1368
- },
1369
- async shutdown() {
1370
- stopped = true;
1371
- if (timer !== void 0) {
1372
- clearInterval(timer);
1373
- timer = void 0;
1374
- }
1375
- await doFlush();
1376
- }
1377
- };
1378
- timer = setInterval(() => {
1379
- if (pending.length > 0) void doFlush();
1380
- }, flushIntervalMs);
1381
- if (typeof timer === "object" && "unref" in timer) {
1382
- ;
1383
- timer.unref();
1384
- }
1385
- async function doFlush() {
1386
- if (pending.length === 0) return;
1387
- const batch = pending.splice(0);
1388
- const body = {
1389
- resourceSpans: [
1390
- {
1391
- resource: {
1392
- attributes: toAttributes2({
1393
- "service.name": serviceName,
1394
- ...resourceAttrs
1395
- })
1396
- },
1397
- scopeSpans: [{ scope: OTEL_AGENT_EVAL_SCOPE, spans: batch }]
1398
- }
1399
- ]
1400
- };
1401
- const url = `${endpoint.replace(/\/+$/, "")}/v1/traces`;
1402
- try {
1403
- await fetch(url, {
1404
- method: "POST",
1405
- headers: {
1406
- "content-type": "application/json",
1407
- ...headers
1408
- },
1409
- body: JSON.stringify(body)
1410
- });
1411
- } catch {
1412
- }
1413
- }
1414
- return exporter;
1415
- }
1416
- function parseHeadersFromEnv() {
1417
- if (typeof process === "undefined") return {};
1418
- const raw = process.env.OTEL_EXPORTER_OTLP_HEADERS;
1419
- if (!raw) return {};
1420
- const out = {};
1421
- for (const pair of raw.split(",")) {
1422
- const eq = pair.indexOf("=");
1423
- if (eq < 0) continue;
1424
- const key = pair.slice(0, eq).trim();
1425
- const value = pair.slice(eq + 1).trim();
1426
- if (key) out[key] = value;
1427
- }
1428
- return out;
1429
- }
1430
- function toOtlpSpan(span) {
1431
- const endedAt = span.endedAt ?? span.startedAt;
1432
- const attrs = {};
1433
- if (span.attributes) {
1434
- for (const [k, v] of Object.entries(span.attributes)) {
1435
- if (typeof v === "string" || typeof v === "number" || typeof v === "boolean") attrs[k] = v;
1436
- }
1437
- }
1438
- attrs[OPENINFERENCE_SPAN_KIND] = traceSpanKindToOpenInferenceKind(span.kind);
1439
- applyLlmSpanOtlpAttributes(attrs, span);
1440
- if (span.tool) applyToolSpanOtlpAttributes(attrs, span.tool);
1441
- return {
1442
- traceId: padTraceId(span.traceId),
1443
- spanId: padSpanId2(span.spanId),
1444
- parentSpanId: span.parentSpanId ? padSpanId2(span.parentSpanId) : void 0,
1445
- name: span.name,
1446
- kind: 1,
1447
- // SPAN_KIND_INTERNAL
1448
- startTimeUnixNano: msToNs2(span.startedAt),
1449
- endTimeUnixNano: msToNs2(endedAt),
1450
- attributes: toAttributes2(attrs),
1451
- status: span.status === "error" ? { code: 2, message: span.error } : { code: 1 }
1452
- };
1453
- }
1454
- function toAttributes2(record) {
1455
- return Object.entries(record).map(([key, value]) => ({
1456
- key,
1457
- value: typeof value === "number" ? Number.isInteger(value) ? { intValue: value.toString() } : { doubleValue: value } : typeof value === "boolean" ? { boolValue: value } : { stringValue: value }
1458
- }));
1459
- }
1460
- function msToNs2(ms) {
1461
- return (BigInt(Math.floor(ms)) * 1000000n).toString();
1462
- }
1463
- function padSpanId2(id) {
1464
- const cleaned = id.replace(/-/g, "");
1465
- return cleaned.slice(0, 16).padEnd(16, "0");
1466
- }
1467
- function padTraceId(id) {
1468
- const cleaned = id.replace(/-/g, "");
1469
- return cleaned.slice(0, 32).padEnd(32, "0");
1470
- }
1471
-
1472
- // src/trace/store-to-otlp.ts
1473
- import { readdirSync, readFileSync, statSync, writeFileSync } from "fs";
1474
- import { join } from "path";
1475
- function convertTraceStoresToOtlp(source, outPath, opts = {}) {
1476
- const sources = Array.isArray(source) ? [...source] : typeof source === "string" ? [{ root: source, layout: "celled" }] : [source];
1477
- const defaultServiceName = opts.serviceName ?? "agent-eval";
1478
- const resourceAttributes = opts.resourceAttributes ?? (() => ({}));
1479
- const runAttributes = opts.runAttributes ?? (() => ({}));
1480
- const lines = [];
1481
- let spanCount = 0;
1482
- let runCount = 0;
1483
- let cellCount = 0;
1484
- let cellErrorCount = 0;
1485
- for (const src of sources) {
1486
- const serviceName = src.serviceName ?? defaultServiceName;
1487
- const cellDirs = src.layout === "flat" ? [{ label: "<root>", dir: src.root }] : listCells(src.root).map((name) => ({ label: name, dir: join(src.root, name) }));
1488
- for (const cell of cellDirs) {
1489
- try {
1490
- const result = projectCell({
1491
- cellDir: cell.dir,
1492
- serviceName,
1493
- resourceAttributes,
1494
- runAttributes
1495
- });
1496
- for (const line of result.lines) lines.push(line);
1497
- spanCount += result.spanCount;
1498
- runCount += result.runCount;
1499
- cellCount += 1;
1500
- } catch (err) {
1501
- console.warn(
1502
- `[traces-to-otlp] cell ${cell.label} (${cell.dir}) skipped: ${err instanceof Error ? err.message : String(err)}`
1503
- );
1504
- cellErrorCount += 1;
1505
- }
1506
- }
1507
- }
1508
- writeFileSync(outPath, lines.join("\n") + (lines.length > 0 ? "\n" : ""));
1509
- return { spanCount, runCount, cellCount, cellErrorCount };
1510
- }
1511
- function projectCell(args) {
1512
- const { cellDir, serviceName, resourceAttributes, runAttributes } = args;
1513
- const lines = [];
1514
- let runCount = 0;
1515
- let spanCount = 0;
1516
- const runs = readMergedShards(cellDir, "runs", "runId");
1517
- const spans = readMergedShards(cellDir, "spans", "spanId");
1518
- const events = readShards(cellDir, "events");
1519
- const runByRunId = /* @__PURE__ */ new Map();
1520
- for (const r of runs) runByRunId.set(r.runId, r);
1521
- const spanBySpanId = /* @__PURE__ */ new Map();
1522
- for (const s of spans) spanBySpanId.set(s.spanId, s);
1523
- const eventsBySpanId = /* @__PURE__ */ new Map();
1524
- for (const e of events) {
1525
- if (e.kind === "state_mutation" && e.payload && typeof e.payload === "object") {
1526
- const entity = e.payload.entity;
1527
- if (entity === "run") {
1528
- const run = e.payload.run;
1529
- if (run?.runId) runByRunId.set(run.runId, run);
1530
- continue;
1531
- }
1532
- if (entity === "run.update") {
1533
- const patch = e.payload.patch;
1534
- if (patch && e.runId) {
1535
- const prior = runByRunId.get(e.runId);
1536
- if (prior) runByRunId.set(e.runId, { ...prior, ...patch });
1537
- }
1538
- continue;
1539
- }
1540
- if (entity === "span") {
1541
- const span = e.payload.span;
1542
- if (span?.spanId) spanBySpanId.set(span.spanId, span);
1543
- continue;
1544
- }
1545
- if (entity === "span.update") {
1546
- const spanId = e.payload.spanId;
1547
- const patch = e.payload.patch;
1548
- if (spanId && patch) {
1549
- const prior = spanBySpanId.get(spanId);
1550
- if (prior) spanBySpanId.set(spanId, { ...prior, ...patch });
1551
- }
1552
- continue;
1553
- }
1554
- }
1555
- if (!e.spanId) continue;
1556
- const arr = eventsBySpanId.get(e.spanId) ?? [];
1557
- arr.push(e);
1558
- eventsBySpanId.set(e.spanId, arr);
1559
- }
1560
- for (const run of runByRunId.values()) {
1561
- const traceId = padTraceId2(run.runId);
1562
- const agentName = run.variantId ?? run.scenarioId;
1563
- const sharedResource = {
1564
- attributes: {
1565
- "service.name": serviceName,
1566
- "agent.name": agentName,
1567
- "run.id": run.runId,
1568
- "run.status": run.status,
1569
- ...resourceAttributes(run)
1570
- }
1571
- };
1572
- const runSpanId = padSpanId3(`run-${run.runId}`);
1573
- const runStart = msToIso(run.startedAt);
1574
- const runEnd = msToIso(run.endedAt ?? run.startedAt);
1575
- const runStatus = run.outcome?.failureClass && run.outcome.failureClass !== "success" ? "STATUS_CODE_ERROR" : "STATUS_CODE_OK";
1576
- const runAttrs = {
1577
- [OPENINFERENCE_SPAN_KIND]: "AGENT",
1578
- "agent.name": agentName,
1579
- "agent.workflow.name": serviceName,
1580
- ...runAttributes(run)
1581
- };
1582
- lines.push(
1583
- JSON.stringify(
1584
- toLine({
1585
- traceId,
1586
- spanId: runSpanId,
1587
- parentSpanId: "",
1588
- name: `run.${agentName}`,
1589
- kind: "SPAN_KIND_INTERNAL",
1590
- startTime: runStart,
1591
- endTime: runEnd,
1592
- statusCode: runStatus,
1593
- statusMessage: run.outcome?.notes ?? "",
1594
- resource: sharedResource,
1595
- attributes: runAttrs
1596
- })
1597
- )
1598
- );
1599
- runCount += 1;
1600
- for (const span of spanBySpanId.values()) {
1601
- if (span.runId !== run.runId) continue;
1602
- const spanAttrs = spanToAttributes(span, eventsBySpanId.get(span.spanId) ?? []);
1603
- const statusCode = span.status === "error" ? "STATUS_CODE_ERROR" : "STATUS_CODE_OK";
1604
- lines.push(
1605
- JSON.stringify(
1606
- toLine({
1607
- traceId,
1608
- spanId: padSpanId3(span.spanId),
1609
- parentSpanId: span.parentSpanId ? padSpanId3(span.parentSpanId) : runSpanId,
1610
- name: span.name,
1611
- kind: spanKindToOtlpKind(span.kind),
1612
- startTime: msToIso(span.startedAt),
1613
- endTime: msToIso(span.endedAt ?? span.startedAt),
1614
- statusCode,
1615
- statusMessage: span.error ?? "",
1616
- resource: sharedResource,
1617
- attributes: spanAttrs
1618
- })
1619
- )
1620
- );
1621
- spanCount += 1;
1622
- }
1623
- }
1624
- return { lines, runCount, spanCount };
1625
- }
1626
- function listCells(root) {
1627
- try {
1628
- return readdirSync(root, { withFileTypes: true }).filter((d) => d.isDirectory()).map((d) => d.name).sort();
1629
- } catch {
1630
- return [];
1631
- }
1632
- }
1633
- function readShards(cellDir, name) {
1634
- let entries;
1635
- try {
1636
- entries = readdirSync(cellDir);
1637
- } catch {
1638
- return [];
1639
- }
1640
- const shards = entries.filter((f) => (f === `${name}.ndjson` || f.startsWith(`${name}.`)) && f.endsWith(".ndjson")).map((f) => ({ file: f, path: join(cellDir, f) })).map((s) => {
1641
- let mtime = 0;
1642
- try {
1643
- mtime = statSync(s.path).mtimeMs;
1644
- } catch {
1645
- }
1646
- return { ...s, mtime };
1647
- }).sort((a, b) => a.mtime - b.mtime || a.file.localeCompare(b.file));
1648
- const rows = [];
1649
- for (const shard of shards) {
1650
- let text;
1651
- try {
1652
- text = readFileSync(shard.path, "utf-8");
1653
- } catch {
1654
- continue;
1655
- }
1656
- for (const line of text.split("\n")) {
1657
- const trimmed = line.trim();
1658
- if (!trimmed) continue;
1659
- try {
1660
- rows.push(JSON.parse(trimmed));
1661
- } catch {
1662
- }
1663
- }
1664
- }
1665
- return rows;
1666
- }
1667
- function readMergedShards(cellDir, name, idKey) {
1668
- const rows = readShards(cellDir, name);
1669
- const byId = /* @__PURE__ */ new Map();
1670
- for (const row of rows) {
1671
- const id = row[idKey];
1672
- if (!id) continue;
1673
- const prior = byId.get(id);
1674
- if (prior && row._update) {
1675
- byId.set(id, { ...prior, ...row, _update: void 0 });
1676
- } else {
1677
- byId.set(id, row);
1678
- }
1679
- }
1680
- return [...byId.values()];
1681
- }
1682
- function spanToAttributes(span, events) {
1683
- const attrs = {
1684
- [OPENINFERENCE_SPAN_KIND]: traceSpanKindToOpenInferenceKind(span.kind)
1685
- };
1686
- if (span.kind === "llm") {
1687
- applyLlmSpanOtlpAttributes(attrs, span);
1688
- if (Array.isArray(span.messages)) {
1689
- attrs["llm.input_messages"] = JSON.stringify(span.messages.slice(-6));
1690
- }
1691
- if (typeof span.output === "string") {
1692
- attrs["llm.output_messages"] = JSON.stringify([{ role: "assistant", content: span.output }]);
1693
- }
1694
- } else if (span.kind === "tool") {
1695
- applyToolSpanOtlpAttributes(attrs, span);
1696
- } else if (span.kind === "judge") {
1697
- attrs["judge.id"] = span.judgeId;
1698
- attrs["judge.dimension"] = span.dimension;
1699
- attrs["judge.score"] = span.score;
1700
- attrs["judge.target_span_id"] = span.targetSpanId;
1701
- }
1702
- if (span.attributes) {
1703
- for (const [k, v] of Object.entries(span.attributes)) {
1704
- attrs[`agent_eval.${k}`] = v;
1705
- }
1706
- }
1707
- if (events.length > 0) {
1708
- attrs["agent_eval.event_count"] = events.length;
1709
- attrs["agent_eval.event_kinds"] = JSON.stringify(events.map((e) => e.kind));
1710
- }
1711
- return attrs;
1712
- }
1713
- function spanKindToOtlpKind(kind) {
1714
- switch (kind) {
1715
- case "llm":
1716
- return "SPAN_KIND_CLIENT";
1717
- case "retrieval":
1718
- return "SPAN_KIND_CLIENT";
1719
- default:
1720
- return "SPAN_KIND_INTERNAL";
1721
- }
1722
- }
1723
- function toLine(args) {
1724
- return {
1725
- trace_id: args.traceId,
1726
- span_id: args.spanId,
1727
- parent_span_id: args.parentSpanId,
1728
- name: args.name,
1729
- kind: args.kind,
1730
- start_time: args.startTime,
1731
- end_time: args.endTime,
1732
- status: { code: args.statusCode, message: args.statusMessage },
1733
- resource: args.resource,
1734
- attributes: args.attributes
1735
- };
1736
- }
1737
- function msToIso(ms) {
1738
- if (!Number.isFinite(ms) || ms <= 0) return (/* @__PURE__ */ new Date(0)).toISOString();
1739
- return new Date(ms).toISOString();
1740
- }
1741
- function padSpanId3(id) {
1742
- const cleaned = id.replace(/[^a-f0-9]/gi, "").toLowerCase();
1743
- if (cleaned.length >= 16) return cleaned.slice(0, 16);
1744
- return foldTo16Hex(id);
1745
- }
1746
- function padTraceId2(id) {
1747
- const cleaned = id.replace(/[^a-f0-9]/gi, "").toLowerCase();
1748
- if (cleaned.length >= 32) return cleaned.slice(0, 32);
1749
- return foldTo32Hex(id);
1750
- }
1751
- function foldTo16Hex(s) {
1752
- let h1 = 2166136261;
1753
- for (const ch of s) {
1754
- h1 ^= ch.charCodeAt(0);
1755
- h1 = Math.imul(h1, 16777619) >>> 0;
1756
- }
1757
- const part = h1.toString(16).padStart(8, "0");
1758
- return (part + part).slice(0, 16);
1759
- }
1760
- function foldTo32Hex(s) {
1761
- return foldTo16Hex(s) + foldTo16Hex(`${s}::trace`).slice(0, 16);
1762
- }
1763
-
1764
- // src/replay.ts
1765
- var ReplayCacheMissError = class extends ReplayError {
1766
- constructor(url, requestKey2, message) {
1767
- super(message ?? `replay cache miss for ${url} (key=${requestKey2})`);
1768
- this.url = url;
1769
- this.requestKey = requestKey2;
1770
- }
1771
- url;
1772
- requestKey;
1773
- };
1774
- var ReplayCache = class _ReplayCache {
1775
- byKey = /* @__PURE__ */ new Map();
1776
- orphans = 0;
1777
- byProvider = {};
1778
- byModel = {};
1779
- /**
1780
- * Build a cache from a sink's events. The sink must implement `list()`.
1781
- * Filter by `runId` / `spanId` to scope to a specific replay.
1782
- */
1783
- static async fromSink(sink, filter = {}) {
1784
- if (!sink.list) {
1785
- throw new ReplayError("ReplayCache.fromSink: sink must implement list() to be replayable.");
1786
- }
1787
- const events = await sink.list(filter);
1788
- return _ReplayCache.fromEvents(events);
1789
- }
1790
- /** Build a cache from an in-memory event list. */
1791
- static async fromEvents(events) {
1792
- const cache = new _ReplayCache();
1793
- const groups = /* @__PURE__ */ new Map();
1794
- for (const e of events) {
1795
- const k = `${e.runId ?? ""}::${e.spanId ?? ""}::${e.attemptIndex}`;
1796
- const g = groups.get(k) ?? {};
1797
- if (e.direction === "request") g.req = e;
1798
- else g.res = e;
1799
- groups.set(k, g);
1800
- }
1801
- for (const g of groups.values()) {
1802
- if (!g.req) continue;
1803
- if (!g.res) {
1804
- cache.orphans += 1;
1805
- continue;
1806
- }
1807
- const key = await requestKey(g.req);
1808
- cache.byKey.set(key, { request: g.req, response: g.res });
1809
- cache.byProvider[g.req.provider] = (cache.byProvider[g.req.provider] ?? 0) + 1;
1810
- cache.byModel[g.req.model] = (cache.byModel[g.req.model] ?? 0) + 1;
1811
- }
1812
- return cache;
1813
- }
1814
- /** Number of cacheable (request, response) pairs in the cache. */
1815
- size() {
1816
- return this.byKey.size;
1817
- }
1818
- stats() {
1819
- return {
1820
- total: this.byKey.size,
1821
- byProvider: { ...this.byProvider },
1822
- byModel: { ...this.byModel },
1823
- orphanRequests: this.orphans
1824
- };
1825
- }
1826
- /** Iterate every cached `(request, response)` pair in insertion order. */
1827
- *entries() {
1828
- for (const entry of this.byKey.values()) yield entry;
1829
- }
1830
- /**
1831
- * Look up a cached response by hashing the (model, messages, temperature,
1832
- * maxTokens, response_format) shape. Returns `undefined` on miss; the
1833
- * caller decides whether to throw, fall back to the network, or skip.
1834
- */
1835
- async lookup(requestBody) {
1836
- const key = await keyFromBody(requestBody);
1837
- return this.byKey.get(key);
1838
- }
1839
- };
1840
- function createReplayFetch(cache, opts = {}) {
1841
- const onMiss = opts.onMiss ?? "throw";
1842
- const fallback = opts.fallbackFetch ?? globalThis.fetch?.bind(globalThis);
1843
- return (async (input, init) => {
1844
- const url = typeof input === "string" ? input : input instanceof URL ? input.toString() : input.url;
1845
- if (!/\/chat\/completions(?:[?#].*)?$/.test(url)) {
1846
- if (!fallback)
1847
- throw new ReplayError(
1848
- `replay fetch: non-completions URL ${url} but no fallbackFetch configured`
1849
- );
1850
- return fallback(input, init);
1851
- }
1852
- let bodyParsed;
1853
- if (init?.body && typeof init.body === "string") {
1854
- try {
1855
- bodyParsed = JSON.parse(init.body);
1856
- } catch {
1857
- }
1858
- }
1859
- const hit = bodyParsed === void 0 ? void 0 : await cache.lookup(bodyParsed);
1860
- if (hit) {
1861
- opts.onHit?.({ url, provider: hit.request.provider, model: hit.request.model });
1862
- const status = hit.response.statusCode ?? 200;
1863
- const headers = new Headers(
1864
- Object.entries(hit.response.responseHeaders ?? { "Content-Type": "application/json" })
1865
- );
1866
- const bodyText = typeof hit.response.responseBody === "string" ? hit.response.responseBody : JSON.stringify(hit.response.responseBody ?? {});
1867
- return new Response(bodyText, { status, headers });
1868
- }
1869
- opts.onMissNotify?.({ url, requestBody: bodyParsed });
1870
- if (onMiss === "throw") {
1871
- const key = bodyParsed === void 0 ? "<unparseable>" : await keyFromBody(bodyParsed);
1872
- throw new ReplayCacheMissError(url, key);
1873
- }
1874
- if (onMiss === "fail-closed") {
1875
- return new Response(JSON.stringify({ error: "replay_cache_miss" }), { status: 599 });
1876
- }
1877
- if (!fallback)
1878
- throw new ReplayError("replay fetch: onMiss=fallback but no fallbackFetch configured");
1879
- return fallback(input, init);
1880
- });
1881
- }
1882
- async function* iterateRawCalls(sink, filter = {}) {
1883
- if (!sink.list) {
1884
- throw new ReplayError("iterateRawCalls: sink must implement list().");
1885
- }
1886
- const events = await sink.list(filter);
1887
- const cache = await ReplayCache.fromEvents(events);
1888
- for (const entry of cache.entries()) yield entry;
1889
- }
1890
- async function requestKey(event) {
1891
- return keyFromBody(event.requestBody);
1892
- }
1893
- async function keyFromBody(body) {
1894
- if (body == null || typeof body !== "object") return hashJson({ raw: String(body) });
1895
- const b = body;
1896
- const reduced = canonicalize({
1897
- model: b.model ?? null,
1898
- messages: b.messages ?? null,
1899
- temperature: b.temperature ?? null,
1900
- max_tokens: b.max_tokens ?? null,
1901
- max_completion_tokens: b.max_completion_tokens ?? null,
1902
- response_format: b.response_format ?? null
1903
- });
1904
- return hashJson(reduced);
1905
- }
1906
-
1907
- export {
1908
- traceAnalystOnRunComplete,
1909
- tokenizeDomainWords,
1910
- inferDomainKeywords,
1911
- domainEvidencePattern,
1912
- describeTraceInsightScope,
1913
- planTraceInsightQuestions,
1914
- buildTraceInsightContext,
1915
- scoreTraceInsightReadiness,
1916
- defaultTraceInsightPanel,
1917
- buildTraceInsightPrompt,
1918
- flattenOtlpExportToNdjson,
1919
- otlpToRunRecords,
1920
- otlpRowsToRunRecords,
1921
- otlpToTraceRunRecords,
1922
- otlpRowsToTraceRunRecords,
1923
- InMemoryTraceStore,
1924
- FileSystemTraceStore,
1925
- captureFetchToRawSink,
1926
- OTEL_AGENT_EVAL_SCOPE,
1927
- exportRunAsOtlp,
1928
- otelRunCompleteHook,
1929
- createOtelTracingStore,
1930
- createOtelExporter,
1931
- convertTraceStoresToOtlp,
1932
- ReplayCacheMissError,
1933
- ReplayCache,
1934
- createReplayFetch,
1935
- iterateRawCalls
1936
- };
1937
- //# sourceMappingURL=chunk-ULOKLHIQ.js.map