@arizeai/phoenix-client 6.10.0 → 6.11.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (244) hide show
  1. package/README.md +62 -0
  2. package/dist/esm/__generated__/api/v1.d.ts +452 -28
  3. package/dist/esm/__generated__/api/v1.d.ts.map +1 -1
  4. package/dist/esm/experiments/helpers/getExampleGlobalId.d.ts +8 -0
  5. package/dist/esm/experiments/helpers/getExampleGlobalId.d.ts.map +1 -0
  6. package/dist/esm/experiments/helpers/getExampleGlobalId.js +9 -0
  7. package/dist/esm/experiments/helpers/getExampleGlobalId.js.map +1 -0
  8. package/dist/esm/experiments/resumeEvaluation.d.ts.map +1 -1
  9. package/dist/esm/experiments/resumeEvaluation.js +2 -1
  10. package/dist/esm/experiments/resumeEvaluation.js.map +1 -1
  11. package/dist/esm/experiments/resumeExperiment.d.ts.map +1 -1
  12. package/dist/esm/experiments/resumeExperiment.js +3 -2
  13. package/dist/esm/experiments/resumeExperiment.js.map +1 -1
  14. package/dist/esm/experiments/runExperiment.d.ts.map +1 -1
  15. package/dist/esm/experiments/runExperiment.js +6 -3
  16. package/dist/esm/experiments/runExperiment.js.map +1 -1
  17. package/dist/esm/jest/index.d.ts +5 -0
  18. package/dist/esm/jest/index.d.ts.map +1 -0
  19. package/dist/esm/jest/index.js +49 -0
  20. package/dist/esm/jest/index.js.map +1 -0
  21. package/dist/esm/jest/reporter.d.ts +13 -0
  22. package/dist/esm/jest/reporter.d.ts.map +1 -0
  23. package/dist/esm/jest/reporter.js +19 -0
  24. package/dist/esm/jest/reporter.js.map +1 -0
  25. package/dist/esm/prompts/sdks/toAI.d.ts +2 -2
  26. package/dist/esm/prompts/sdks/toAI.d.ts.map +1 -1
  27. package/dist/esm/prompts/sdks/toAI.js.map +1 -1
  28. package/dist/esm/prompts/sdks/toAnthropic.d.ts +2 -2
  29. package/dist/esm/prompts/sdks/toAnthropic.d.ts.map +1 -1
  30. package/dist/esm/prompts/sdks/toAnthropic.js.map +1 -1
  31. package/dist/esm/prompts/sdks/toOpenAI.d.ts +2 -2
  32. package/dist/esm/prompts/sdks/toOpenAI.d.ts.map +1 -1
  33. package/dist/esm/prompts/sdks/toOpenAI.js.map +1 -1
  34. package/dist/esm/prompts/sdks/toSDK.d.ts +8 -8
  35. package/dist/esm/prompts/sdks/toSDK.d.ts.map +1 -1
  36. package/dist/esm/prompts/sdks/toSDK.js.map +1 -1
  37. package/dist/esm/prompts/sdks/types.d.ts +2 -2
  38. package/dist/esm/prompts/sdks/types.d.ts.map +1 -1
  39. package/dist/esm/schemas/llm/anthropic/converters.d.ts +8 -8
  40. package/dist/esm/schemas/llm/anthropic/messagePartSchemas.d.ts +4 -4
  41. package/dist/esm/schemas/llm/anthropic/messageSchemas.d.ts +6 -6
  42. package/dist/esm/schemas/llm/constants.d.ts +3 -3
  43. package/dist/esm/schemas/llm/converters.d.ts +12 -12
  44. package/dist/esm/schemas/llm/openai/converters.d.ts +3 -3
  45. package/dist/esm/schemas/llm/schemas.d.ts +2 -2
  46. package/dist/esm/testing/acceptance.d.ts +20 -0
  47. package/dist/esm/testing/acceptance.d.ts.map +1 -0
  48. package/dist/esm/testing/acceptance.js +129 -0
  49. package/dist/esm/testing/acceptance.js.map +1 -0
  50. package/dist/esm/testing/define-api.d.ts +157 -0
  51. package/dist/esm/testing/define-api.d.ts.map +1 -0
  52. package/dist/esm/testing/define-api.js +78 -0
  53. package/dist/esm/testing/define-api.js.map +1 -0
  54. package/dist/esm/testing/helpers.d.ts +55 -0
  55. package/dist/esm/testing/helpers.d.ts.map +1 -0
  56. package/dist/esm/testing/helpers.js +179 -0
  57. package/dist/esm/testing/helpers.js.map +1 -0
  58. package/dist/esm/testing/phoenix-test-tracking.d.ts +68 -0
  59. package/dist/esm/testing/phoenix-test-tracking.d.ts.map +1 -0
  60. package/dist/esm/testing/phoenix-test-tracking.js +521 -0
  61. package/dist/esm/testing/phoenix-test-tracking.js.map +1 -0
  62. package/dist/esm/testing/report-artifacts.d.ts +45 -0
  63. package/dist/esm/testing/report-artifacts.d.ts.map +1 -0
  64. package/dist/esm/testing/report-artifacts.js +218 -0
  65. package/dist/esm/testing/report-artifacts.js.map +1 -0
  66. package/dist/esm/testing/report-run.d.ts +22 -0
  67. package/dist/esm/testing/report-run.d.ts.map +1 -0
  68. package/dist/esm/testing/report-run.js +41 -0
  69. package/dist/esm/testing/report-run.js.map +1 -0
  70. package/dist/esm/testing/reporter-format.d.ts +83 -0
  71. package/dist/esm/testing/reporter-format.d.ts.map +1 -0
  72. package/dist/esm/testing/reporter-format.js +852 -0
  73. package/dist/esm/testing/reporter-format.js.map +1 -0
  74. package/dist/esm/testing/runner.d.ts +31 -0
  75. package/dist/esm/testing/runner.d.ts.map +1 -0
  76. package/dist/esm/testing/runner.js +238 -0
  77. package/dist/esm/testing/runner.js.map +1 -0
  78. package/dist/esm/testing/state.d.ts +138 -0
  79. package/dist/esm/testing/state.d.ts.map +1 -0
  80. package/dist/esm/testing/state.js +31 -0
  81. package/dist/esm/testing/state.js.map +1 -0
  82. package/dist/esm/testing/types.d.ts +319 -0
  83. package/dist/esm/testing/types.d.ts.map +1 -0
  84. package/dist/esm/testing/types.js +9 -0
  85. package/dist/esm/testing/types.js.map +1 -0
  86. package/dist/esm/tsconfig.esm.tsbuildinfo +1 -1
  87. package/dist/esm/utils/channel.d.ts +7 -7
  88. package/dist/esm/utils/channel.d.ts.map +1 -1
  89. package/dist/esm/utils/channel.js +1 -1
  90. package/dist/esm/utils/channel.js.map +1 -1
  91. package/dist/esm/utils/formatPromptMessages.d.ts.map +1 -1
  92. package/dist/esm/utils/getPromptBySelector.d.ts.map +1 -1
  93. package/dist/esm/utils/promisifyResult.d.ts +1 -1
  94. package/dist/esm/utils/promisifyResult.d.ts.map +1 -1
  95. package/dist/esm/utils/promisifyResult.js.map +1 -1
  96. package/dist/esm/utils/schemaMatches.d.ts +5 -5
  97. package/dist/esm/utils/schemaMatches.d.ts.map +1 -1
  98. package/dist/esm/utils/schemaMatches.js.map +1 -1
  99. package/dist/esm/vitest/index.d.ts +5 -0
  100. package/dist/esm/vitest/index.d.ts.map +1 -0
  101. package/dist/esm/vitest/index.js +15 -0
  102. package/dist/esm/vitest/index.js.map +1 -0
  103. package/dist/esm/vitest/reporter.d.ts +19 -0
  104. package/dist/esm/vitest/reporter.d.ts.map +1 -0
  105. package/dist/esm/vitest/reporter.js +27 -0
  106. package/dist/esm/vitest/reporter.js.map +1 -0
  107. package/dist/src/__generated__/api/v1.d.ts +452 -28
  108. package/dist/src/__generated__/api/v1.d.ts.map +1 -1
  109. package/dist/src/experiments/helpers/getExampleGlobalId.d.ts +8 -0
  110. package/dist/src/experiments/helpers/getExampleGlobalId.d.ts.map +1 -0
  111. package/dist/src/experiments/helpers/getExampleGlobalId.js +13 -0
  112. package/dist/src/experiments/helpers/getExampleGlobalId.js.map +1 -0
  113. package/dist/src/experiments/resumeEvaluation.d.ts.map +1 -1
  114. package/dist/src/experiments/resumeEvaluation.js +2 -1
  115. package/dist/src/experiments/resumeEvaluation.js.map +1 -1
  116. package/dist/src/experiments/resumeExperiment.d.ts.map +1 -1
  117. package/dist/src/experiments/resumeExperiment.js +3 -2
  118. package/dist/src/experiments/resumeExperiment.js.map +1 -1
  119. package/dist/src/experiments/runExperiment.d.ts.map +1 -1
  120. package/dist/src/experiments/runExperiment.js +6 -3
  121. package/dist/src/experiments/runExperiment.js.map +1 -1
  122. package/dist/src/jest/index.d.ts +5 -0
  123. package/dist/src/jest/index.d.ts.map +1 -0
  124. package/dist/src/jest/index.js +58 -0
  125. package/dist/src/jest/index.js.map +1 -0
  126. package/dist/src/jest/reporter.d.ts +13 -0
  127. package/dist/src/jest/reporter.d.ts.map +1 -0
  128. package/dist/src/jest/reporter.js +23 -0
  129. package/dist/src/jest/reporter.js.map +1 -0
  130. package/dist/src/prompts/sdks/toAI.d.ts +2 -2
  131. package/dist/src/prompts/sdks/toAI.d.ts.map +1 -1
  132. package/dist/src/prompts/sdks/toAI.js.map +1 -1
  133. package/dist/src/prompts/sdks/toAnthropic.d.ts +2 -2
  134. package/dist/src/prompts/sdks/toAnthropic.d.ts.map +1 -1
  135. package/dist/src/prompts/sdks/toAnthropic.js.map +1 -1
  136. package/dist/src/prompts/sdks/toOpenAI.d.ts +2 -2
  137. package/dist/src/prompts/sdks/toOpenAI.d.ts.map +1 -1
  138. package/dist/src/prompts/sdks/toOpenAI.js.map +1 -1
  139. package/dist/src/prompts/sdks/toSDK.d.ts +8 -8
  140. package/dist/src/prompts/sdks/toSDK.d.ts.map +1 -1
  141. package/dist/src/prompts/sdks/toSDK.js.map +1 -1
  142. package/dist/src/prompts/sdks/types.d.ts +2 -2
  143. package/dist/src/prompts/sdks/types.d.ts.map +1 -1
  144. package/dist/src/schemas/llm/anthropic/converters.d.ts +8 -8
  145. package/dist/src/schemas/llm/anthropic/messagePartSchemas.d.ts +4 -4
  146. package/dist/src/schemas/llm/anthropic/messageSchemas.d.ts +6 -6
  147. package/dist/src/schemas/llm/constants.d.ts +3 -3
  148. package/dist/src/schemas/llm/converters.d.ts +12 -12
  149. package/dist/src/schemas/llm/openai/converters.d.ts +3 -3
  150. package/dist/src/schemas/llm/schemas.d.ts +2 -2
  151. package/dist/src/testing/acceptance.d.ts +20 -0
  152. package/dist/src/testing/acceptance.d.ts.map +1 -0
  153. package/dist/src/testing/acceptance.js +114 -0
  154. package/dist/src/testing/acceptance.js.map +1 -0
  155. package/dist/src/testing/define-api.d.ts +157 -0
  156. package/dist/src/testing/define-api.d.ts.map +1 -0
  157. package/dist/src/testing/define-api.js +81 -0
  158. package/dist/src/testing/define-api.js.map +1 -0
  159. package/dist/src/testing/helpers.d.ts +55 -0
  160. package/dist/src/testing/helpers.d.ts.map +1 -0
  161. package/dist/src/testing/helpers.js +182 -0
  162. package/dist/src/testing/helpers.js.map +1 -0
  163. package/dist/src/testing/phoenix-test-tracking.d.ts +68 -0
  164. package/dist/src/testing/phoenix-test-tracking.d.ts.map +1 -0
  165. package/dist/src/testing/phoenix-test-tracking.js +530 -0
  166. package/dist/src/testing/phoenix-test-tracking.js.map +1 -0
  167. package/dist/src/testing/report-artifacts.d.ts +45 -0
  168. package/dist/src/testing/report-artifacts.d.ts.map +1 -0
  169. package/dist/src/testing/report-artifacts.js +225 -0
  170. package/dist/src/testing/report-artifacts.js.map +1 -0
  171. package/dist/src/testing/report-run.d.ts +22 -0
  172. package/dist/src/testing/report-run.d.ts.map +1 -0
  173. package/dist/src/testing/report-run.js +47 -0
  174. package/dist/src/testing/report-run.js.map +1 -0
  175. package/dist/src/testing/reporter-format.d.ts +83 -0
  176. package/dist/src/testing/reporter-format.d.ts.map +1 -0
  177. package/dist/src/testing/reporter-format.js +870 -0
  178. package/dist/src/testing/reporter-format.js.map +1 -0
  179. package/dist/src/testing/runner.d.ts +31 -0
  180. package/dist/src/testing/runner.d.ts.map +1 -0
  181. package/dist/src/testing/runner.js +258 -0
  182. package/dist/src/testing/runner.js.map +1 -0
  183. package/dist/src/testing/state.d.ts +138 -0
  184. package/dist/src/testing/state.d.ts.map +1 -0
  185. package/dist/src/testing/state.js +38 -0
  186. package/dist/src/testing/state.js.map +1 -0
  187. package/dist/src/testing/types.d.ts +319 -0
  188. package/dist/src/testing/types.d.ts.map +1 -0
  189. package/dist/src/testing/types.js +13 -0
  190. package/dist/src/testing/types.js.map +1 -0
  191. package/dist/src/utils/channel.d.ts +7 -7
  192. package/dist/src/utils/channel.d.ts.map +1 -1
  193. package/dist/src/utils/channel.js +1 -1
  194. package/dist/src/utils/channel.js.map +1 -1
  195. package/dist/src/utils/formatPromptMessages.d.ts.map +1 -1
  196. package/dist/src/utils/getPromptBySelector.d.ts.map +1 -1
  197. package/dist/src/utils/promisifyResult.d.ts +1 -1
  198. package/dist/src/utils/promisifyResult.d.ts.map +1 -1
  199. package/dist/src/utils/promisifyResult.js.map +1 -1
  200. package/dist/src/utils/schemaMatches.d.ts +5 -5
  201. package/dist/src/utils/schemaMatches.d.ts.map +1 -1
  202. package/dist/src/utils/schemaMatches.js.map +1 -1
  203. package/dist/src/vitest/index.d.ts +5 -0
  204. package/dist/src/vitest/index.d.ts.map +1 -0
  205. package/dist/src/vitest/index.js +23 -0
  206. package/dist/src/vitest/index.js.map +1 -0
  207. package/dist/src/vitest/reporter.d.ts +19 -0
  208. package/dist/src/vitest/reporter.d.ts.map +1 -0
  209. package/dist/src/vitest/reporter.js +34 -0
  210. package/dist/src/vitest/reporter.js.map +1 -0
  211. package/dist/tsconfig.tsbuildinfo +1 -1
  212. package/docs/ci-evals-annotations.mdx +190 -0
  213. package/docs/ci-evals-jest.mdx +78 -0
  214. package/docs/ci-evals-vitest.mdx +240 -0
  215. package/docs/ci-evals.mdx +263 -0
  216. package/docs/overview.mdx +9 -1
  217. package/package.json +49 -17
  218. package/src/__generated__/api/v1.ts +452 -28
  219. package/src/experiments/helpers/getExampleGlobalId.ts +12 -0
  220. package/src/experiments/resumeEvaluation.ts +2 -1
  221. package/src/experiments/resumeExperiment.ts +3 -2
  222. package/src/experiments/runExperiment.ts +6 -3
  223. package/src/jest/index.ts +124 -0
  224. package/src/jest/reporter.ts +22 -0
  225. package/src/prompts/sdks/toAI.ts +4 -3
  226. package/src/prompts/sdks/toAnthropic.ts +4 -3
  227. package/src/prompts/sdks/toOpenAI.ts +4 -3
  228. package/src/prompts/sdks/toSDK.ts +16 -11
  229. package/src/prompts/sdks/types.ts +2 -2
  230. package/src/testing/acceptance.ts +190 -0
  231. package/src/testing/define-api.ts +279 -0
  232. package/src/testing/helpers.ts +251 -0
  233. package/src/testing/phoenix-test-tracking.ts +637 -0
  234. package/src/testing/report-artifacts.ts +272 -0
  235. package/src/testing/report-run.ts +44 -0
  236. package/src/testing/reporter-format.ts +1072 -0
  237. package/src/testing/runner.ts +350 -0
  238. package/src/testing/state.ts +165 -0
  239. package/src/testing/types.ts +366 -0
  240. package/src/utils/channel.ts +17 -15
  241. package/src/utils/promisifyResult.ts +6 -4
  242. package/src/utils/schemaMatches.ts +12 -10
  243. package/src/vitest/index.ts +57 -0
  244. package/src/vitest/reporter.ts +32 -0
@@ -0,0 +1,852 @@
1
+ import { formatAcceptanceResult } from "./acceptance.js";
2
+ // ---------------------------------------------------------------------------
3
+ // Option / environment resolution
4
+ // ---------------------------------------------------------------------------
5
+ function isTruthyFlag(value) {
6
+ const v = (value ?? "").toLowerCase();
7
+ return v === "true" || v === "1" || v === "on" || v === "yes";
8
+ }
9
+ function isFalsyFlag(value) {
10
+ const v = (value ?? "").toLowerCase();
11
+ return v === "false" || v === "0" || v === "off" || v === "no";
12
+ }
13
+ /**
14
+ * Resolve {@link RenderOptions} from the environment and output stream.
15
+ *
16
+ * Verbosity: `PHOENIX_TEST_REPORTER=verbose` (or the `PHOENIX_TEST_VERBOSE=1`
17
+ * alias) restores the full per-test dump; the default is the compact view.
18
+ * `PHOENIX_TEST_REPORTER_MAX_ROWS` caps the per-suite rows (default 10).
19
+ *
20
+ * Color follows the common ecosystem rules: off when `NO_COLOR` is set, in CI,
21
+ * on a non-TTY, or a "dumb" terminal; `PHOENIX_TEST_COLOR` / `FORCE_COLOR`
22
+ * force it on or off.
23
+ */
24
+ export function resolveRenderOptions(env = process.env, stream = process.stdout) {
25
+ const verbose = (env.PHOENIX_TEST_REPORTER ?? "").toLowerCase() === "verbose" ||
26
+ isTruthyFlag(env.PHOENIX_TEST_VERBOSE);
27
+ const parsedRows = Number.parseInt(env.PHOENIX_TEST_REPORTER_MAX_ROWS ?? "", 10);
28
+ const maxRows = Number.isFinite(parsedRows) && parsedRows > 0 ? parsedRows : 10;
29
+ const color = resolveColor(env, stream);
30
+ // When piped (no TTY width) assume a roomy-but-safe 100 columns so the
31
+ // overview table doesn't over-truncate suite names in CI logs.
32
+ const columns = typeof stream.columns === "number" && stream.columns > 0
33
+ ? stream.columns
34
+ : 100;
35
+ const maxWidth = Math.min(columns, 120);
36
+ return { verbose, color, maxRows, maxWidth };
37
+ }
38
+ function resolveColor(env, stream) {
39
+ if (isTruthyFlag(env.PHOENIX_TEST_COLOR))
40
+ return true;
41
+ if (isFalsyFlag(env.PHOENIX_TEST_COLOR))
42
+ return false;
43
+ if (env.FORCE_COLOR != null &&
44
+ env.FORCE_COLOR !== "" &&
45
+ env.FORCE_COLOR !== "0")
46
+ return true;
47
+ if (env.NO_COLOR != null && env.NO_COLOR !== "")
48
+ return false;
49
+ if (env.CI != null && env.CI !== "")
50
+ return false;
51
+ if (env.TERM === "dumb")
52
+ return false;
53
+ return stream.isTTY === true;
54
+ }
55
+ // ---------------------------------------------------------------------------
56
+ // Zero-dependency ASCII / ANSI toolkit
57
+ // ---------------------------------------------------------------------------
58
+ const ANSI = {
59
+ green: "\x1b[32m",
60
+ red: "\x1b[31m",
61
+ yellow: "\x1b[33m",
62
+ dim: "\x1b[2m",
63
+ bold: "\x1b[1m",
64
+ reset: "\x1b[0m",
65
+ };
66
+ /** Wrap a string in an ANSI color, or return it unchanged when color is off. */
67
+ function colorize(s, code, o) {
68
+ return o.color ? `${ANSI[code]}${s}${ANSI.reset}` : s;
69
+ }
70
+ // eslint-disable-next-line no-control-regex -- matching the ESC control byte is the point
71
+ const ANSI_PATTERN = /\x1b\[[0-9;]*m/g;
72
+ /** Visible length of a string, ignoring any ANSI escape codes. */
73
+ function visibleLen(s) {
74
+ return s.replace(ANSI_PATTERN, "").length;
75
+ }
76
+ /**
77
+ * Truncate `s` to `max` visible characters, keeping the head and tail with an
78
+ * ellipsis in the middle. Strings containing ANSI codes are returned unchanged
79
+ * to avoid slicing through an escape sequence (colored cells are always short
80
+ * enough to fit, so they never need truncating).
81
+ */
82
+ function truncateMiddle(s, max) {
83
+ if (max <= 1 || ANSI_PATTERN.test(s))
84
+ return s;
85
+ if (s.length <= max)
86
+ return s;
87
+ const head = Math.ceil((max - 1) / 2);
88
+ const tail = Math.floor((max - 1) / 2);
89
+ return `${s.slice(0, head)}…${tail > 0 ? s.slice(s.length - tail) : ""}`;
90
+ }
91
+ /** Left-pad-end a cell to `width`, measuring with {@link visibleLen}. */
92
+ function padCell(s, width) {
93
+ const pad = width - visibleLen(s);
94
+ return pad > 0 ? s + " ".repeat(pad) : s;
95
+ }
96
+ /**
97
+ * Render an aligned ASCII table (no table dependency). Column widths auto-size
98
+ * to content, clamped by `spec.caps`, and the first column is shrunk toward a
99
+ * floor when the table would exceed `o.maxWidth`. Cells longer than their final
100
+ * width are middle-truncated; widths are computed with {@link visibleLen} so
101
+ * ANSI color never breaks alignment.
102
+ */
103
+ function renderTable(headers, rows, o, spec = {}) {
104
+ const colCount = headers.length;
105
+ const gutter = " ";
106
+ const allRows = [headers, ...rows, ...(spec.footer ? [spec.footer] : [])];
107
+ const widths = headers.map((_, i) => {
108
+ const natural = Math.max(...allRows.map((r) => visibleLen(r[i] ?? "")));
109
+ const cap = spec.caps?.[i];
110
+ return cap ? Math.min(natural, cap) : natural;
111
+ });
112
+ const FIRST_COL_FLOOR = 16;
113
+ const totalWidth = () => widths.reduce((a, b) => a + b, 0) + gutter.length * (colCount - 1);
114
+ while (totalWidth() > o.maxWidth && widths[0] > FIRST_COL_FLOOR) {
115
+ widths[0]--;
116
+ }
117
+ const lastCol = colCount - 1;
118
+ const formatRow = (cells) => cells
119
+ .map((cell, i) => {
120
+ const text = truncateMiddle(cell ?? "", widths[i]);
121
+ // The last column is never padded — trailing spaces are wasted tokens.
122
+ return i === lastCol ? text : padCell(text, widths[i]);
123
+ })
124
+ .join(gutter)
125
+ // Drop the dangling gutter when the final cell is empty (e.g. a clean
126
+ // suite's blank Result column).
127
+ .trimEnd();
128
+ const out = [formatRow(headers), ...rows.map(formatRow)];
129
+ if (spec.footer)
130
+ out.push(formatRow(spec.footer));
131
+ return out;
132
+ }
133
+ /**
134
+ * Aggregate every (non-`pass`) annotation across results, preserving the order
135
+ * in which annotation names are first seen. Numeric and boolean annotations are
136
+ * tracked separately; if a name appears as both, the first kind seen wins.
137
+ */
138
+ function computeAnnotationStats(results) {
139
+ const order = [];
140
+ const numeric = new Map();
141
+ const boolean = new Map();
142
+ for (const result of results) {
143
+ for (const ann of result.annotations) {
144
+ if (ann.name === "pass")
145
+ continue;
146
+ if (typeof ann.score === "number" && Number.isFinite(ann.score)) {
147
+ if (!numeric.has(ann.name) && !boolean.has(ann.name))
148
+ order.push(ann.name);
149
+ const arr = numeric.get(ann.name);
150
+ if (arr)
151
+ arr.push(ann.score);
152
+ else if (!boolean.has(ann.name))
153
+ numeric.set(ann.name, [ann.score]);
154
+ }
155
+ else if (typeof ann.score === "boolean") {
156
+ if (!numeric.has(ann.name) && !boolean.has(ann.name))
157
+ order.push(ann.name);
158
+ const cur = boolean.get(ann.name);
159
+ if (cur) {
160
+ cur.total++;
161
+ if (ann.score)
162
+ cur.t++;
163
+ }
164
+ else if (!numeric.has(ann.name)) {
165
+ boolean.set(ann.name, { t: ann.score ? 1 : 0, total: 1 });
166
+ }
167
+ }
168
+ }
169
+ }
170
+ return order.map((name) => {
171
+ const nums = numeric.get(name);
172
+ if (nums) {
173
+ return {
174
+ name,
175
+ kind: "number",
176
+ avg: nums.reduce((a, b) => a + b, 0) / nums.length,
177
+ count: nums.length,
178
+ };
179
+ }
180
+ const bools = boolean.get(name);
181
+ return { name, kind: "boolean", trueCount: bools.t, count: bools.total };
182
+ });
183
+ }
184
+ /** Format an annotation stat as the aggregate-line string used historically. */
185
+ function formatStat(stat) {
186
+ const samples = `${stat.count} sample${stat.count === 1 ? "" : "s"}`;
187
+ return stat.kind === "number"
188
+ ? `avg ${stat.avg.toFixed(3)} (${samples})`
189
+ : `${stat.trueCount}/${stat.count} true`;
190
+ }
191
+ /**
192
+ * Aggregate annotations into the legacy `name -> summary` string map (kept for
193
+ * the verbose view and any external callers).
194
+ */
195
+ function aggregateAnnotations(results) {
196
+ const out = {};
197
+ for (const stat of computeAnnotationStats(results)) {
198
+ out[stat.name] = formatStat(stat);
199
+ }
200
+ return out;
201
+ }
202
+ /**
203
+ * Per-annotation score bar a single run must clear to avoid counting as a
204
+ * "miss". Only `average` criteria contribute one: the aggregate `threshold`
205
+ * reused as a per-run heuristic (the suite-level acceptance block still reports
206
+ * the true aggregate verdict). `passRate` criteria decide passing with an
207
+ * arbitrary `passFn` predicate — there is no static numeric bar to highlight
208
+ * against — so their rows fall back to the default miss heuristic.
209
+ */
210
+ function buildAcceptanceBars(suite) {
211
+ const bars = new Map();
212
+ for (const result of suite.acceptanceResults ?? []) {
213
+ if (result.metric !== "average")
214
+ continue;
215
+ bars.set(result.annotationName, {
216
+ bar: result.threshold,
217
+ direction: result.direction ?? "maximize",
218
+ });
219
+ }
220
+ return bars;
221
+ }
222
+ /**
223
+ * Whether a passing test's evaluator scores fall short. When maximizing, a
224
+ * boolean `false` or a numeric score below its bar is a miss; when minimizing,
225
+ * a boolean `true` or a score above its bar is a miss. With no criterion for an
226
+ * annotation, only a non-positive score counts (keeps zero-config suites quiet).
227
+ */
228
+ function isMiss(result, bars) {
229
+ for (const ann of result.annotations) {
230
+ if (ann.name === "pass")
231
+ continue;
232
+ const acceptanceBar = bars.get(ann.name);
233
+ const minimizing = acceptanceBar?.direction === "minimize";
234
+ if (typeof ann.score === "boolean") {
235
+ if (minimizing ? ann.score : !ann.score)
236
+ return true;
237
+ }
238
+ else if (typeof ann.score === "number" && Number.isFinite(ann.score)) {
239
+ if (acceptanceBar === undefined) {
240
+ if (ann.score <= 0)
241
+ return true;
242
+ }
243
+ else if (minimizing
244
+ ? ann.score > acceptanceBar.bar
245
+ : ann.score < acceptanceBar.bar) {
246
+ return true;
247
+ }
248
+ }
249
+ }
250
+ return false;
251
+ }
252
+ /**
253
+ * Annotation columns for a suite's table: the acceptance-gated metrics first
254
+ * (the ones a user cares about), else the union of annotation names by
255
+ * first-seen order. Capped at three to keep the table narrow.
256
+ */
257
+ function selectAnnotationColumns(suite) {
258
+ const MAX_COLUMNS = 3;
259
+ const gated = (suite.acceptanceResults ?? []).map((r) => r.annotationName);
260
+ const ordered = gated.length > 0
261
+ ? gated
262
+ : computeAnnotationStats(suite.results).map((s) => s.name);
263
+ return [...new Set(ordered)].slice(0, MAX_COLUMNS);
264
+ }
265
+ /** The aggregate cell for an annotation column. */
266
+ function aggregateCell(stats, name) {
267
+ const stat = stats.find((s) => s.name === name);
268
+ if (!stat)
269
+ return "—";
270
+ return stat.kind === "number"
271
+ ? `avg ${stat.avg.toFixed(2)}`
272
+ : `${stat.trueCount}/${stat.count}`;
273
+ }
274
+ function computeSuiteVitals(suite) {
275
+ const bars = buildAcceptanceBars(suite);
276
+ const total = suite.results.length;
277
+ const passed = suite.results.filter((r) => r.status === "passed").length;
278
+ const failed = suite.results.filter((r) => r.status === "failed").length;
279
+ const misses = suite.results
280
+ .filter((r) => r.status === "passed" && isMiss(r, bars))
281
+ .sort((a, b) => worstScore(a) - worstScore(b));
282
+ const acceptanceFailed = (suite.acceptanceResults ?? []).some((r) => !r.passed);
283
+ const status = failed > 0 || acceptanceFailed
284
+ ? "fail"
285
+ : misses.length > 0
286
+ ? "miss"
287
+ : "pass";
288
+ const meanLatencyMs = total > 0 ? suite.results.reduce((a, r) => a + r.durationMs, 0) / total : 0;
289
+ return {
290
+ total,
291
+ passed,
292
+ failed,
293
+ missCount: misses.length,
294
+ status,
295
+ meanLatencyMs,
296
+ problems: [
297
+ ...suite.results.filter((r) => r.status === "failed"),
298
+ ...misses,
299
+ ],
300
+ acceptanceFailed,
301
+ };
302
+ }
303
+ /** The worst (lowest) evaluator score on a row, for ordering most-broken first. */
304
+ function worstScore(result) {
305
+ let worst = Number.POSITIVE_INFINITY;
306
+ for (const ann of result.annotations) {
307
+ if (ann.name === "pass")
308
+ continue;
309
+ worst = Math.min(worst, annotationScoreValue(ann));
310
+ }
311
+ return worst;
312
+ }
313
+ /** ANSI color matching a status (green pass / yellow miss / red fail). */
314
+ function statusColor(status) {
315
+ return status === "fail" ? "red" : status === "miss" ? "yellow" : "green";
316
+ }
317
+ /** `2/2 passed · 1 failed · 3 misses`, dropping any zero clause. */
318
+ function countsLabel(v) {
319
+ const parts = [`${v.passed}/${v.total} passed`];
320
+ if (v.failed > 0)
321
+ parts.push(`${v.failed} failed`);
322
+ if (v.missCount > 0)
323
+ parts.push(`${v.missCount} miss${v.missCount === 1 ? "" : "es"}`);
324
+ return parts.join(" · ");
325
+ }
326
+ /** Setup / upload problems worth surfacing regardless of test status. */
327
+ function warningLines(suite, o) {
328
+ const out = [];
329
+ if (suite.setupError?.message) {
330
+ out.push(` ${colorize("setup error:", "red", o)} ${suite.setupError.message}`);
331
+ }
332
+ const n = suite.uploadFailureCount ?? 0;
333
+ if (n > 0) {
334
+ out.push(` ${colorize("warning:", "yellow", o)} ${n} upload${n === 1 ? "" : "s"} failed (auth or network?)`);
335
+ }
336
+ return out;
337
+ }
338
+ /**
339
+ * Render a single suite. A clean suite collapses to one line; a suite with
340
+ * failures or misses expands into a per-row diagnosis (scores, rationale,
341
+ * output, and the Phoenix ids needed to pull the trace). Pass a verbose
342
+ * {@link RenderOptions} to restore the full per-test dump.
343
+ */
344
+ export function formatSuiteSummary(suite, o = resolveRenderOptions()) {
345
+ return o.verbose
346
+ ? formatVerboseSuite(suite)
347
+ : formatSuiteDetail(suite, computeSuiteVitals(suite), o);
348
+ }
349
+ /** Verbatim acceptance-criteria block (kept stable for downstream parsers). */
350
+ function acceptanceLines(suite) {
351
+ if (!suite.acceptanceResults || suite.acceptanceResults.length === 0)
352
+ return [];
353
+ return [
354
+ " Acceptance Criteria:",
355
+ ...suite.acceptanceResults.map((r) => ` ${formatAcceptanceResult(r)}`),
356
+ ];
357
+ }
358
+ function linkLines(suite) {
359
+ return suite.links.map((link) => ` ${link.label}: ${link.url}`);
360
+ }
361
+ /** Dim one-line roll-up of every metric average plus mean latency. */
362
+ function vitalsInline(suite, v, o) {
363
+ const parts = computeAnnotationStats(suite.results).map((s) => s.kind === "number"
364
+ ? `${s.name} ${s.avg.toFixed(2)}`
365
+ : `${s.name} ${s.trueCount}/${s.count}`);
366
+ parts.push(`avg ${formatDuration(v.meanLatencyMs)}`);
367
+ return colorize(parts.join(" "), "dim", o);
368
+ }
369
+ function formatSuiteDetail(suite, v, o) {
370
+ const title = `${colorize(suite.name, statusColor(v.status), o)} ${colorize(countsLabel(v), "dim", o)}`;
371
+ const warnings = warningLines(suite, o);
372
+ // Clean suite with nothing to warn about: one line is the whole story.
373
+ if (v.status === "pass" && warnings.length === 0) {
374
+ return `${title} ${vitalsInline(suite, v, o)}`;
375
+ }
376
+ const lines = [title, ` ${vitalsInline(suite, v, o)}`, ...warnings];
377
+ // Never hide a hard failure; cap the number of below-bar misses shown.
378
+ const failures = v.problems.filter((r) => r.status === "failed");
379
+ const misses = v.problems.filter((r) => r.status !== "failed");
380
+ const shownMisses = misses.slice(0, Math.max(o.maxRows - failures.length, 0));
381
+ for (const r of [...failures, ...shownMisses]) {
382
+ lines.push(...problemEntry(r, o));
383
+ }
384
+ const hidden = misses.length - shownMisses.length;
385
+ if (hidden > 0) {
386
+ lines.push(colorize(` … ${hidden} more miss${hidden === 1 ? "" : "es"}`, "dim", o));
387
+ }
388
+ for (const a of suite.acceptanceResults ?? []) {
389
+ if (!a.passed) {
390
+ lines.push(` ${colorize("✗ acceptance", "red", o)} ${formatAcceptanceResult(a).replace(/^FAIL /, "")}`);
391
+ }
392
+ }
393
+ for (const link of suite.links) {
394
+ lines.push(` ${link.label}: ${link.url}`);
395
+ }
396
+ return lines.join("\n");
397
+ }
398
+ /**
399
+ * One failing / missing row: a weighted title with its scores, then the dim
400
+ * detail an agent needs to fix it — rationale, model output, and trace ids.
401
+ */
402
+ function problemEntry(result, o) {
403
+ const mark = colorize("✗", result.status === "failed" ? "red" : "yellow", o);
404
+ const name = truncateEnd(humanizeLabel(result.testName), 72);
405
+ const dry = result.dryRun ? colorize(" (dry run)", "dim", o) : "";
406
+ const lines = [` ${mark} ${colorize(name, "bold", o)}${dry}`];
407
+ const indent = " ";
408
+ const err = compactError(result.error);
409
+ if (err)
410
+ lines.push(`${indent}${colorize(err, "red", o)}`);
411
+ // The sub-perfect evaluators that dragged the row down, worst score first —
412
+ // a clean `1.0` metric isn't what broke it, so it stays out of the way.
413
+ const rationales = [...result.annotations]
414
+ .filter((a) => a.name !== "pass" && annotationScoreValue(a) < 1)
415
+ .sort((a, b) => annotationScoreValue(a) - annotationScoreValue(b))
416
+ .slice(0, 3);
417
+ for (const ann of rationales) {
418
+ const reason = ann.explanation ?? ann.label;
419
+ const tail = reason
420
+ ? ` ${colorize("·", "dim", o)} ${truncateSummary(reason, 160)}`
421
+ : "";
422
+ lines.push(`${indent}${colorize(ann.name, "dim", o)} ${formatScore(ann)}${tail}`);
423
+ }
424
+ const output = summarizeValue(result.output, 160);
425
+ if (output !== null) {
426
+ lines.push(`${indent}${colorize("output", "dim", o)} ${output}`);
427
+ }
428
+ const ids = formatResultIds(result);
429
+ if (ids)
430
+ lines.push(`${indent}${colorize(ids, "dim", o)}`);
431
+ return lines;
432
+ }
433
+ /** The legacy verbose view: every test as a delimited block including output. */
434
+ function formatVerboseSuite(suite) {
435
+ const lines = [];
436
+ lines.push("");
437
+ lines.push(suite.name);
438
+ if (suite.trackingDisabled) {
439
+ lines.push(` (tracking disabled — ${friendlyTrackingReason(suite)})`);
440
+ }
441
+ const total = suite.results.length;
442
+ const passed = suite.results.filter((r) => r.status === "passed").length;
443
+ const failed = suite.results.filter((r) => r.status === "failed").length;
444
+ lines.push(` ${passed}/${total} passed${failed ? `, ${failed} failed` : ""}`);
445
+ if (suite.uploadFailureCount && suite.uploadFailureCount > 0) {
446
+ lines.push(` warning: ${suite.uploadFailureCount} upload${suite.uploadFailureCount === 1 ? "" : "s"} failed (auth or network?)`);
447
+ }
448
+ const aggregated = aggregateAnnotations(suite.results);
449
+ for (const [name, summary] of Object.entries(aggregated)) {
450
+ lines.push(` ${name}: ${summary}`);
451
+ }
452
+ lines.push(...acceptanceLines(suite));
453
+ for (const result of suite.results) {
454
+ const status = result.status === "passed"
455
+ ? "PASS"
456
+ : result.status === "failed"
457
+ ? "FAIL"
458
+ : "SKIP";
459
+ const tag = result.dryRun ? " (dry run — not uploaded)" : "";
460
+ const annotations = formatAnnotationsInline(result.annotations);
461
+ const annotationSuffix = annotations ? ` → ${annotations}` : "";
462
+ lines.push("");
463
+ lines.push(` [${status}] ${humanizeLabel(result.testName)} (${formatDuration(result.durationMs)})${tag}${annotationSuffix}`);
464
+ if (result.error) {
465
+ lines.push(` error: ${result.error}`);
466
+ }
467
+ if (result.output !== undefined) {
468
+ lines.push(` output: ${stringifyForLog(result.output)}`);
469
+ }
470
+ for (const ann of result.annotations) {
471
+ if (ann.name !== "pass" && ann.explanation) {
472
+ lines.push(` why (${ann.name}): ${ann.explanation}`);
473
+ }
474
+ }
475
+ const ids = formatResultIds(result);
476
+ if (ids) {
477
+ lines.push(` ids: ${ids}`);
478
+ }
479
+ }
480
+ lines.push(...linkLines(suite));
481
+ return lines.join("\n");
482
+ }
483
+ // ---------------------------------------------------------------------------
484
+ // Cross-suite overview
485
+ // ---------------------------------------------------------------------------
486
+ /** Friendly, env-var-free reason a suite ran locally. */
487
+ function friendlyTrackingReason(suite) {
488
+ return suite.setupError?.message ?? "local only";
489
+ }
490
+ /** A single tracking note when every suite ran locally for the same reason. */
491
+ function sharedTrackingNote(suites) {
492
+ return suites.length > 0 &&
493
+ suites.every((s) => s.trackingDisabled && !s.setupError)
494
+ ? "tracking disabled (local only)"
495
+ : undefined;
496
+ }
497
+ /** The primary metric stat for a suite's overview row, or `null`. */
498
+ function primaryStat(suite) {
499
+ const stats = computeAnnotationStats(suite.results);
500
+ const primary = selectAnnotationColumns(suite)[0];
501
+ return stats.find((s) => s.name === primary) ?? null;
502
+ }
503
+ /**
504
+ * Render the run header (totals + tracking note) and, for multi-suite runs, an
505
+ * aligned overview table: one row per suite with its pass count, primary metric,
506
+ * acceptance verdict, mean latency, and a miss/fail note. This is the index;
507
+ * only suites with problems are expanded into a detail block below it.
508
+ */
509
+ export function formatScoreboard(suites, o = resolveRenderOptions()) {
510
+ if (suites.length === 0)
511
+ return "";
512
+ const vitals = suites.map(computeSuiteVitals);
513
+ const passed = vitals.reduce((a, v) => a + v.passed, 0);
514
+ const total = vitals.reduce((a, v) => a + v.total, 0);
515
+ const failedTests = vitals.reduce((a, v) => a + v.failed, 0);
516
+ const misses = vitals.reduce((a, v) => a + v.missCount, 0);
517
+ const acceptFails = vitals.filter((v) => v.acceptanceFailed).length;
518
+ const uploadFails = suites.reduce((a, s) => a + (s.uploadFailureCount ?? 0), 0);
519
+ // Totals are test/row-level so the clauses stay in one unit; the per-suite
520
+ // breakdown lives in the table below.
521
+ const header = [
522
+ "Eval Results",
523
+ `${suites.length} suite${suites.length === 1 ? "" : "s"}`,
524
+ `${passed}/${total} passed`,
525
+ ];
526
+ if (failedTests > 0)
527
+ header.push(`${failedTests} failed`);
528
+ if (misses > 0)
529
+ header.push(`${misses} miss${misses === 1 ? "" : "es"}`);
530
+ if (acceptFails > 0) {
531
+ header.push(`${acceptFails} acceptance failure${acceptFails === 1 ? "" : "s"}`);
532
+ }
533
+ if (uploadFails > 0)
534
+ header.push(`${uploadFails} uploads failed`);
535
+ const note = sharedTrackingNote(suites);
536
+ if (note)
537
+ header.push(note);
538
+ const headerLine = colorize(header.join(" · "), "bold", o);
539
+ // A single suite's own detail block is the overview; just print the header.
540
+ if (suites.length === 1)
541
+ return headerLine;
542
+ // Columns appear only when at least one suite has something to put in them.
543
+ const anyScore = suites.some((s) => primaryStat(s) !== null);
544
+ const anyAccept = suites.some((s) => (s.acceptanceResults ?? []).length > 0);
545
+ const anyLink = suites.some((s) => s.links.length > 0);
546
+ const primaries = suites.map((s) => selectAnnotationColumns(s)[0]);
547
+ const shared = primaries.every((p) => p && p === primaries[0]) && primaries[0]
548
+ ? primaries[0]
549
+ : undefined;
550
+ const headers = ["Suite", "Tests"];
551
+ const caps = [34, 7];
552
+ if (anyScore) {
553
+ headers.push(shared ?? "Score");
554
+ caps.push(22);
555
+ }
556
+ if (anyAccept) {
557
+ headers.push("Accept");
558
+ caps.push(7);
559
+ }
560
+ headers.push("Latency", "Result");
561
+ caps.push(8, 12);
562
+ if (anyLink) {
563
+ headers.push("Link");
564
+ caps.push(48);
565
+ }
566
+ const rows = suites.map((suite, i) => {
567
+ const v = vitals[i];
568
+ const stats = computeAnnotationStats(suite.results);
569
+ const stat = primaryStat(suite);
570
+ const row = [suite.name, `${v.passed}/${v.total}`];
571
+ if (anyScore) {
572
+ row.push(!stat
573
+ ? "—"
574
+ : shared
575
+ ? aggregateCell(stats, stat.name)
576
+ : `${stat.name} ${aggregateCell(stats, stat.name).replace(/^avg /, "")}`);
577
+ }
578
+ if (anyAccept) {
579
+ row.push((suite.acceptanceResults ?? []).length === 0
580
+ ? "—"
581
+ : v.acceptanceFailed
582
+ ? colorize("FAIL", "red", o)
583
+ : colorize("PASS", "green", o));
584
+ }
585
+ row.push(formatDuration(v.meanLatencyMs), resultNote(v, o));
586
+ if (anyLink)
587
+ row.push(suite.links[0]?.url ?? "—");
588
+ return row;
589
+ });
590
+ return [headerLine, "", ...renderTable(headers, rows, o, { caps })].join("\n");
591
+ }
592
+ /** The overview "Result" cell: what went wrong, colored, or blank when clean. */
593
+ function resultNote(v, o) {
594
+ if (v.failed > 0)
595
+ return colorize(`${v.failed} failed`, "red", o);
596
+ if (v.acceptanceFailed)
597
+ return colorize("accept ✗", "red", o);
598
+ if (v.missCount > 0) {
599
+ return colorize(`${v.missCount} miss${v.missCount === 1 ? "" : "es"}`, "yellow", o);
600
+ }
601
+ return "";
602
+ }
603
+ // ---------------------------------------------------------------------------
604
+ // Entry point
605
+ // ---------------------------------------------------------------------------
606
+ /**
607
+ * Print the run summary: the overview header (and, for multi-suite runs, the
608
+ * index table), then an expanded detail block for every suite that failed or
609
+ * had misses. Clean suites are fully described by their overview row. A
610
+ * single-suite run always prints its block; verbose prints every block.
611
+ */
612
+ export function printSuiteSummaries(suites) {
613
+ const o = resolveRenderOptions();
614
+ const overview = formatScoreboard(suites, o);
615
+ // eslint-disable-next-line no-console
616
+ if (overview)
617
+ console.log(overview);
618
+ const expand = o.verbose
619
+ ? suites
620
+ : suites.length === 1
621
+ ? suites
622
+ : suites.filter((s) => computeSuiteVitals(s).status !== "pass");
623
+ for (const suite of expand) {
624
+ // eslint-disable-next-line no-console
625
+ console.log(`\n${formatSuiteSummary(suite, o)}`);
626
+ }
627
+ }
628
+ // ---------------------------------------------------------------------------
629
+ // Shared formatters
630
+ // ---------------------------------------------------------------------------
631
+ /**
632
+ * Render a test's annotations as a compact, single-line `name=score` list for
633
+ * the verbose per-test header. The implicit `pass` annotation is omitted.
634
+ */
635
+ function formatAnnotationsInline(annotations) {
636
+ return annotations
637
+ .filter((ann) => ann.name !== "pass")
638
+ .map((ann) => `${ann.name}=${formatScore(ann)}`)
639
+ .join(", ");
640
+ }
641
+ function formatScore(ann) {
642
+ if (typeof ann.score === "number")
643
+ return ann.score.toString();
644
+ if (typeof ann.score === "boolean")
645
+ return ann.score ? "true" : "false";
646
+ if (ann.label)
647
+ return ann.label;
648
+ return "(no score)";
649
+ }
650
+ function formatDuration(ms) {
651
+ if (ms < 1000)
652
+ return `${Math.round(ms)}ms`;
653
+ return `${(ms / 1000).toFixed(2)}s`;
654
+ }
655
+ function stringifyForLog(value) {
656
+ try {
657
+ const json = JSON.stringify(value);
658
+ return json && json.length > 200
659
+ ? `${json.slice(0, 197)}...`
660
+ : (json ?? "");
661
+ }
662
+ catch {
663
+ return String(value);
664
+ }
665
+ }
666
+ // ---------------------------------------------------------------------------
667
+ // Per-failure detail helpers
668
+ //
669
+ // A problem row's block answers the two questions an agent needs to fix it:
670
+ // *why* (the judge's rationale and the model output) and *where* (the Phoenix
671
+ // trace / run / example ids it can pull for the full picture).
672
+ // ---------------------------------------------------------------------------
673
+ /** A run's numeric score for sorting (booleans as 1/0, missing as +∞). */
674
+ function annotationScoreValue(ann) {
675
+ if (typeof ann.score === "number")
676
+ return ann.score;
677
+ if (typeof ann.score === "boolean")
678
+ return ann.score ? 1 : 0;
679
+ return Number.POSITIVE_INFINITY;
680
+ }
681
+ /** First non-empty line of a multi-line error, truncated for one-line display. */
682
+ function compactError(error) {
683
+ if (!error)
684
+ return null;
685
+ const firstLine = error
686
+ .split("\n")
687
+ .map((line) => line.trim())
688
+ .find((line) => line.length > 0);
689
+ return firstLine ? truncateSummary(firstLine, 160) : null;
690
+ }
691
+ /** Phoenix ids for a run as a single `trace=… run=… example=…` string. */
692
+ function formatResultIds(result) {
693
+ const parts = [];
694
+ if (result.traceId)
695
+ parts.push(`trace=${result.traceId}`);
696
+ if (result.runId)
697
+ parts.push(`run=${result.runId}`);
698
+ if (result.exampleId)
699
+ parts.push(`example=${result.exampleId}`);
700
+ return parts.length > 0 ? parts.join(" ") : null;
701
+ }
702
+ // ---------------------------------------------------------------------------
703
+ // Label humanization
704
+ // ---------------------------------------------------------------------------
705
+ /**
706
+ * Turn a machine test name into a readable title. A `test.each` row is named by
707
+ * stringifying its input (`{"userQuery":"Show active users"}`); we surface the
708
+ * value of a single-field object directly (`Show active users`) and fold a
709
+ * multi-field object to `key=value` pairs. Non-JSON names pass through.
710
+ */
711
+ function humanizeLabel(name) {
712
+ const trimmed = name.trim();
713
+ if (!(trimmed.startsWith("{") || trimmed.startsWith("[")))
714
+ return name;
715
+ let parsed;
716
+ try {
717
+ parsed = JSON.parse(trimmed);
718
+ }
719
+ catch {
720
+ return name;
721
+ }
722
+ if (Array.isArray(parsed))
723
+ return summarizeValue(parsed) ?? name;
724
+ if (!parsed || typeof parsed !== "object")
725
+ return name;
726
+ const entries = Object.entries(parsed).filter(([, v]) => v != null && v !== "");
727
+ if (entries.length === 0)
728
+ return name;
729
+ if (entries.length === 1 && typeof entries[0][1] === "string") {
730
+ return entries[0][1];
731
+ }
732
+ return entries
733
+ .map(([k, v]) => `${k}=${summaryPrimitive(v) ?? ""}`)
734
+ .join(" ");
735
+ }
736
+ /** Truncate keeping the head (most identifying for a title), trailing ellipsis. */
737
+ function truncateEnd(s, max) {
738
+ return s.length <= max ? s : `${s.slice(0, max - 1)}…`;
739
+ }
740
+ // ---------------------------------------------------------------------------
741
+ // Token-efficient value summarization (adapted from the vitest-evals reporter)
742
+ //
743
+ // Replaces a blind JSON truncation with a key-preferring summary: the salient
744
+ // keys of an eval output (`score`, `output`, `error`, …) come first, primitives
745
+ // are rendered compactly, and JSON-encoded strings are parsed so the same
746
+ // summary applies. The result is far denser and more legible per token.
747
+ // ---------------------------------------------------------------------------
748
+ /** Keys surfaced first when summarizing a record, in priority order. */
749
+ const PREFERRED_SUMMARY_KEYS = [
750
+ "score",
751
+ "label",
752
+ "pass",
753
+ "passed",
754
+ "output",
755
+ "result",
756
+ "answer",
757
+ "response",
758
+ "reason",
759
+ "rationale",
760
+ "explanation",
761
+ "error",
762
+ "message",
763
+ "name",
764
+ "id",
765
+ "status",
766
+ ];
767
+ function truncateSummary(value, maxLength = 96) {
768
+ return value.length <= maxLength
769
+ ? value
770
+ : `${value.slice(0, maxLength - 1)}…`;
771
+ }
772
+ /** Render a single value as a short token: scalars inline, containers as counts. */
773
+ function summaryPrimitive(value) {
774
+ if (value === undefined)
775
+ return null;
776
+ if (value === null)
777
+ return "null";
778
+ if (typeof value === "string") {
779
+ const truncated = truncateSummary(value, 48);
780
+ // Bare-word strings stay unquoted; anything with spaces/punctuation is
781
+ // quoted so the key=value pairs remain unambiguous.
782
+ return /^[\w.:/@-]+$/.test(truncated)
783
+ ? truncated
784
+ : JSON.stringify(truncated);
785
+ }
786
+ if (typeof value === "number" || typeof value === "boolean") {
787
+ return String(value);
788
+ }
789
+ if (Array.isArray(value))
790
+ return `array(${value.length})`;
791
+ if (typeof value === "object") {
792
+ return `object(${Object.keys(value).length})`;
793
+ }
794
+ return String(value);
795
+ }
796
+ function summarizeRecord(record, maxLength) {
797
+ const keys = Object.keys(record);
798
+ if (keys.length === 0)
799
+ return "object(0)";
800
+ const ordered = [
801
+ ...PREFERRED_SUMMARY_KEYS.filter((key) => keys.includes(key)),
802
+ ...keys.filter((key) => !PREFERRED_SUMMARY_KEYS.includes(key)),
803
+ ].slice(0, 4);
804
+ const parts = ordered
805
+ .map((key) => {
806
+ const formatted = summaryPrimitive(record[key]);
807
+ return formatted === null ? null : `${key}=${formatted}`;
808
+ })
809
+ .filter((part) => part !== null);
810
+ if (parts.length === 0)
811
+ return null;
812
+ const suffix = keys.length > ordered.length ? " …" : "";
813
+ return truncateSummary(`${parts.join(" ")}${suffix}`, maxLength);
814
+ }
815
+ /**
816
+ * Summarize an arbitrary value to a compact, single-line string, or `null` when
817
+ * there's nothing to show (`undefined`). JSON-encoded strings are parsed first
818
+ * so the key-preferring record summary still applies.
819
+ */
820
+ function summarizeValue(value, maxLength = 96) {
821
+ if (value === undefined)
822
+ return null;
823
+ if (value === null)
824
+ return "null";
825
+ if (typeof value === "string") {
826
+ const trimmed = value.trim();
827
+ if (trimmed.startsWith("{") || trimmed.startsWith("[")) {
828
+ try {
829
+ return summarizeValue(JSON.parse(trimmed), maxLength);
830
+ }
831
+ catch {
832
+ // Not valid JSON — fall through to plain-string handling.
833
+ }
834
+ }
835
+ return truncateSummary(value, maxLength);
836
+ }
837
+ if (typeof value === "number" || typeof value === "boolean") {
838
+ return String(value);
839
+ }
840
+ if (Array.isArray(value)) {
841
+ if (value.length === 0)
842
+ return "array(0)";
843
+ const first = summaryPrimitive(value[0]);
844
+ const suffix = value.length > 1 ? " …" : "";
845
+ return truncateSummary(`array(${value.length}) ${first ?? ""}${suffix}`.trim(), maxLength);
846
+ }
847
+ if (typeof value === "object") {
848
+ return summarizeRecord(value, maxLength);
849
+ }
850
+ return truncateSummary(String(value), maxLength);
851
+ }
852
+ //# sourceMappingURL=reporter-format.js.map