@arizeai/phoenix-client 6.10.1 → 6.11.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (214) hide show
  1. package/README.md +62 -0
  2. package/dist/esm/__generated__/api/v1.d.ts +254 -2
  3. package/dist/esm/__generated__/api/v1.d.ts.map +1 -1
  4. package/dist/esm/jest/index.d.ts +5 -0
  5. package/dist/esm/jest/index.d.ts.map +1 -0
  6. package/dist/esm/jest/index.js +49 -0
  7. package/dist/esm/jest/index.js.map +1 -0
  8. package/dist/esm/jest/reporter.d.ts +13 -0
  9. package/dist/esm/jest/reporter.d.ts.map +1 -0
  10. package/dist/esm/jest/reporter.js +19 -0
  11. package/dist/esm/jest/reporter.js.map +1 -0
  12. package/dist/esm/prompts/sdks/toAI.d.ts +2 -2
  13. package/dist/esm/prompts/sdks/toAI.d.ts.map +1 -1
  14. package/dist/esm/prompts/sdks/toAI.js.map +1 -1
  15. package/dist/esm/prompts/sdks/toAnthropic.d.ts +2 -2
  16. package/dist/esm/prompts/sdks/toAnthropic.d.ts.map +1 -1
  17. package/dist/esm/prompts/sdks/toAnthropic.js.map +1 -1
  18. package/dist/esm/prompts/sdks/toOpenAI.d.ts +2 -2
  19. package/dist/esm/prompts/sdks/toOpenAI.d.ts.map +1 -1
  20. package/dist/esm/prompts/sdks/toOpenAI.js.map +1 -1
  21. package/dist/esm/prompts/sdks/toSDK.d.ts +8 -8
  22. package/dist/esm/prompts/sdks/toSDK.d.ts.map +1 -1
  23. package/dist/esm/prompts/sdks/toSDK.js.map +1 -1
  24. package/dist/esm/prompts/sdks/types.d.ts +2 -2
  25. package/dist/esm/prompts/sdks/types.d.ts.map +1 -1
  26. package/dist/esm/schemas/llm/anthropic/converters.d.ts +8 -8
  27. package/dist/esm/schemas/llm/anthropic/messagePartSchemas.d.ts +4 -4
  28. package/dist/esm/schemas/llm/anthropic/messageSchemas.d.ts +6 -6
  29. package/dist/esm/schemas/llm/constants.d.ts +3 -3
  30. package/dist/esm/schemas/llm/converters.d.ts +12 -12
  31. package/dist/esm/schemas/llm/openai/converters.d.ts +3 -3
  32. package/dist/esm/schemas/llm/schemas.d.ts +2 -2
  33. package/dist/esm/testing/acceptance.d.ts +20 -0
  34. package/dist/esm/testing/acceptance.d.ts.map +1 -0
  35. package/dist/esm/testing/acceptance.js +129 -0
  36. package/dist/esm/testing/acceptance.js.map +1 -0
  37. package/dist/esm/testing/define-api.d.ts +157 -0
  38. package/dist/esm/testing/define-api.d.ts.map +1 -0
  39. package/dist/esm/testing/define-api.js +78 -0
  40. package/dist/esm/testing/define-api.js.map +1 -0
  41. package/dist/esm/testing/helpers.d.ts +55 -0
  42. package/dist/esm/testing/helpers.d.ts.map +1 -0
  43. package/dist/esm/testing/helpers.js +179 -0
  44. package/dist/esm/testing/helpers.js.map +1 -0
  45. package/dist/esm/testing/phoenix-test-tracking.d.ts +68 -0
  46. package/dist/esm/testing/phoenix-test-tracking.d.ts.map +1 -0
  47. package/dist/esm/testing/phoenix-test-tracking.js +521 -0
  48. package/dist/esm/testing/phoenix-test-tracking.js.map +1 -0
  49. package/dist/esm/testing/report-artifacts.d.ts +45 -0
  50. package/dist/esm/testing/report-artifacts.d.ts.map +1 -0
  51. package/dist/esm/testing/report-artifacts.js +218 -0
  52. package/dist/esm/testing/report-artifacts.js.map +1 -0
  53. package/dist/esm/testing/report-run.d.ts +22 -0
  54. package/dist/esm/testing/report-run.d.ts.map +1 -0
  55. package/dist/esm/testing/report-run.js +41 -0
  56. package/dist/esm/testing/report-run.js.map +1 -0
  57. package/dist/esm/testing/reporter-format.d.ts +83 -0
  58. package/dist/esm/testing/reporter-format.d.ts.map +1 -0
  59. package/dist/esm/testing/reporter-format.js +852 -0
  60. package/dist/esm/testing/reporter-format.js.map +1 -0
  61. package/dist/esm/testing/runner.d.ts +31 -0
  62. package/dist/esm/testing/runner.d.ts.map +1 -0
  63. package/dist/esm/testing/runner.js +238 -0
  64. package/dist/esm/testing/runner.js.map +1 -0
  65. package/dist/esm/testing/state.d.ts +138 -0
  66. package/dist/esm/testing/state.d.ts.map +1 -0
  67. package/dist/esm/testing/state.js +31 -0
  68. package/dist/esm/testing/state.js.map +1 -0
  69. package/dist/esm/testing/types.d.ts +319 -0
  70. package/dist/esm/testing/types.d.ts.map +1 -0
  71. package/dist/esm/testing/types.js +9 -0
  72. package/dist/esm/testing/types.js.map +1 -0
  73. package/dist/esm/tsconfig.esm.tsbuildinfo +1 -1
  74. package/dist/esm/utils/channel.d.ts +7 -7
  75. package/dist/esm/utils/channel.d.ts.map +1 -1
  76. package/dist/esm/utils/channel.js +1 -1
  77. package/dist/esm/utils/channel.js.map +1 -1
  78. package/dist/esm/utils/formatPromptMessages.d.ts.map +1 -1
  79. package/dist/esm/utils/getPromptBySelector.d.ts.map +1 -1
  80. package/dist/esm/utils/promisifyResult.d.ts +1 -1
  81. package/dist/esm/utils/promisifyResult.d.ts.map +1 -1
  82. package/dist/esm/utils/promisifyResult.js.map +1 -1
  83. package/dist/esm/utils/schemaMatches.d.ts +5 -5
  84. package/dist/esm/utils/schemaMatches.d.ts.map +1 -1
  85. package/dist/esm/utils/schemaMatches.js.map +1 -1
  86. package/dist/esm/vitest/index.d.ts +5 -0
  87. package/dist/esm/vitest/index.d.ts.map +1 -0
  88. package/dist/esm/vitest/index.js +15 -0
  89. package/dist/esm/vitest/index.js.map +1 -0
  90. package/dist/esm/vitest/reporter.d.ts +19 -0
  91. package/dist/esm/vitest/reporter.d.ts.map +1 -0
  92. package/dist/esm/vitest/reporter.js +27 -0
  93. package/dist/esm/vitest/reporter.js.map +1 -0
  94. package/dist/src/__generated__/api/v1.d.ts +254 -2
  95. package/dist/src/__generated__/api/v1.d.ts.map +1 -1
  96. package/dist/src/jest/index.d.ts +5 -0
  97. package/dist/src/jest/index.d.ts.map +1 -0
  98. package/dist/src/jest/index.js +58 -0
  99. package/dist/src/jest/index.js.map +1 -0
  100. package/dist/src/jest/reporter.d.ts +13 -0
  101. package/dist/src/jest/reporter.d.ts.map +1 -0
  102. package/dist/src/jest/reporter.js +23 -0
  103. package/dist/src/jest/reporter.js.map +1 -0
  104. package/dist/src/prompts/sdks/toAI.d.ts +2 -2
  105. package/dist/src/prompts/sdks/toAI.d.ts.map +1 -1
  106. package/dist/src/prompts/sdks/toAI.js.map +1 -1
  107. package/dist/src/prompts/sdks/toAnthropic.d.ts +2 -2
  108. package/dist/src/prompts/sdks/toAnthropic.d.ts.map +1 -1
  109. package/dist/src/prompts/sdks/toAnthropic.js.map +1 -1
  110. package/dist/src/prompts/sdks/toOpenAI.d.ts +2 -2
  111. package/dist/src/prompts/sdks/toOpenAI.d.ts.map +1 -1
  112. package/dist/src/prompts/sdks/toOpenAI.js.map +1 -1
  113. package/dist/src/prompts/sdks/toSDK.d.ts +8 -8
  114. package/dist/src/prompts/sdks/toSDK.d.ts.map +1 -1
  115. package/dist/src/prompts/sdks/toSDK.js.map +1 -1
  116. package/dist/src/prompts/sdks/types.d.ts +2 -2
  117. package/dist/src/prompts/sdks/types.d.ts.map +1 -1
  118. package/dist/src/schemas/llm/anthropic/converters.d.ts +8 -8
  119. package/dist/src/schemas/llm/anthropic/messagePartSchemas.d.ts +4 -4
  120. package/dist/src/schemas/llm/anthropic/messageSchemas.d.ts +6 -6
  121. package/dist/src/schemas/llm/constants.d.ts +3 -3
  122. package/dist/src/schemas/llm/converters.d.ts +12 -12
  123. package/dist/src/schemas/llm/openai/converters.d.ts +3 -3
  124. package/dist/src/schemas/llm/schemas.d.ts +2 -2
  125. package/dist/src/testing/acceptance.d.ts +20 -0
  126. package/dist/src/testing/acceptance.d.ts.map +1 -0
  127. package/dist/src/testing/acceptance.js +114 -0
  128. package/dist/src/testing/acceptance.js.map +1 -0
  129. package/dist/src/testing/define-api.d.ts +157 -0
  130. package/dist/src/testing/define-api.d.ts.map +1 -0
  131. package/dist/src/testing/define-api.js +81 -0
  132. package/dist/src/testing/define-api.js.map +1 -0
  133. package/dist/src/testing/helpers.d.ts +55 -0
  134. package/dist/src/testing/helpers.d.ts.map +1 -0
  135. package/dist/src/testing/helpers.js +182 -0
  136. package/dist/src/testing/helpers.js.map +1 -0
  137. package/dist/src/testing/phoenix-test-tracking.d.ts +68 -0
  138. package/dist/src/testing/phoenix-test-tracking.d.ts.map +1 -0
  139. package/dist/src/testing/phoenix-test-tracking.js +530 -0
  140. package/dist/src/testing/phoenix-test-tracking.js.map +1 -0
  141. package/dist/src/testing/report-artifacts.d.ts +45 -0
  142. package/dist/src/testing/report-artifacts.d.ts.map +1 -0
  143. package/dist/src/testing/report-artifacts.js +225 -0
  144. package/dist/src/testing/report-artifacts.js.map +1 -0
  145. package/dist/src/testing/report-run.d.ts +22 -0
  146. package/dist/src/testing/report-run.d.ts.map +1 -0
  147. package/dist/src/testing/report-run.js +47 -0
  148. package/dist/src/testing/report-run.js.map +1 -0
  149. package/dist/src/testing/reporter-format.d.ts +83 -0
  150. package/dist/src/testing/reporter-format.d.ts.map +1 -0
  151. package/dist/src/testing/reporter-format.js +870 -0
  152. package/dist/src/testing/reporter-format.js.map +1 -0
  153. package/dist/src/testing/runner.d.ts +31 -0
  154. package/dist/src/testing/runner.d.ts.map +1 -0
  155. package/dist/src/testing/runner.js +258 -0
  156. package/dist/src/testing/runner.js.map +1 -0
  157. package/dist/src/testing/state.d.ts +138 -0
  158. package/dist/src/testing/state.d.ts.map +1 -0
  159. package/dist/src/testing/state.js +38 -0
  160. package/dist/src/testing/state.js.map +1 -0
  161. package/dist/src/testing/types.d.ts +319 -0
  162. package/dist/src/testing/types.d.ts.map +1 -0
  163. package/dist/src/testing/types.js +13 -0
  164. package/dist/src/testing/types.js.map +1 -0
  165. package/dist/src/utils/channel.d.ts +7 -7
  166. package/dist/src/utils/channel.d.ts.map +1 -1
  167. package/dist/src/utils/channel.js +1 -1
  168. package/dist/src/utils/channel.js.map +1 -1
  169. package/dist/src/utils/formatPromptMessages.d.ts.map +1 -1
  170. package/dist/src/utils/getPromptBySelector.d.ts.map +1 -1
  171. package/dist/src/utils/promisifyResult.d.ts +1 -1
  172. package/dist/src/utils/promisifyResult.d.ts.map +1 -1
  173. package/dist/src/utils/promisifyResult.js.map +1 -1
  174. package/dist/src/utils/schemaMatches.d.ts +5 -5
  175. package/dist/src/utils/schemaMatches.d.ts.map +1 -1
  176. package/dist/src/utils/schemaMatches.js.map +1 -1
  177. package/dist/src/vitest/index.d.ts +5 -0
  178. package/dist/src/vitest/index.d.ts.map +1 -0
  179. package/dist/src/vitest/index.js +23 -0
  180. package/dist/src/vitest/index.js.map +1 -0
  181. package/dist/src/vitest/reporter.d.ts +19 -0
  182. package/dist/src/vitest/reporter.d.ts.map +1 -0
  183. package/dist/src/vitest/reporter.js +34 -0
  184. package/dist/src/vitest/reporter.js.map +1 -0
  185. package/dist/tsconfig.tsbuildinfo +1 -1
  186. package/docs/ci-evals-annotations.mdx +190 -0
  187. package/docs/ci-evals-jest.mdx +78 -0
  188. package/docs/ci-evals-vitest.mdx +240 -0
  189. package/docs/ci-evals.mdx +263 -0
  190. package/docs/overview.mdx +9 -1
  191. package/package.json +49 -17
  192. package/src/__generated__/api/v1.ts +254 -2
  193. package/src/jest/index.ts +124 -0
  194. package/src/jest/reporter.ts +22 -0
  195. package/src/prompts/sdks/toAI.ts +4 -3
  196. package/src/prompts/sdks/toAnthropic.ts +4 -3
  197. package/src/prompts/sdks/toOpenAI.ts +4 -3
  198. package/src/prompts/sdks/toSDK.ts +16 -11
  199. package/src/prompts/sdks/types.ts +2 -2
  200. package/src/testing/acceptance.ts +190 -0
  201. package/src/testing/define-api.ts +279 -0
  202. package/src/testing/helpers.ts +251 -0
  203. package/src/testing/phoenix-test-tracking.ts +637 -0
  204. package/src/testing/report-artifacts.ts +272 -0
  205. package/src/testing/report-run.ts +44 -0
  206. package/src/testing/reporter-format.ts +1072 -0
  207. package/src/testing/runner.ts +350 -0
  208. package/src/testing/state.ts +165 -0
  209. package/src/testing/types.ts +366 -0
  210. package/src/utils/channel.ts +17 -15
  211. package/src/utils/promisifyResult.ts +6 -4
  212. package/src/utils/schemaMatches.ts +12 -10
  213. package/src/vitest/index.ts +57 -0
  214. package/src/vitest/reporter.ts +32 -0
@@ -0,0 +1,870 @@
1
+ "use strict";
2
+ Object.defineProperty(exports, "__esModule", { value: true });
3
+ exports.resolveRenderOptions = resolveRenderOptions;
4
+ exports.formatSuiteSummary = formatSuiteSummary;
5
+ exports.formatScoreboard = formatScoreboard;
6
+ exports.printSuiteSummaries = printSuiteSummaries;
7
+ const acceptance_1 = require("./acceptance");
8
+ // ---------------------------------------------------------------------------
9
+ // Option / environment resolution
10
+ // ---------------------------------------------------------------------------
11
+ function isTruthyFlag(value) {
12
+ const v = (value !== null && value !== void 0 ? value : "").toLowerCase();
13
+ return v === "true" || v === "1" || v === "on" || v === "yes";
14
+ }
15
+ function isFalsyFlag(value) {
16
+ const v = (value !== null && value !== void 0 ? value : "").toLowerCase();
17
+ return v === "false" || v === "0" || v === "off" || v === "no";
18
+ }
19
+ /**
20
+ * Resolve {@link RenderOptions} from the environment and output stream.
21
+ *
22
+ * Verbosity: `PHOENIX_TEST_REPORTER=verbose` (or the `PHOENIX_TEST_VERBOSE=1`
23
+ * alias) restores the full per-test dump; the default is the compact view.
24
+ * `PHOENIX_TEST_REPORTER_MAX_ROWS` caps the per-suite rows (default 10).
25
+ *
26
+ * Color follows the common ecosystem rules: off when `NO_COLOR` is set, in CI,
27
+ * on a non-TTY, or a "dumb" terminal; `PHOENIX_TEST_COLOR` / `FORCE_COLOR`
28
+ * force it on or off.
29
+ */
30
+ function resolveRenderOptions(env = process.env, stream = process.stdout) {
31
+ var _a, _b;
32
+ const verbose = ((_a = env.PHOENIX_TEST_REPORTER) !== null && _a !== void 0 ? _a : "").toLowerCase() === "verbose" ||
33
+ isTruthyFlag(env.PHOENIX_TEST_VERBOSE);
34
+ const parsedRows = Number.parseInt((_b = env.PHOENIX_TEST_REPORTER_MAX_ROWS) !== null && _b !== void 0 ? _b : "", 10);
35
+ const maxRows = Number.isFinite(parsedRows) && parsedRows > 0 ? parsedRows : 10;
36
+ const color = resolveColor(env, stream);
37
+ // When piped (no TTY width) assume a roomy-but-safe 100 columns so the
38
+ // overview table doesn't over-truncate suite names in CI logs.
39
+ const columns = typeof stream.columns === "number" && stream.columns > 0
40
+ ? stream.columns
41
+ : 100;
42
+ const maxWidth = Math.min(columns, 120);
43
+ return { verbose, color, maxRows, maxWidth };
44
+ }
45
+ function resolveColor(env, stream) {
46
+ if (isTruthyFlag(env.PHOENIX_TEST_COLOR))
47
+ return true;
48
+ if (isFalsyFlag(env.PHOENIX_TEST_COLOR))
49
+ return false;
50
+ if (env.FORCE_COLOR != null &&
51
+ env.FORCE_COLOR !== "" &&
52
+ env.FORCE_COLOR !== "0")
53
+ return true;
54
+ if (env.NO_COLOR != null && env.NO_COLOR !== "")
55
+ return false;
56
+ if (env.CI != null && env.CI !== "")
57
+ return false;
58
+ if (env.TERM === "dumb")
59
+ return false;
60
+ return stream.isTTY === true;
61
+ }
62
+ // ---------------------------------------------------------------------------
63
+ // Zero-dependency ASCII / ANSI toolkit
64
+ // ---------------------------------------------------------------------------
65
+ const ANSI = {
66
+ green: "\x1b[32m",
67
+ red: "\x1b[31m",
68
+ yellow: "\x1b[33m",
69
+ dim: "\x1b[2m",
70
+ bold: "\x1b[1m",
71
+ reset: "\x1b[0m",
72
+ };
73
+ /** Wrap a string in an ANSI color, or return it unchanged when color is off. */
74
+ function colorize(s, code, o) {
75
+ return o.color ? `${ANSI[code]}${s}${ANSI.reset}` : s;
76
+ }
77
+ // eslint-disable-next-line no-control-regex -- matching the ESC control byte is the point
78
+ const ANSI_PATTERN = /\x1b\[[0-9;]*m/g;
79
+ /** Visible length of a string, ignoring any ANSI escape codes. */
80
+ function visibleLen(s) {
81
+ return s.replace(ANSI_PATTERN, "").length;
82
+ }
83
+ /**
84
+ * Truncate `s` to `max` visible characters, keeping the head and tail with an
85
+ * ellipsis in the middle. Strings containing ANSI codes are returned unchanged
86
+ * to avoid slicing through an escape sequence (colored cells are always short
87
+ * enough to fit, so they never need truncating).
88
+ */
89
+ function truncateMiddle(s, max) {
90
+ if (max <= 1 || ANSI_PATTERN.test(s))
91
+ return s;
92
+ if (s.length <= max)
93
+ return s;
94
+ const head = Math.ceil((max - 1) / 2);
95
+ const tail = Math.floor((max - 1) / 2);
96
+ return `${s.slice(0, head)}…${tail > 0 ? s.slice(s.length - tail) : ""}`;
97
+ }
98
+ /** Left-pad-end a cell to `width`, measuring with {@link visibleLen}. */
99
+ function padCell(s, width) {
100
+ const pad = width - visibleLen(s);
101
+ return pad > 0 ? s + " ".repeat(pad) : s;
102
+ }
103
+ /**
104
+ * Render an aligned ASCII table (no table dependency). Column widths auto-size
105
+ * to content, clamped by `spec.caps`, and the first column is shrunk toward a
106
+ * floor when the table would exceed `o.maxWidth`. Cells longer than their final
107
+ * width are middle-truncated; widths are computed with {@link visibleLen} so
108
+ * ANSI color never breaks alignment.
109
+ */
110
+ function renderTable(headers, rows, o, spec = {}) {
111
+ const colCount = headers.length;
112
+ const gutter = " ";
113
+ const allRows = [headers, ...rows, ...(spec.footer ? [spec.footer] : [])];
114
+ const widths = headers.map((_, i) => {
115
+ var _a;
116
+ const natural = Math.max(...allRows.map((r) => { var _a; return visibleLen((_a = r[i]) !== null && _a !== void 0 ? _a : ""); }));
117
+ const cap = (_a = spec.caps) === null || _a === void 0 ? void 0 : _a[i];
118
+ return cap ? Math.min(natural, cap) : natural;
119
+ });
120
+ const FIRST_COL_FLOOR = 16;
121
+ const totalWidth = () => widths.reduce((a, b) => a + b, 0) + gutter.length * (colCount - 1);
122
+ while (totalWidth() > o.maxWidth && widths[0] > FIRST_COL_FLOOR) {
123
+ widths[0]--;
124
+ }
125
+ const lastCol = colCount - 1;
126
+ const formatRow = (cells) => cells
127
+ .map((cell, i) => {
128
+ const text = truncateMiddle(cell !== null && cell !== void 0 ? cell : "", widths[i]);
129
+ // The last column is never padded — trailing spaces are wasted tokens.
130
+ return i === lastCol ? text : padCell(text, widths[i]);
131
+ })
132
+ .join(gutter)
133
+ // Drop the dangling gutter when the final cell is empty (e.g. a clean
134
+ // suite's blank Result column).
135
+ .trimEnd();
136
+ const out = [formatRow(headers), ...rows.map(formatRow)];
137
+ if (spec.footer)
138
+ out.push(formatRow(spec.footer));
139
+ return out;
140
+ }
141
+ /**
142
+ * Aggregate every (non-`pass`) annotation across results, preserving the order
143
+ * in which annotation names are first seen. Numeric and boolean annotations are
144
+ * tracked separately; if a name appears as both, the first kind seen wins.
145
+ */
146
+ function computeAnnotationStats(results) {
147
+ const order = [];
148
+ const numeric = new Map();
149
+ const boolean = new Map();
150
+ for (const result of results) {
151
+ for (const ann of result.annotations) {
152
+ if (ann.name === "pass")
153
+ continue;
154
+ if (typeof ann.score === "number" && Number.isFinite(ann.score)) {
155
+ if (!numeric.has(ann.name) && !boolean.has(ann.name))
156
+ order.push(ann.name);
157
+ const arr = numeric.get(ann.name);
158
+ if (arr)
159
+ arr.push(ann.score);
160
+ else if (!boolean.has(ann.name))
161
+ numeric.set(ann.name, [ann.score]);
162
+ }
163
+ else if (typeof ann.score === "boolean") {
164
+ if (!numeric.has(ann.name) && !boolean.has(ann.name))
165
+ order.push(ann.name);
166
+ const cur = boolean.get(ann.name);
167
+ if (cur) {
168
+ cur.total++;
169
+ if (ann.score)
170
+ cur.t++;
171
+ }
172
+ else if (!numeric.has(ann.name)) {
173
+ boolean.set(ann.name, { t: ann.score ? 1 : 0, total: 1 });
174
+ }
175
+ }
176
+ }
177
+ }
178
+ return order.map((name) => {
179
+ const nums = numeric.get(name);
180
+ if (nums) {
181
+ return {
182
+ name,
183
+ kind: "number",
184
+ avg: nums.reduce((a, b) => a + b, 0) / nums.length,
185
+ count: nums.length,
186
+ };
187
+ }
188
+ const bools = boolean.get(name);
189
+ return { name, kind: "boolean", trueCount: bools.t, count: bools.total };
190
+ });
191
+ }
192
+ /** Format an annotation stat as the aggregate-line string used historically. */
193
+ function formatStat(stat) {
194
+ const samples = `${stat.count} sample${stat.count === 1 ? "" : "s"}`;
195
+ return stat.kind === "number"
196
+ ? `avg ${stat.avg.toFixed(3)} (${samples})`
197
+ : `${stat.trueCount}/${stat.count} true`;
198
+ }
199
+ /**
200
+ * Aggregate annotations into the legacy `name -> summary` string map (kept for
201
+ * the verbose view and any external callers).
202
+ */
203
+ function aggregateAnnotations(results) {
204
+ const out = {};
205
+ for (const stat of computeAnnotationStats(results)) {
206
+ out[stat.name] = formatStat(stat);
207
+ }
208
+ return out;
209
+ }
210
+ /**
211
+ * Per-annotation score bar a single run must clear to avoid counting as a
212
+ * "miss". Only `average` criteria contribute one: the aggregate `threshold`
213
+ * reused as a per-run heuristic (the suite-level acceptance block still reports
214
+ * the true aggregate verdict). `passRate` criteria decide passing with an
215
+ * arbitrary `passFn` predicate — there is no static numeric bar to highlight
216
+ * against — so their rows fall back to the default miss heuristic.
217
+ */
218
+ function buildAcceptanceBars(suite) {
219
+ var _a, _b;
220
+ const bars = new Map();
221
+ for (const result of (_a = suite.acceptanceResults) !== null && _a !== void 0 ? _a : []) {
222
+ if (result.metric !== "average")
223
+ continue;
224
+ bars.set(result.annotationName, {
225
+ bar: result.threshold,
226
+ direction: (_b = result.direction) !== null && _b !== void 0 ? _b : "maximize",
227
+ });
228
+ }
229
+ return bars;
230
+ }
231
+ /**
232
+ * Whether a passing test's evaluator scores fall short. When maximizing, a
233
+ * boolean `false` or a numeric score below its bar is a miss; when minimizing,
234
+ * a boolean `true` or a score above its bar is a miss. With no criterion for an
235
+ * annotation, only a non-positive score counts (keeps zero-config suites quiet).
236
+ */
237
+ function isMiss(result, bars) {
238
+ for (const ann of result.annotations) {
239
+ if (ann.name === "pass")
240
+ continue;
241
+ const acceptanceBar = bars.get(ann.name);
242
+ const minimizing = (acceptanceBar === null || acceptanceBar === void 0 ? void 0 : acceptanceBar.direction) === "minimize";
243
+ if (typeof ann.score === "boolean") {
244
+ if (minimizing ? ann.score : !ann.score)
245
+ return true;
246
+ }
247
+ else if (typeof ann.score === "number" && Number.isFinite(ann.score)) {
248
+ if (acceptanceBar === undefined) {
249
+ if (ann.score <= 0)
250
+ return true;
251
+ }
252
+ else if (minimizing
253
+ ? ann.score > acceptanceBar.bar
254
+ : ann.score < acceptanceBar.bar) {
255
+ return true;
256
+ }
257
+ }
258
+ }
259
+ return false;
260
+ }
261
+ /**
262
+ * Annotation columns for a suite's table: the acceptance-gated metrics first
263
+ * (the ones a user cares about), else the union of annotation names by
264
+ * first-seen order. Capped at three to keep the table narrow.
265
+ */
266
+ function selectAnnotationColumns(suite) {
267
+ var _a;
268
+ const MAX_COLUMNS = 3;
269
+ const gated = ((_a = suite.acceptanceResults) !== null && _a !== void 0 ? _a : []).map((r) => r.annotationName);
270
+ const ordered = gated.length > 0
271
+ ? gated
272
+ : computeAnnotationStats(suite.results).map((s) => s.name);
273
+ return [...new Set(ordered)].slice(0, MAX_COLUMNS);
274
+ }
275
+ /** The aggregate cell for an annotation column. */
276
+ function aggregateCell(stats, name) {
277
+ const stat = stats.find((s) => s.name === name);
278
+ if (!stat)
279
+ return "—";
280
+ return stat.kind === "number"
281
+ ? `avg ${stat.avg.toFixed(2)}`
282
+ : `${stat.trueCount}/${stat.count}`;
283
+ }
284
+ function computeSuiteVitals(suite) {
285
+ var _a;
286
+ const bars = buildAcceptanceBars(suite);
287
+ const total = suite.results.length;
288
+ const passed = suite.results.filter((r) => r.status === "passed").length;
289
+ const failed = suite.results.filter((r) => r.status === "failed").length;
290
+ const misses = suite.results
291
+ .filter((r) => r.status === "passed" && isMiss(r, bars))
292
+ .sort((a, b) => worstScore(a) - worstScore(b));
293
+ const acceptanceFailed = ((_a = suite.acceptanceResults) !== null && _a !== void 0 ? _a : []).some((r) => !r.passed);
294
+ const status = failed > 0 || acceptanceFailed
295
+ ? "fail"
296
+ : misses.length > 0
297
+ ? "miss"
298
+ : "pass";
299
+ const meanLatencyMs = total > 0 ? suite.results.reduce((a, r) => a + r.durationMs, 0) / total : 0;
300
+ return {
301
+ total,
302
+ passed,
303
+ failed,
304
+ missCount: misses.length,
305
+ status,
306
+ meanLatencyMs,
307
+ problems: [
308
+ ...suite.results.filter((r) => r.status === "failed"),
309
+ ...misses,
310
+ ],
311
+ acceptanceFailed,
312
+ };
313
+ }
314
+ /** The worst (lowest) evaluator score on a row, for ordering most-broken first. */
315
+ function worstScore(result) {
316
+ let worst = Number.POSITIVE_INFINITY;
317
+ for (const ann of result.annotations) {
318
+ if (ann.name === "pass")
319
+ continue;
320
+ worst = Math.min(worst, annotationScoreValue(ann));
321
+ }
322
+ return worst;
323
+ }
324
+ /** ANSI color matching a status (green pass / yellow miss / red fail). */
325
+ function statusColor(status) {
326
+ return status === "fail" ? "red" : status === "miss" ? "yellow" : "green";
327
+ }
328
+ /** `2/2 passed · 1 failed · 3 misses`, dropping any zero clause. */
329
+ function countsLabel(v) {
330
+ const parts = [`${v.passed}/${v.total} passed`];
331
+ if (v.failed > 0)
332
+ parts.push(`${v.failed} failed`);
333
+ if (v.missCount > 0)
334
+ parts.push(`${v.missCount} miss${v.missCount === 1 ? "" : "es"}`);
335
+ return parts.join(" · ");
336
+ }
337
+ /** Setup / upload problems worth surfacing regardless of test status. */
338
+ function warningLines(suite, o) {
339
+ var _a, _b;
340
+ const out = [];
341
+ if ((_a = suite.setupError) === null || _a === void 0 ? void 0 : _a.message) {
342
+ out.push(` ${colorize("setup error:", "red", o)} ${suite.setupError.message}`);
343
+ }
344
+ const n = (_b = suite.uploadFailureCount) !== null && _b !== void 0 ? _b : 0;
345
+ if (n > 0) {
346
+ out.push(` ${colorize("warning:", "yellow", o)} ${n} upload${n === 1 ? "" : "s"} failed (auth or network?)`);
347
+ }
348
+ return out;
349
+ }
350
+ /**
351
+ * Render a single suite. A clean suite collapses to one line; a suite with
352
+ * failures or misses expands into a per-row diagnosis (scores, rationale,
353
+ * output, and the Phoenix ids needed to pull the trace). Pass a verbose
354
+ * {@link RenderOptions} to restore the full per-test dump.
355
+ */
356
+ function formatSuiteSummary(suite, o = resolveRenderOptions()) {
357
+ return o.verbose
358
+ ? formatVerboseSuite(suite)
359
+ : formatSuiteDetail(suite, computeSuiteVitals(suite), o);
360
+ }
361
+ /** Verbatim acceptance-criteria block (kept stable for downstream parsers). */
362
+ function acceptanceLines(suite) {
363
+ if (!suite.acceptanceResults || suite.acceptanceResults.length === 0)
364
+ return [];
365
+ return [
366
+ " Acceptance Criteria:",
367
+ ...suite.acceptanceResults.map((r) => ` ${(0, acceptance_1.formatAcceptanceResult)(r)}`),
368
+ ];
369
+ }
370
+ function linkLines(suite) {
371
+ return suite.links.map((link) => ` ${link.label}: ${link.url}`);
372
+ }
373
+ /** Dim one-line roll-up of every metric average plus mean latency. */
374
+ function vitalsInline(suite, v, o) {
375
+ const parts = computeAnnotationStats(suite.results).map((s) => s.kind === "number"
376
+ ? `${s.name} ${s.avg.toFixed(2)}`
377
+ : `${s.name} ${s.trueCount}/${s.count}`);
378
+ parts.push(`avg ${formatDuration(v.meanLatencyMs)}`);
379
+ return colorize(parts.join(" "), "dim", o);
380
+ }
381
+ function formatSuiteDetail(suite, v, o) {
382
+ var _a;
383
+ const title = `${colorize(suite.name, statusColor(v.status), o)} ${colorize(countsLabel(v), "dim", o)}`;
384
+ const warnings = warningLines(suite, o);
385
+ // Clean suite with nothing to warn about: one line is the whole story.
386
+ if (v.status === "pass" && warnings.length === 0) {
387
+ return `${title} ${vitalsInline(suite, v, o)}`;
388
+ }
389
+ const lines = [title, ` ${vitalsInline(suite, v, o)}`, ...warnings];
390
+ // Never hide a hard failure; cap the number of below-bar misses shown.
391
+ const failures = v.problems.filter((r) => r.status === "failed");
392
+ const misses = v.problems.filter((r) => r.status !== "failed");
393
+ const shownMisses = misses.slice(0, Math.max(o.maxRows - failures.length, 0));
394
+ for (const r of [...failures, ...shownMisses]) {
395
+ lines.push(...problemEntry(r, o));
396
+ }
397
+ const hidden = misses.length - shownMisses.length;
398
+ if (hidden > 0) {
399
+ lines.push(colorize(` … ${hidden} more miss${hidden === 1 ? "" : "es"}`, "dim", o));
400
+ }
401
+ for (const a of (_a = suite.acceptanceResults) !== null && _a !== void 0 ? _a : []) {
402
+ if (!a.passed) {
403
+ lines.push(` ${colorize("✗ acceptance", "red", o)} ${(0, acceptance_1.formatAcceptanceResult)(a).replace(/^FAIL /, "")}`);
404
+ }
405
+ }
406
+ for (const link of suite.links) {
407
+ lines.push(` ${link.label}: ${link.url}`);
408
+ }
409
+ return lines.join("\n");
410
+ }
411
+ /**
412
+ * One failing / missing row: a weighted title with its scores, then the dim
413
+ * detail an agent needs to fix it — rationale, model output, and trace ids.
414
+ */
415
+ function problemEntry(result, o) {
416
+ var _a;
417
+ const mark = colorize("✗", result.status === "failed" ? "red" : "yellow", o);
418
+ const name = truncateEnd(humanizeLabel(result.testName), 72);
419
+ const dry = result.dryRun ? colorize(" (dry run)", "dim", o) : "";
420
+ const lines = [` ${mark} ${colorize(name, "bold", o)}${dry}`];
421
+ const indent = " ";
422
+ const err = compactError(result.error);
423
+ if (err)
424
+ lines.push(`${indent}${colorize(err, "red", o)}`);
425
+ // The sub-perfect evaluators that dragged the row down, worst score first —
426
+ // a clean `1.0` metric isn't what broke it, so it stays out of the way.
427
+ const rationales = [...result.annotations]
428
+ .filter((a) => a.name !== "pass" && annotationScoreValue(a) < 1)
429
+ .sort((a, b) => annotationScoreValue(a) - annotationScoreValue(b))
430
+ .slice(0, 3);
431
+ for (const ann of rationales) {
432
+ const reason = (_a = ann.explanation) !== null && _a !== void 0 ? _a : ann.label;
433
+ const tail = reason
434
+ ? ` ${colorize("·", "dim", o)} ${truncateSummary(reason, 160)}`
435
+ : "";
436
+ lines.push(`${indent}${colorize(ann.name, "dim", o)} ${formatScore(ann)}${tail}`);
437
+ }
438
+ const output = summarizeValue(result.output, 160);
439
+ if (output !== null) {
440
+ lines.push(`${indent}${colorize("output", "dim", o)} ${output}`);
441
+ }
442
+ const ids = formatResultIds(result);
443
+ if (ids)
444
+ lines.push(`${indent}${colorize(ids, "dim", o)}`);
445
+ return lines;
446
+ }
447
+ /** The legacy verbose view: every test as a delimited block including output. */
448
+ function formatVerboseSuite(suite) {
449
+ const lines = [];
450
+ lines.push("");
451
+ lines.push(suite.name);
452
+ if (suite.trackingDisabled) {
453
+ lines.push(` (tracking disabled — ${friendlyTrackingReason(suite)})`);
454
+ }
455
+ const total = suite.results.length;
456
+ const passed = suite.results.filter((r) => r.status === "passed").length;
457
+ const failed = suite.results.filter((r) => r.status === "failed").length;
458
+ lines.push(` ${passed}/${total} passed${failed ? `, ${failed} failed` : ""}`);
459
+ if (suite.uploadFailureCount && suite.uploadFailureCount > 0) {
460
+ lines.push(` warning: ${suite.uploadFailureCount} upload${suite.uploadFailureCount === 1 ? "" : "s"} failed (auth or network?)`);
461
+ }
462
+ const aggregated = aggregateAnnotations(suite.results);
463
+ for (const [name, summary] of Object.entries(aggregated)) {
464
+ lines.push(` ${name}: ${summary}`);
465
+ }
466
+ lines.push(...acceptanceLines(suite));
467
+ for (const result of suite.results) {
468
+ const status = result.status === "passed"
469
+ ? "PASS"
470
+ : result.status === "failed"
471
+ ? "FAIL"
472
+ : "SKIP";
473
+ const tag = result.dryRun ? " (dry run — not uploaded)" : "";
474
+ const annotations = formatAnnotationsInline(result.annotations);
475
+ const annotationSuffix = annotations ? ` → ${annotations}` : "";
476
+ lines.push("");
477
+ lines.push(` [${status}] ${humanizeLabel(result.testName)} (${formatDuration(result.durationMs)})${tag}${annotationSuffix}`);
478
+ if (result.error) {
479
+ lines.push(` error: ${result.error}`);
480
+ }
481
+ if (result.output !== undefined) {
482
+ lines.push(` output: ${stringifyForLog(result.output)}`);
483
+ }
484
+ for (const ann of result.annotations) {
485
+ if (ann.name !== "pass" && ann.explanation) {
486
+ lines.push(` why (${ann.name}): ${ann.explanation}`);
487
+ }
488
+ }
489
+ const ids = formatResultIds(result);
490
+ if (ids) {
491
+ lines.push(` ids: ${ids}`);
492
+ }
493
+ }
494
+ lines.push(...linkLines(suite));
495
+ return lines.join("\n");
496
+ }
497
+ // ---------------------------------------------------------------------------
498
+ // Cross-suite overview
499
+ // ---------------------------------------------------------------------------
500
+ /** Friendly, env-var-free reason a suite ran locally. */
501
+ function friendlyTrackingReason(suite) {
502
+ var _a, _b;
503
+ return (_b = (_a = suite.setupError) === null || _a === void 0 ? void 0 : _a.message) !== null && _b !== void 0 ? _b : "local only";
504
+ }
505
+ /** A single tracking note when every suite ran locally for the same reason. */
506
+ function sharedTrackingNote(suites) {
507
+ return suites.length > 0 &&
508
+ suites.every((s) => s.trackingDisabled && !s.setupError)
509
+ ? "tracking disabled (local only)"
510
+ : undefined;
511
+ }
512
+ /** The primary metric stat for a suite's overview row, or `null`. */
513
+ function primaryStat(suite) {
514
+ var _a;
515
+ const stats = computeAnnotationStats(suite.results);
516
+ const primary = selectAnnotationColumns(suite)[0];
517
+ return (_a = stats.find((s) => s.name === primary)) !== null && _a !== void 0 ? _a : null;
518
+ }
519
+ /**
520
+ * Render the run header (totals + tracking note) and, for multi-suite runs, an
521
+ * aligned overview table: one row per suite with its pass count, primary metric,
522
+ * acceptance verdict, mean latency, and a miss/fail note. This is the index;
523
+ * only suites with problems are expanded into a detail block below it.
524
+ */
525
+ function formatScoreboard(suites, o = resolveRenderOptions()) {
526
+ if (suites.length === 0)
527
+ return "";
528
+ const vitals = suites.map(computeSuiteVitals);
529
+ const passed = vitals.reduce((a, v) => a + v.passed, 0);
530
+ const total = vitals.reduce((a, v) => a + v.total, 0);
531
+ const failedTests = vitals.reduce((a, v) => a + v.failed, 0);
532
+ const misses = vitals.reduce((a, v) => a + v.missCount, 0);
533
+ const acceptFails = vitals.filter((v) => v.acceptanceFailed).length;
534
+ const uploadFails = suites.reduce((a, s) => { var _a; return a + ((_a = s.uploadFailureCount) !== null && _a !== void 0 ? _a : 0); }, 0);
535
+ // Totals are test/row-level so the clauses stay in one unit; the per-suite
536
+ // breakdown lives in the table below.
537
+ const header = [
538
+ "Eval Results",
539
+ `${suites.length} suite${suites.length === 1 ? "" : "s"}`,
540
+ `${passed}/${total} passed`,
541
+ ];
542
+ if (failedTests > 0)
543
+ header.push(`${failedTests} failed`);
544
+ if (misses > 0)
545
+ header.push(`${misses} miss${misses === 1 ? "" : "es"}`);
546
+ if (acceptFails > 0) {
547
+ header.push(`${acceptFails} acceptance failure${acceptFails === 1 ? "" : "s"}`);
548
+ }
549
+ if (uploadFails > 0)
550
+ header.push(`${uploadFails} uploads failed`);
551
+ const note = sharedTrackingNote(suites);
552
+ if (note)
553
+ header.push(note);
554
+ const headerLine = colorize(header.join(" · "), "bold", o);
555
+ // A single suite's own detail block is the overview; just print the header.
556
+ if (suites.length === 1)
557
+ return headerLine;
558
+ // Columns appear only when at least one suite has something to put in them.
559
+ const anyScore = suites.some((s) => primaryStat(s) !== null);
560
+ const anyAccept = suites.some((s) => { var _a; return ((_a = s.acceptanceResults) !== null && _a !== void 0 ? _a : []).length > 0; });
561
+ const anyLink = suites.some((s) => s.links.length > 0);
562
+ const primaries = suites.map((s) => selectAnnotationColumns(s)[0]);
563
+ const shared = primaries.every((p) => p && p === primaries[0]) && primaries[0]
564
+ ? primaries[0]
565
+ : undefined;
566
+ const headers = ["Suite", "Tests"];
567
+ const caps = [34, 7];
568
+ if (anyScore) {
569
+ headers.push(shared !== null && shared !== void 0 ? shared : "Score");
570
+ caps.push(22);
571
+ }
572
+ if (anyAccept) {
573
+ headers.push("Accept");
574
+ caps.push(7);
575
+ }
576
+ headers.push("Latency", "Result");
577
+ caps.push(8, 12);
578
+ if (anyLink) {
579
+ headers.push("Link");
580
+ caps.push(48);
581
+ }
582
+ const rows = suites.map((suite, i) => {
583
+ var _a, _b, _c;
584
+ const v = vitals[i];
585
+ const stats = computeAnnotationStats(suite.results);
586
+ const stat = primaryStat(suite);
587
+ const row = [suite.name, `${v.passed}/${v.total}`];
588
+ if (anyScore) {
589
+ row.push(!stat
590
+ ? "—"
591
+ : shared
592
+ ? aggregateCell(stats, stat.name)
593
+ : `${stat.name} ${aggregateCell(stats, stat.name).replace(/^avg /, "")}`);
594
+ }
595
+ if (anyAccept) {
596
+ row.push(((_a = suite.acceptanceResults) !== null && _a !== void 0 ? _a : []).length === 0
597
+ ? "—"
598
+ : v.acceptanceFailed
599
+ ? colorize("FAIL", "red", o)
600
+ : colorize("PASS", "green", o));
601
+ }
602
+ row.push(formatDuration(v.meanLatencyMs), resultNote(v, o));
603
+ if (anyLink)
604
+ row.push((_c = (_b = suite.links[0]) === null || _b === void 0 ? void 0 : _b.url) !== null && _c !== void 0 ? _c : "—");
605
+ return row;
606
+ });
607
+ return [headerLine, "", ...renderTable(headers, rows, o, { caps })].join("\n");
608
+ }
609
+ /** The overview "Result" cell: what went wrong, colored, or blank when clean. */
610
+ function resultNote(v, o) {
611
+ if (v.failed > 0)
612
+ return colorize(`${v.failed} failed`, "red", o);
613
+ if (v.acceptanceFailed)
614
+ return colorize("accept ✗", "red", o);
615
+ if (v.missCount > 0) {
616
+ return colorize(`${v.missCount} miss${v.missCount === 1 ? "" : "es"}`, "yellow", o);
617
+ }
618
+ return "";
619
+ }
620
+ // ---------------------------------------------------------------------------
621
+ // Entry point
622
+ // ---------------------------------------------------------------------------
623
+ /**
624
+ * Print the run summary: the overview header (and, for multi-suite runs, the
625
+ * index table), then an expanded detail block for every suite that failed or
626
+ * had misses. Clean suites are fully described by their overview row. A
627
+ * single-suite run always prints its block; verbose prints every block.
628
+ */
629
+ function printSuiteSummaries(suites) {
630
+ const o = resolveRenderOptions();
631
+ const overview = formatScoreboard(suites, o);
632
+ // eslint-disable-next-line no-console
633
+ if (overview)
634
+ console.log(overview);
635
+ const expand = o.verbose
636
+ ? suites
637
+ : suites.length === 1
638
+ ? suites
639
+ : suites.filter((s) => computeSuiteVitals(s).status !== "pass");
640
+ for (const suite of expand) {
641
+ // eslint-disable-next-line no-console
642
+ console.log(`\n${formatSuiteSummary(suite, o)}`);
643
+ }
644
+ }
645
+ // ---------------------------------------------------------------------------
646
+ // Shared formatters
647
+ // ---------------------------------------------------------------------------
648
+ /**
649
+ * Render a test's annotations as a compact, single-line `name=score` list for
650
+ * the verbose per-test header. The implicit `pass` annotation is omitted.
651
+ */
652
+ function formatAnnotationsInline(annotations) {
653
+ return annotations
654
+ .filter((ann) => ann.name !== "pass")
655
+ .map((ann) => `${ann.name}=${formatScore(ann)}`)
656
+ .join(", ");
657
+ }
658
+ function formatScore(ann) {
659
+ if (typeof ann.score === "number")
660
+ return ann.score.toString();
661
+ if (typeof ann.score === "boolean")
662
+ return ann.score ? "true" : "false";
663
+ if (ann.label)
664
+ return ann.label;
665
+ return "(no score)";
666
+ }
667
+ function formatDuration(ms) {
668
+ if (ms < 1000)
669
+ return `${Math.round(ms)}ms`;
670
+ return `${(ms / 1000).toFixed(2)}s`;
671
+ }
672
+ function stringifyForLog(value) {
673
+ try {
674
+ const json = JSON.stringify(value);
675
+ return json && json.length > 200
676
+ ? `${json.slice(0, 197)}...`
677
+ : (json !== null && json !== void 0 ? json : "");
678
+ }
679
+ catch (_a) {
680
+ return String(value);
681
+ }
682
+ }
683
+ // ---------------------------------------------------------------------------
684
+ // Per-failure detail helpers
685
+ //
686
+ // A problem row's block answers the two questions an agent needs to fix it:
687
+ // *why* (the judge's rationale and the model output) and *where* (the Phoenix
688
+ // trace / run / example ids it can pull for the full picture).
689
+ // ---------------------------------------------------------------------------
690
+ /** A run's numeric score for sorting (booleans as 1/0, missing as +∞). */
691
+ function annotationScoreValue(ann) {
692
+ if (typeof ann.score === "number")
693
+ return ann.score;
694
+ if (typeof ann.score === "boolean")
695
+ return ann.score ? 1 : 0;
696
+ return Number.POSITIVE_INFINITY;
697
+ }
698
+ /** First non-empty line of a multi-line error, truncated for one-line display. */
699
+ function compactError(error) {
700
+ if (!error)
701
+ return null;
702
+ const firstLine = error
703
+ .split("\n")
704
+ .map((line) => line.trim())
705
+ .find((line) => line.length > 0);
706
+ return firstLine ? truncateSummary(firstLine, 160) : null;
707
+ }
708
+ /** Phoenix ids for a run as a single `trace=… run=… example=…` string. */
709
+ function formatResultIds(result) {
710
+ const parts = [];
711
+ if (result.traceId)
712
+ parts.push(`trace=${result.traceId}`);
713
+ if (result.runId)
714
+ parts.push(`run=${result.runId}`);
715
+ if (result.exampleId)
716
+ parts.push(`example=${result.exampleId}`);
717
+ return parts.length > 0 ? parts.join(" ") : null;
718
+ }
719
+ // ---------------------------------------------------------------------------
720
+ // Label humanization
721
+ // ---------------------------------------------------------------------------
722
+ /**
723
+ * Turn a machine test name into a readable title. A `test.each` row is named by
724
+ * stringifying its input (`{"userQuery":"Show active users"}`); we surface the
725
+ * value of a single-field object directly (`Show active users`) and fold a
726
+ * multi-field object to `key=value` pairs. Non-JSON names pass through.
727
+ */
728
+ function humanizeLabel(name) {
729
+ var _a;
730
+ const trimmed = name.trim();
731
+ if (!(trimmed.startsWith("{") || trimmed.startsWith("[")))
732
+ return name;
733
+ let parsed;
734
+ try {
735
+ parsed = JSON.parse(trimmed);
736
+ }
737
+ catch (_b) {
738
+ return name;
739
+ }
740
+ if (Array.isArray(parsed))
741
+ return (_a = summarizeValue(parsed)) !== null && _a !== void 0 ? _a : name;
742
+ if (!parsed || typeof parsed !== "object")
743
+ return name;
744
+ const entries = Object.entries(parsed).filter(([, v]) => v != null && v !== "");
745
+ if (entries.length === 0)
746
+ return name;
747
+ if (entries.length === 1 && typeof entries[0][1] === "string") {
748
+ return entries[0][1];
749
+ }
750
+ return entries
751
+ .map(([k, v]) => { var _a; return `${k}=${(_a = summaryPrimitive(v)) !== null && _a !== void 0 ? _a : ""}`; })
752
+ .join(" ");
753
+ }
754
+ /** Truncate keeping the head (most identifying for a title), trailing ellipsis. */
755
+ function truncateEnd(s, max) {
756
+ return s.length <= max ? s : `${s.slice(0, max - 1)}…`;
757
+ }
758
+ // ---------------------------------------------------------------------------
759
+ // Token-efficient value summarization (adapted from the vitest-evals reporter)
760
+ //
761
+ // Replaces a blind JSON truncation with a key-preferring summary: the salient
762
+ // keys of an eval output (`score`, `output`, `error`, …) come first, primitives
763
+ // are rendered compactly, and JSON-encoded strings are parsed so the same
764
+ // summary applies. The result is far denser and more legible per token.
765
+ // ---------------------------------------------------------------------------
766
+ /** Keys surfaced first when summarizing a record, in priority order. */
767
+ const PREFERRED_SUMMARY_KEYS = [
768
+ "score",
769
+ "label",
770
+ "pass",
771
+ "passed",
772
+ "output",
773
+ "result",
774
+ "answer",
775
+ "response",
776
+ "reason",
777
+ "rationale",
778
+ "explanation",
779
+ "error",
780
+ "message",
781
+ "name",
782
+ "id",
783
+ "status",
784
+ ];
785
+ function truncateSummary(value, maxLength = 96) {
786
+ return value.length <= maxLength
787
+ ? value
788
+ : `${value.slice(0, maxLength - 1)}…`;
789
+ }
790
+ /** Render a single value as a short token: scalars inline, containers as counts. */
791
+ function summaryPrimitive(value) {
792
+ if (value === undefined)
793
+ return null;
794
+ if (value === null)
795
+ return "null";
796
+ if (typeof value === "string") {
797
+ const truncated = truncateSummary(value, 48);
798
+ // Bare-word strings stay unquoted; anything with spaces/punctuation is
799
+ // quoted so the key=value pairs remain unambiguous.
800
+ return /^[\w.:/@-]+$/.test(truncated)
801
+ ? truncated
802
+ : JSON.stringify(truncated);
803
+ }
804
+ if (typeof value === "number" || typeof value === "boolean") {
805
+ return String(value);
806
+ }
807
+ if (Array.isArray(value))
808
+ return `array(${value.length})`;
809
+ if (typeof value === "object") {
810
+ return `object(${Object.keys(value).length})`;
811
+ }
812
+ return String(value);
813
+ }
814
+ function summarizeRecord(record, maxLength) {
815
+ const keys = Object.keys(record);
816
+ if (keys.length === 0)
817
+ return "object(0)";
818
+ const ordered = [
819
+ ...PREFERRED_SUMMARY_KEYS.filter((key) => keys.includes(key)),
820
+ ...keys.filter((key) => !PREFERRED_SUMMARY_KEYS.includes(key)),
821
+ ].slice(0, 4);
822
+ const parts = ordered
823
+ .map((key) => {
824
+ const formatted = summaryPrimitive(record[key]);
825
+ return formatted === null ? null : `${key}=${formatted}`;
826
+ })
827
+ .filter((part) => part !== null);
828
+ if (parts.length === 0)
829
+ return null;
830
+ const suffix = keys.length > ordered.length ? " …" : "";
831
+ return truncateSummary(`${parts.join(" ")}${suffix}`, maxLength);
832
+ }
833
+ /**
834
+ * Summarize an arbitrary value to a compact, single-line string, or `null` when
835
+ * there's nothing to show (`undefined`). JSON-encoded strings are parsed first
836
+ * so the key-preferring record summary still applies.
837
+ */
838
+ function summarizeValue(value, maxLength = 96) {
839
+ if (value === undefined)
840
+ return null;
841
+ if (value === null)
842
+ return "null";
843
+ if (typeof value === "string") {
844
+ const trimmed = value.trim();
845
+ if (trimmed.startsWith("{") || trimmed.startsWith("[")) {
846
+ try {
847
+ return summarizeValue(JSON.parse(trimmed), maxLength);
848
+ }
849
+ catch (_a) {
850
+ // Not valid JSON — fall through to plain-string handling.
851
+ }
852
+ }
853
+ return truncateSummary(value, maxLength);
854
+ }
855
+ if (typeof value === "number" || typeof value === "boolean") {
856
+ return String(value);
857
+ }
858
+ if (Array.isArray(value)) {
859
+ if (value.length === 0)
860
+ return "array(0)";
861
+ const first = summaryPrimitive(value[0]);
862
+ const suffix = value.length > 1 ? " …" : "";
863
+ return truncateSummary(`array(${value.length}) ${first !== null && first !== void 0 ? first : ""}${suffix}`.trim(), maxLength);
864
+ }
865
+ if (typeof value === "object") {
866
+ return summarizeRecord(value, maxLength);
867
+ }
868
+ return truncateSummary(String(value), maxLength);
869
+ }
870
+ //# sourceMappingURL=reporter-format.js.map