@arizeai/phoenix-client 6.10.0 → 6.11.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (244) hide show
  1. package/README.md +62 -0
  2. package/dist/esm/__generated__/api/v1.d.ts +452 -28
  3. package/dist/esm/__generated__/api/v1.d.ts.map +1 -1
  4. package/dist/esm/experiments/helpers/getExampleGlobalId.d.ts +8 -0
  5. package/dist/esm/experiments/helpers/getExampleGlobalId.d.ts.map +1 -0
  6. package/dist/esm/experiments/helpers/getExampleGlobalId.js +9 -0
  7. package/dist/esm/experiments/helpers/getExampleGlobalId.js.map +1 -0
  8. package/dist/esm/experiments/resumeEvaluation.d.ts.map +1 -1
  9. package/dist/esm/experiments/resumeEvaluation.js +2 -1
  10. package/dist/esm/experiments/resumeEvaluation.js.map +1 -1
  11. package/dist/esm/experiments/resumeExperiment.d.ts.map +1 -1
  12. package/dist/esm/experiments/resumeExperiment.js +3 -2
  13. package/dist/esm/experiments/resumeExperiment.js.map +1 -1
  14. package/dist/esm/experiments/runExperiment.d.ts.map +1 -1
  15. package/dist/esm/experiments/runExperiment.js +6 -3
  16. package/dist/esm/experiments/runExperiment.js.map +1 -1
  17. package/dist/esm/jest/index.d.ts +5 -0
  18. package/dist/esm/jest/index.d.ts.map +1 -0
  19. package/dist/esm/jest/index.js +49 -0
  20. package/dist/esm/jest/index.js.map +1 -0
  21. package/dist/esm/jest/reporter.d.ts +13 -0
  22. package/dist/esm/jest/reporter.d.ts.map +1 -0
  23. package/dist/esm/jest/reporter.js +19 -0
  24. package/dist/esm/jest/reporter.js.map +1 -0
  25. package/dist/esm/prompts/sdks/toAI.d.ts +2 -2
  26. package/dist/esm/prompts/sdks/toAI.d.ts.map +1 -1
  27. package/dist/esm/prompts/sdks/toAI.js.map +1 -1
  28. package/dist/esm/prompts/sdks/toAnthropic.d.ts +2 -2
  29. package/dist/esm/prompts/sdks/toAnthropic.d.ts.map +1 -1
  30. package/dist/esm/prompts/sdks/toAnthropic.js.map +1 -1
  31. package/dist/esm/prompts/sdks/toOpenAI.d.ts +2 -2
  32. package/dist/esm/prompts/sdks/toOpenAI.d.ts.map +1 -1
  33. package/dist/esm/prompts/sdks/toOpenAI.js.map +1 -1
  34. package/dist/esm/prompts/sdks/toSDK.d.ts +8 -8
  35. package/dist/esm/prompts/sdks/toSDK.d.ts.map +1 -1
  36. package/dist/esm/prompts/sdks/toSDK.js.map +1 -1
  37. package/dist/esm/prompts/sdks/types.d.ts +2 -2
  38. package/dist/esm/prompts/sdks/types.d.ts.map +1 -1
  39. package/dist/esm/schemas/llm/anthropic/converters.d.ts +8 -8
  40. package/dist/esm/schemas/llm/anthropic/messagePartSchemas.d.ts +4 -4
  41. package/dist/esm/schemas/llm/anthropic/messageSchemas.d.ts +6 -6
  42. package/dist/esm/schemas/llm/constants.d.ts +3 -3
  43. package/dist/esm/schemas/llm/converters.d.ts +12 -12
  44. package/dist/esm/schemas/llm/openai/converters.d.ts +3 -3
  45. package/dist/esm/schemas/llm/schemas.d.ts +2 -2
  46. package/dist/esm/testing/acceptance.d.ts +20 -0
  47. package/dist/esm/testing/acceptance.d.ts.map +1 -0
  48. package/dist/esm/testing/acceptance.js +129 -0
  49. package/dist/esm/testing/acceptance.js.map +1 -0
  50. package/dist/esm/testing/define-api.d.ts +157 -0
  51. package/dist/esm/testing/define-api.d.ts.map +1 -0
  52. package/dist/esm/testing/define-api.js +78 -0
  53. package/dist/esm/testing/define-api.js.map +1 -0
  54. package/dist/esm/testing/helpers.d.ts +55 -0
  55. package/dist/esm/testing/helpers.d.ts.map +1 -0
  56. package/dist/esm/testing/helpers.js +179 -0
  57. package/dist/esm/testing/helpers.js.map +1 -0
  58. package/dist/esm/testing/phoenix-test-tracking.d.ts +68 -0
  59. package/dist/esm/testing/phoenix-test-tracking.d.ts.map +1 -0
  60. package/dist/esm/testing/phoenix-test-tracking.js +521 -0
  61. package/dist/esm/testing/phoenix-test-tracking.js.map +1 -0
  62. package/dist/esm/testing/report-artifacts.d.ts +45 -0
  63. package/dist/esm/testing/report-artifacts.d.ts.map +1 -0
  64. package/dist/esm/testing/report-artifacts.js +218 -0
  65. package/dist/esm/testing/report-artifacts.js.map +1 -0
  66. package/dist/esm/testing/report-run.d.ts +22 -0
  67. package/dist/esm/testing/report-run.d.ts.map +1 -0
  68. package/dist/esm/testing/report-run.js +41 -0
  69. package/dist/esm/testing/report-run.js.map +1 -0
  70. package/dist/esm/testing/reporter-format.d.ts +83 -0
  71. package/dist/esm/testing/reporter-format.d.ts.map +1 -0
  72. package/dist/esm/testing/reporter-format.js +852 -0
  73. package/dist/esm/testing/reporter-format.js.map +1 -0
  74. package/dist/esm/testing/runner.d.ts +31 -0
  75. package/dist/esm/testing/runner.d.ts.map +1 -0
  76. package/dist/esm/testing/runner.js +238 -0
  77. package/dist/esm/testing/runner.js.map +1 -0
  78. package/dist/esm/testing/state.d.ts +138 -0
  79. package/dist/esm/testing/state.d.ts.map +1 -0
  80. package/dist/esm/testing/state.js +31 -0
  81. package/dist/esm/testing/state.js.map +1 -0
  82. package/dist/esm/testing/types.d.ts +319 -0
  83. package/dist/esm/testing/types.d.ts.map +1 -0
  84. package/dist/esm/testing/types.js +9 -0
  85. package/dist/esm/testing/types.js.map +1 -0
  86. package/dist/esm/tsconfig.esm.tsbuildinfo +1 -1
  87. package/dist/esm/utils/channel.d.ts +7 -7
  88. package/dist/esm/utils/channel.d.ts.map +1 -1
  89. package/dist/esm/utils/channel.js +1 -1
  90. package/dist/esm/utils/channel.js.map +1 -1
  91. package/dist/esm/utils/formatPromptMessages.d.ts.map +1 -1
  92. package/dist/esm/utils/getPromptBySelector.d.ts.map +1 -1
  93. package/dist/esm/utils/promisifyResult.d.ts +1 -1
  94. package/dist/esm/utils/promisifyResult.d.ts.map +1 -1
  95. package/dist/esm/utils/promisifyResult.js.map +1 -1
  96. package/dist/esm/utils/schemaMatches.d.ts +5 -5
  97. package/dist/esm/utils/schemaMatches.d.ts.map +1 -1
  98. package/dist/esm/utils/schemaMatches.js.map +1 -1
  99. package/dist/esm/vitest/index.d.ts +5 -0
  100. package/dist/esm/vitest/index.d.ts.map +1 -0
  101. package/dist/esm/vitest/index.js +15 -0
  102. package/dist/esm/vitest/index.js.map +1 -0
  103. package/dist/esm/vitest/reporter.d.ts +19 -0
  104. package/dist/esm/vitest/reporter.d.ts.map +1 -0
  105. package/dist/esm/vitest/reporter.js +27 -0
  106. package/dist/esm/vitest/reporter.js.map +1 -0
  107. package/dist/src/__generated__/api/v1.d.ts +452 -28
  108. package/dist/src/__generated__/api/v1.d.ts.map +1 -1
  109. package/dist/src/experiments/helpers/getExampleGlobalId.d.ts +8 -0
  110. package/dist/src/experiments/helpers/getExampleGlobalId.d.ts.map +1 -0
  111. package/dist/src/experiments/helpers/getExampleGlobalId.js +13 -0
  112. package/dist/src/experiments/helpers/getExampleGlobalId.js.map +1 -0
  113. package/dist/src/experiments/resumeEvaluation.d.ts.map +1 -1
  114. package/dist/src/experiments/resumeEvaluation.js +2 -1
  115. package/dist/src/experiments/resumeEvaluation.js.map +1 -1
  116. package/dist/src/experiments/resumeExperiment.d.ts.map +1 -1
  117. package/dist/src/experiments/resumeExperiment.js +3 -2
  118. package/dist/src/experiments/resumeExperiment.js.map +1 -1
  119. package/dist/src/experiments/runExperiment.d.ts.map +1 -1
  120. package/dist/src/experiments/runExperiment.js +6 -3
  121. package/dist/src/experiments/runExperiment.js.map +1 -1
  122. package/dist/src/jest/index.d.ts +5 -0
  123. package/dist/src/jest/index.d.ts.map +1 -0
  124. package/dist/src/jest/index.js +58 -0
  125. package/dist/src/jest/index.js.map +1 -0
  126. package/dist/src/jest/reporter.d.ts +13 -0
  127. package/dist/src/jest/reporter.d.ts.map +1 -0
  128. package/dist/src/jest/reporter.js +23 -0
  129. package/dist/src/jest/reporter.js.map +1 -0
  130. package/dist/src/prompts/sdks/toAI.d.ts +2 -2
  131. package/dist/src/prompts/sdks/toAI.d.ts.map +1 -1
  132. package/dist/src/prompts/sdks/toAI.js.map +1 -1
  133. package/dist/src/prompts/sdks/toAnthropic.d.ts +2 -2
  134. package/dist/src/prompts/sdks/toAnthropic.d.ts.map +1 -1
  135. package/dist/src/prompts/sdks/toAnthropic.js.map +1 -1
  136. package/dist/src/prompts/sdks/toOpenAI.d.ts +2 -2
  137. package/dist/src/prompts/sdks/toOpenAI.d.ts.map +1 -1
  138. package/dist/src/prompts/sdks/toOpenAI.js.map +1 -1
  139. package/dist/src/prompts/sdks/toSDK.d.ts +8 -8
  140. package/dist/src/prompts/sdks/toSDK.d.ts.map +1 -1
  141. package/dist/src/prompts/sdks/toSDK.js.map +1 -1
  142. package/dist/src/prompts/sdks/types.d.ts +2 -2
  143. package/dist/src/prompts/sdks/types.d.ts.map +1 -1
  144. package/dist/src/schemas/llm/anthropic/converters.d.ts +8 -8
  145. package/dist/src/schemas/llm/anthropic/messagePartSchemas.d.ts +4 -4
  146. package/dist/src/schemas/llm/anthropic/messageSchemas.d.ts +6 -6
  147. package/dist/src/schemas/llm/constants.d.ts +3 -3
  148. package/dist/src/schemas/llm/converters.d.ts +12 -12
  149. package/dist/src/schemas/llm/openai/converters.d.ts +3 -3
  150. package/dist/src/schemas/llm/schemas.d.ts +2 -2
  151. package/dist/src/testing/acceptance.d.ts +20 -0
  152. package/dist/src/testing/acceptance.d.ts.map +1 -0
  153. package/dist/src/testing/acceptance.js +114 -0
  154. package/dist/src/testing/acceptance.js.map +1 -0
  155. package/dist/src/testing/define-api.d.ts +157 -0
  156. package/dist/src/testing/define-api.d.ts.map +1 -0
  157. package/dist/src/testing/define-api.js +81 -0
  158. package/dist/src/testing/define-api.js.map +1 -0
  159. package/dist/src/testing/helpers.d.ts +55 -0
  160. package/dist/src/testing/helpers.d.ts.map +1 -0
  161. package/dist/src/testing/helpers.js +182 -0
  162. package/dist/src/testing/helpers.js.map +1 -0
  163. package/dist/src/testing/phoenix-test-tracking.d.ts +68 -0
  164. package/dist/src/testing/phoenix-test-tracking.d.ts.map +1 -0
  165. package/dist/src/testing/phoenix-test-tracking.js +530 -0
  166. package/dist/src/testing/phoenix-test-tracking.js.map +1 -0
  167. package/dist/src/testing/report-artifacts.d.ts +45 -0
  168. package/dist/src/testing/report-artifacts.d.ts.map +1 -0
  169. package/dist/src/testing/report-artifacts.js +225 -0
  170. package/dist/src/testing/report-artifacts.js.map +1 -0
  171. package/dist/src/testing/report-run.d.ts +22 -0
  172. package/dist/src/testing/report-run.d.ts.map +1 -0
  173. package/dist/src/testing/report-run.js +47 -0
  174. package/dist/src/testing/report-run.js.map +1 -0
  175. package/dist/src/testing/reporter-format.d.ts +83 -0
  176. package/dist/src/testing/reporter-format.d.ts.map +1 -0
  177. package/dist/src/testing/reporter-format.js +870 -0
  178. package/dist/src/testing/reporter-format.js.map +1 -0
  179. package/dist/src/testing/runner.d.ts +31 -0
  180. package/dist/src/testing/runner.d.ts.map +1 -0
  181. package/dist/src/testing/runner.js +258 -0
  182. package/dist/src/testing/runner.js.map +1 -0
  183. package/dist/src/testing/state.d.ts +138 -0
  184. package/dist/src/testing/state.d.ts.map +1 -0
  185. package/dist/src/testing/state.js +38 -0
  186. package/dist/src/testing/state.js.map +1 -0
  187. package/dist/src/testing/types.d.ts +319 -0
  188. package/dist/src/testing/types.d.ts.map +1 -0
  189. package/dist/src/testing/types.js +13 -0
  190. package/dist/src/testing/types.js.map +1 -0
  191. package/dist/src/utils/channel.d.ts +7 -7
  192. package/dist/src/utils/channel.d.ts.map +1 -1
  193. package/dist/src/utils/channel.js +1 -1
  194. package/dist/src/utils/channel.js.map +1 -1
  195. package/dist/src/utils/formatPromptMessages.d.ts.map +1 -1
  196. package/dist/src/utils/getPromptBySelector.d.ts.map +1 -1
  197. package/dist/src/utils/promisifyResult.d.ts +1 -1
  198. package/dist/src/utils/promisifyResult.d.ts.map +1 -1
  199. package/dist/src/utils/promisifyResult.js.map +1 -1
  200. package/dist/src/utils/schemaMatches.d.ts +5 -5
  201. package/dist/src/utils/schemaMatches.d.ts.map +1 -1
  202. package/dist/src/utils/schemaMatches.js.map +1 -1
  203. package/dist/src/vitest/index.d.ts +5 -0
  204. package/dist/src/vitest/index.d.ts.map +1 -0
  205. package/dist/src/vitest/index.js +23 -0
  206. package/dist/src/vitest/index.js.map +1 -0
  207. package/dist/src/vitest/reporter.d.ts +19 -0
  208. package/dist/src/vitest/reporter.d.ts.map +1 -0
  209. package/dist/src/vitest/reporter.js +34 -0
  210. package/dist/src/vitest/reporter.js.map +1 -0
  211. package/dist/tsconfig.tsbuildinfo +1 -1
  212. package/docs/ci-evals-annotations.mdx +190 -0
  213. package/docs/ci-evals-jest.mdx +78 -0
  214. package/docs/ci-evals-vitest.mdx +240 -0
  215. package/docs/ci-evals.mdx +263 -0
  216. package/docs/overview.mdx +9 -1
  217. package/package.json +49 -17
  218. package/src/__generated__/api/v1.ts +452 -28
  219. package/src/experiments/helpers/getExampleGlobalId.ts +12 -0
  220. package/src/experiments/resumeEvaluation.ts +2 -1
  221. package/src/experiments/resumeExperiment.ts +3 -2
  222. package/src/experiments/runExperiment.ts +6 -3
  223. package/src/jest/index.ts +124 -0
  224. package/src/jest/reporter.ts +22 -0
  225. package/src/prompts/sdks/toAI.ts +4 -3
  226. package/src/prompts/sdks/toAnthropic.ts +4 -3
  227. package/src/prompts/sdks/toOpenAI.ts +4 -3
  228. package/src/prompts/sdks/toSDK.ts +16 -11
  229. package/src/prompts/sdks/types.ts +2 -2
  230. package/src/testing/acceptance.ts +190 -0
  231. package/src/testing/define-api.ts +279 -0
  232. package/src/testing/helpers.ts +251 -0
  233. package/src/testing/phoenix-test-tracking.ts +637 -0
  234. package/src/testing/report-artifacts.ts +272 -0
  235. package/src/testing/report-run.ts +44 -0
  236. package/src/testing/reporter-format.ts +1072 -0
  237. package/src/testing/runner.ts +350 -0
  238. package/src/testing/state.ts +165 -0
  239. package/src/testing/types.ts +366 -0
  240. package/src/utils/channel.ts +17 -15
  241. package/src/utils/promisifyResult.ts +6 -4
  242. package/src/utils/schemaMatches.ts +12 -10
  243. package/src/vitest/index.ts +57 -0
  244. package/src/vitest/reporter.ts +32 -0
@@ -0,0 +1,1072 @@
1
+ import { formatAcceptanceResult } from "./acceptance";
2
+ import type { TestResult } from "./state";
3
+ import type {
4
+ AcceptanceResult,
5
+ Annotation,
6
+ OptimizationDirection,
7
+ } from "./types";
8
+
9
+ /**
10
+ * The serializable summary of one suite that the reporter renders. This is the
11
+ * payload written to and read from the artifact files in `report-artifacts.ts`,
12
+ * so it holds only plain data — no clients, tracers, or live state.
13
+ */
14
+ export interface SuiteSummary {
15
+ /** Suite name (also the dataset / experiment name in Phoenix). */
16
+ name: string;
17
+ /** True when the suite did not sync to Phoenix (dry run, disabled, or error). */
18
+ trackingDisabled?: boolean;
19
+ /** Human-readable reason tracking was disabled, when known. */
20
+ trackingDisabledReason?: string;
21
+ /** Setup failure that disabled tracking, reduced to its message for printing. */
22
+ setupError?: { message: string };
23
+ /** Number of best-effort uploads (runs + annotations) that failed. */
24
+ uploadFailureCount?: number;
25
+ /** Per-test outcomes shown in the summary. */
26
+ results: TestResult[];
27
+ /** Aggregate acceptance results shown in the summary. */
28
+ acceptanceResults?: AcceptanceResult[];
29
+ /** Phoenix UI links (dataset / experiment) printed at the end of the block. */
30
+ links: Array<{ label: string; url: string }>;
31
+ }
32
+
33
+ /**
34
+ * Options that control how the reporter renders. Resolved once per run from the
35
+ * environment and the output stream (see {@link resolveRenderOptions}) and then
36
+ * threaded through every formatting function so jest and vitest behave
37
+ * identically without either reporter class needing options of its own.
38
+ */
39
+ export interface RenderOptions {
40
+ /** Show every test row plus the legacy per-test `output:` detail block. */
41
+ verbose: boolean;
42
+ /** Emit ANSI color escapes. */
43
+ color: boolean;
44
+ /** Max test rows shown per suite in compact mode (failures are never hidden). */
45
+ maxRows: number;
46
+ /** Terminal width budget used to size tables. */
47
+ maxWidth: number;
48
+ }
49
+
50
+ // ---------------------------------------------------------------------------
51
+ // Option / environment resolution
52
+ // ---------------------------------------------------------------------------
53
+
54
+ function isTruthyFlag(value: string | undefined): boolean {
55
+ const v = (value ?? "").toLowerCase();
56
+ return v === "true" || v === "1" || v === "on" || v === "yes";
57
+ }
58
+
59
+ function isFalsyFlag(value: string | undefined): boolean {
60
+ const v = (value ?? "").toLowerCase();
61
+ return v === "false" || v === "0" || v === "off" || v === "no";
62
+ }
63
+
64
+ /**
65
+ * Resolve {@link RenderOptions} from the environment and output stream.
66
+ *
67
+ * Verbosity: `PHOENIX_TEST_REPORTER=verbose` (or the `PHOENIX_TEST_VERBOSE=1`
68
+ * alias) restores the full per-test dump; the default is the compact view.
69
+ * `PHOENIX_TEST_REPORTER_MAX_ROWS` caps the per-suite rows (default 10).
70
+ *
71
+ * Color follows the common ecosystem rules: off when `NO_COLOR` is set, in CI,
72
+ * on a non-TTY, or a "dumb" terminal; `PHOENIX_TEST_COLOR` / `FORCE_COLOR`
73
+ * force it on or off.
74
+ */
75
+ export function resolveRenderOptions(
76
+ env: NodeJS.ProcessEnv = process.env,
77
+ stream: { isTTY?: boolean; columns?: number } = process.stdout
78
+ ): RenderOptions {
79
+ const verbose =
80
+ (env.PHOENIX_TEST_REPORTER ?? "").toLowerCase() === "verbose" ||
81
+ isTruthyFlag(env.PHOENIX_TEST_VERBOSE);
82
+
83
+ const parsedRows = Number.parseInt(
84
+ env.PHOENIX_TEST_REPORTER_MAX_ROWS ?? "",
85
+ 10
86
+ );
87
+ const maxRows =
88
+ Number.isFinite(parsedRows) && parsedRows > 0 ? parsedRows : 10;
89
+
90
+ const color = resolveColor(env, stream);
91
+
92
+ // When piped (no TTY width) assume a roomy-but-safe 100 columns so the
93
+ // overview table doesn't over-truncate suite names in CI logs.
94
+ const columns =
95
+ typeof stream.columns === "number" && stream.columns > 0
96
+ ? stream.columns
97
+ : 100;
98
+ const maxWidth = Math.min(columns, 120);
99
+
100
+ return { verbose, color, maxRows, maxWidth };
101
+ }
102
+
103
+ function resolveColor(
104
+ env: NodeJS.ProcessEnv,
105
+ stream: { isTTY?: boolean }
106
+ ): boolean {
107
+ if (isTruthyFlag(env.PHOENIX_TEST_COLOR)) return true;
108
+ if (isFalsyFlag(env.PHOENIX_TEST_COLOR)) return false;
109
+ if (
110
+ env.FORCE_COLOR != null &&
111
+ env.FORCE_COLOR !== "" &&
112
+ env.FORCE_COLOR !== "0"
113
+ )
114
+ return true;
115
+ if (env.NO_COLOR != null && env.NO_COLOR !== "") return false;
116
+ if (env.CI != null && env.CI !== "") return false;
117
+ if (env.TERM === "dumb") return false;
118
+ return stream.isTTY === true;
119
+ }
120
+
121
+ // ---------------------------------------------------------------------------
122
+ // Zero-dependency ASCII / ANSI toolkit
123
+ // ---------------------------------------------------------------------------
124
+
125
+ const ANSI = {
126
+ green: "\x1b[32m",
127
+ red: "\x1b[31m",
128
+ yellow: "\x1b[33m",
129
+ dim: "\x1b[2m",
130
+ bold: "\x1b[1m",
131
+ reset: "\x1b[0m",
132
+ } as const;
133
+
134
+ type AnsiColor = keyof Omit<typeof ANSI, "reset">;
135
+
136
+ /** Wrap a string in an ANSI color, or return it unchanged when color is off. */
137
+ function colorize(s: string, code: AnsiColor, o: RenderOptions): string {
138
+ return o.color ? `${ANSI[code]}${s}${ANSI.reset}` : s;
139
+ }
140
+
141
+ // eslint-disable-next-line no-control-regex -- matching the ESC control byte is the point
142
+ const ANSI_PATTERN = /\x1b\[[0-9;]*m/g;
143
+
144
+ /** Visible length of a string, ignoring any ANSI escape codes. */
145
+ function visibleLen(s: string): string["length"] {
146
+ return s.replace(ANSI_PATTERN, "").length;
147
+ }
148
+
149
+ /**
150
+ * Truncate `s` to `max` visible characters, keeping the head and tail with an
151
+ * ellipsis in the middle. Strings containing ANSI codes are returned unchanged
152
+ * to avoid slicing through an escape sequence (colored cells are always short
153
+ * enough to fit, so they never need truncating).
154
+ */
155
+ function truncateMiddle(s: string, max: number): string {
156
+ if (max <= 1 || ANSI_PATTERN.test(s)) return s;
157
+ if (s.length <= max) return s;
158
+ const head = Math.ceil((max - 1) / 2);
159
+ const tail = Math.floor((max - 1) / 2);
160
+ return `${s.slice(0, head)}…${tail > 0 ? s.slice(s.length - tail) : ""}`;
161
+ }
162
+
163
+ /** Left-pad-end a cell to `width`, measuring with {@link visibleLen}. */
164
+ function padCell(s: string, width: number): string {
165
+ const pad = width - visibleLen(s);
166
+ return pad > 0 ? s + " ".repeat(pad) : s;
167
+ }
168
+
169
+ interface TableSpec {
170
+ /** Per-column max width caps; columns without a cap auto-size. */
171
+ caps?: number[];
172
+ /** A summary row rendered as the final line (e.g. an AGGREGATE row). */
173
+ footer?: string[];
174
+ }
175
+
176
+ /**
177
+ * Render an aligned ASCII table (no table dependency). Column widths auto-size
178
+ * to content, clamped by `spec.caps`, and the first column is shrunk toward a
179
+ * floor when the table would exceed `o.maxWidth`. Cells longer than their final
180
+ * width are middle-truncated; widths are computed with {@link visibleLen} so
181
+ * ANSI color never breaks alignment.
182
+ */
183
+ function renderTable(
184
+ headers: string[],
185
+ rows: string[][],
186
+ o: RenderOptions,
187
+ spec: TableSpec = {}
188
+ ): string[] {
189
+ const colCount = headers.length;
190
+ const gutter = " ";
191
+ const allRows = [headers, ...rows, ...(spec.footer ? [spec.footer] : [])];
192
+
193
+ const widths = headers.map((_, i) => {
194
+ const natural = Math.max(...allRows.map((r) => visibleLen(r[i] ?? "")));
195
+ const cap = spec.caps?.[i];
196
+ return cap ? Math.min(natural, cap) : natural;
197
+ });
198
+
199
+ const FIRST_COL_FLOOR = 16;
200
+ const totalWidth = () =>
201
+ widths.reduce((a, b) => a + b, 0) + gutter.length * (colCount - 1);
202
+ while (totalWidth() > o.maxWidth && widths[0]! > FIRST_COL_FLOOR) {
203
+ widths[0]!--;
204
+ }
205
+
206
+ const lastCol = colCount - 1;
207
+ const formatRow = (cells: string[]) =>
208
+ cells
209
+ .map((cell, i) => {
210
+ const text = truncateMiddle(cell ?? "", widths[i]!);
211
+ // The last column is never padded — trailing spaces are wasted tokens.
212
+ return i === lastCol ? text : padCell(text, widths[i]!);
213
+ })
214
+ .join(gutter)
215
+ // Drop the dangling gutter when the final cell is empty (e.g. a clean
216
+ // suite's blank Result column).
217
+ .trimEnd();
218
+
219
+ const out = [formatRow(headers), ...rows.map(formatRow)];
220
+ if (spec.footer) out.push(formatRow(spec.footer));
221
+ return out;
222
+ }
223
+
224
+ // ---------------------------------------------------------------------------
225
+ // Annotation aggregation
226
+ // ---------------------------------------------------------------------------
227
+
228
+ /** Structured aggregate of one annotation across a suite's results. */
229
+ interface AnnotationStat {
230
+ name: string;
231
+ kind: "number" | "boolean";
232
+ /** Mean score for numeric annotations. */
233
+ avg?: number;
234
+ /** Count of `true` scores for boolean annotations. */
235
+ trueCount?: number;
236
+ /** Number of scored samples. */
237
+ count: number;
238
+ }
239
+
240
+ /**
241
+ * Aggregate every (non-`pass`) annotation across results, preserving the order
242
+ * in which annotation names are first seen. Numeric and boolean annotations are
243
+ * tracked separately; if a name appears as both, the first kind seen wins.
244
+ */
245
+ function computeAnnotationStats(
246
+ results: readonly TestResult[]
247
+ ): AnnotationStat[] {
248
+ const order: string[] = [];
249
+ const numeric = new Map<string, number[]>();
250
+ const boolean = new Map<string, { t: number; total: number }>();
251
+ for (const result of results) {
252
+ for (const ann of result.annotations) {
253
+ if (ann.name === "pass") continue;
254
+ if (typeof ann.score === "number" && Number.isFinite(ann.score)) {
255
+ if (!numeric.has(ann.name) && !boolean.has(ann.name))
256
+ order.push(ann.name);
257
+ const arr = numeric.get(ann.name);
258
+ if (arr) arr.push(ann.score);
259
+ else if (!boolean.has(ann.name)) numeric.set(ann.name, [ann.score]);
260
+ } else if (typeof ann.score === "boolean") {
261
+ if (!numeric.has(ann.name) && !boolean.has(ann.name))
262
+ order.push(ann.name);
263
+ const cur = boolean.get(ann.name);
264
+ if (cur) {
265
+ cur.total++;
266
+ if (ann.score) cur.t++;
267
+ } else if (!numeric.has(ann.name)) {
268
+ boolean.set(ann.name, { t: ann.score ? 1 : 0, total: 1 });
269
+ }
270
+ }
271
+ }
272
+ }
273
+ return order.map((name) => {
274
+ const nums = numeric.get(name);
275
+ if (nums) {
276
+ return {
277
+ name,
278
+ kind: "number",
279
+ avg: nums.reduce((a, b) => a + b, 0) / nums.length,
280
+ count: nums.length,
281
+ };
282
+ }
283
+ const bools = boolean.get(name)!;
284
+ return { name, kind: "boolean", trueCount: bools.t, count: bools.total };
285
+ });
286
+ }
287
+
288
+ /** Format an annotation stat as the aggregate-line string used historically. */
289
+ function formatStat(stat: AnnotationStat): string {
290
+ const samples = `${stat.count} sample${stat.count === 1 ? "" : "s"}`;
291
+ return stat.kind === "number"
292
+ ? `avg ${stat.avg!.toFixed(3)} (${samples})`
293
+ : `${stat.trueCount}/${stat.count} true`;
294
+ }
295
+
296
+ /**
297
+ * Aggregate annotations into the legacy `name -> summary` string map (kept for
298
+ * the verbose view and any external callers).
299
+ */
300
+ function aggregateAnnotations(
301
+ results: readonly TestResult[]
302
+ ): Record<string, string> {
303
+ const out: Record<string, string> = {};
304
+ for (const stat of computeAnnotationStats(results)) {
305
+ out[stat.name] = formatStat(stat);
306
+ }
307
+ return out;
308
+ }
309
+
310
+ // ---------------------------------------------------------------------------
311
+ // Miss detection + row selection
312
+ // ---------------------------------------------------------------------------
313
+
314
+ interface AcceptanceBar {
315
+ bar: number;
316
+ direction: OptimizationDirection;
317
+ }
318
+
319
+ /**
320
+ * Per-annotation score bar a single run must clear to avoid counting as a
321
+ * "miss". Only `average` criteria contribute one: the aggregate `threshold`
322
+ * reused as a per-run heuristic (the suite-level acceptance block still reports
323
+ * the true aggregate verdict). `passRate` criteria decide passing with an
324
+ * arbitrary `passFn` predicate — there is no static numeric bar to highlight
325
+ * against — so their rows fall back to the default miss heuristic.
326
+ */
327
+ function buildAcceptanceBars(suite: SuiteSummary): Map<string, AcceptanceBar> {
328
+ const bars = new Map<string, AcceptanceBar>();
329
+ for (const result of suite.acceptanceResults ?? []) {
330
+ if (result.metric !== "average") continue;
331
+ bars.set(result.annotationName, {
332
+ bar: result.threshold,
333
+ direction: result.direction ?? "maximize",
334
+ });
335
+ }
336
+ return bars;
337
+ }
338
+
339
+ /**
340
+ * Whether a passing test's evaluator scores fall short. When maximizing, a
341
+ * boolean `false` or a numeric score below its bar is a miss; when minimizing,
342
+ * a boolean `true` or a score above its bar is a miss. With no criterion for an
343
+ * annotation, only a non-positive score counts (keeps zero-config suites quiet).
344
+ */
345
+ function isMiss(result: TestResult, bars: Map<string, AcceptanceBar>): boolean {
346
+ for (const ann of result.annotations) {
347
+ if (ann.name === "pass") continue;
348
+ const acceptanceBar = bars.get(ann.name);
349
+ const minimizing = acceptanceBar?.direction === "minimize";
350
+ if (typeof ann.score === "boolean") {
351
+ if (minimizing ? ann.score : !ann.score) return true;
352
+ } else if (typeof ann.score === "number" && Number.isFinite(ann.score)) {
353
+ if (acceptanceBar === undefined) {
354
+ if (ann.score <= 0) return true;
355
+ } else if (
356
+ minimizing
357
+ ? ann.score > acceptanceBar.bar
358
+ : ann.score < acceptanceBar.bar
359
+ ) {
360
+ return true;
361
+ }
362
+ }
363
+ }
364
+ return false;
365
+ }
366
+
367
+ /**
368
+ * Annotation columns for a suite's table: the acceptance-gated metrics first
369
+ * (the ones a user cares about), else the union of annotation names by
370
+ * first-seen order. Capped at three to keep the table narrow.
371
+ */
372
+ function selectAnnotationColumns(suite: SuiteSummary): string[] {
373
+ const MAX_COLUMNS = 3;
374
+ const gated = (suite.acceptanceResults ?? []).map((r) => r.annotationName);
375
+ const ordered =
376
+ gated.length > 0
377
+ ? gated
378
+ : computeAnnotationStats(suite.results).map((s) => s.name);
379
+ return [...new Set(ordered)].slice(0, MAX_COLUMNS);
380
+ }
381
+
382
+ /** The aggregate cell for an annotation column. */
383
+ function aggregateCell(stats: AnnotationStat[], name: string): string {
384
+ const stat = stats.find((s) => s.name === name);
385
+ if (!stat) return "—";
386
+ return stat.kind === "number"
387
+ ? `avg ${stat.avg!.toFixed(2)}`
388
+ : `${stat.trueCount}/${stat.count}`;
389
+ }
390
+
391
+ // ---------------------------------------------------------------------------
392
+ // Suite rendering
393
+ // ---------------------------------------------------------------------------
394
+
395
+ type SuiteStatus = "pass" | "miss" | "fail";
396
+
397
+ /** A suite's headline numbers, computed once and shared by every renderer. */
398
+ interface SuiteVitals {
399
+ total: number;
400
+ passed: number;
401
+ failed: number;
402
+ missCount: number;
403
+ status: SuiteStatus;
404
+ meanLatencyMs: number;
405
+ /** Failing tests first, then below-bar misses, each worst-score first. */
406
+ problems: TestResult[];
407
+ acceptanceFailed: boolean;
408
+ }
409
+
410
+ function computeSuiteVitals(suite: SuiteSummary): SuiteVitals {
411
+ const bars = buildAcceptanceBars(suite);
412
+ const total = suite.results.length;
413
+ const passed = suite.results.filter((r) => r.status === "passed").length;
414
+ const failed = suite.results.filter((r) => r.status === "failed").length;
415
+ const misses = suite.results
416
+ .filter((r) => r.status === "passed" && isMiss(r, bars))
417
+ .sort((a, b) => worstScore(a) - worstScore(b));
418
+ const acceptanceFailed = (suite.acceptanceResults ?? []).some(
419
+ (r) => !r.passed
420
+ );
421
+ const status: SuiteStatus =
422
+ failed > 0 || acceptanceFailed
423
+ ? "fail"
424
+ : misses.length > 0
425
+ ? "miss"
426
+ : "pass";
427
+ const meanLatencyMs =
428
+ total > 0 ? suite.results.reduce((a, r) => a + r.durationMs, 0) / total : 0;
429
+ return {
430
+ total,
431
+ passed,
432
+ failed,
433
+ missCount: misses.length,
434
+ status,
435
+ meanLatencyMs,
436
+ problems: [
437
+ ...suite.results.filter((r) => r.status === "failed"),
438
+ ...misses,
439
+ ],
440
+ acceptanceFailed,
441
+ };
442
+ }
443
+
444
+ /** The worst (lowest) evaluator score on a row, for ordering most-broken first. */
445
+ function worstScore(result: TestResult): number {
446
+ let worst = Number.POSITIVE_INFINITY;
447
+ for (const ann of result.annotations) {
448
+ if (ann.name === "pass") continue;
449
+ worst = Math.min(worst, annotationScoreValue(ann));
450
+ }
451
+ return worst;
452
+ }
453
+
454
+ /** ANSI color matching a status (green pass / yellow miss / red fail). */
455
+ function statusColor(status: SuiteStatus): AnsiColor {
456
+ return status === "fail" ? "red" : status === "miss" ? "yellow" : "green";
457
+ }
458
+
459
+ /** `2/2 passed · 1 failed · 3 misses`, dropping any zero clause. */
460
+ function countsLabel(v: SuiteVitals): string {
461
+ const parts = [`${v.passed}/${v.total} passed`];
462
+ if (v.failed > 0) parts.push(`${v.failed} failed`);
463
+ if (v.missCount > 0)
464
+ parts.push(`${v.missCount} miss${v.missCount === 1 ? "" : "es"}`);
465
+ return parts.join(" · ");
466
+ }
467
+
468
+ /** Setup / upload problems worth surfacing regardless of test status. */
469
+ function warningLines(suite: SuiteSummary, o: RenderOptions): string[] {
470
+ const out: string[] = [];
471
+ if (suite.setupError?.message) {
472
+ out.push(
473
+ ` ${colorize("setup error:", "red", o)} ${suite.setupError.message}`
474
+ );
475
+ }
476
+ const n = suite.uploadFailureCount ?? 0;
477
+ if (n > 0) {
478
+ out.push(
479
+ ` ${colorize("warning:", "yellow", o)} ${n} upload${n === 1 ? "" : "s"} failed (auth or network?)`
480
+ );
481
+ }
482
+ return out;
483
+ }
484
+
485
+ /**
486
+ * Render a single suite. A clean suite collapses to one line; a suite with
487
+ * failures or misses expands into a per-row diagnosis (scores, rationale,
488
+ * output, and the Phoenix ids needed to pull the trace). Pass a verbose
489
+ * {@link RenderOptions} to restore the full per-test dump.
490
+ */
491
+ export function formatSuiteSummary(
492
+ suite: SuiteSummary,
493
+ o: RenderOptions = resolveRenderOptions()
494
+ ): string {
495
+ return o.verbose
496
+ ? formatVerboseSuite(suite)
497
+ : formatSuiteDetail(suite, computeSuiteVitals(suite), o);
498
+ }
499
+
500
+ /** Verbatim acceptance-criteria block (kept stable for downstream parsers). */
501
+ function acceptanceLines(suite: SuiteSummary): string[] {
502
+ if (!suite.acceptanceResults || suite.acceptanceResults.length === 0)
503
+ return [];
504
+ return [
505
+ " Acceptance Criteria:",
506
+ ...suite.acceptanceResults.map((r) => ` ${formatAcceptanceResult(r)}`),
507
+ ];
508
+ }
509
+
510
+ function linkLines(suite: SuiteSummary): string[] {
511
+ return suite.links.map((link) => ` ${link.label}: ${link.url}`);
512
+ }
513
+
514
+ /** Dim one-line roll-up of every metric average plus mean latency. */
515
+ function vitalsInline(
516
+ suite: SuiteSummary,
517
+ v: SuiteVitals,
518
+ o: RenderOptions
519
+ ): string {
520
+ const parts = computeAnnotationStats(suite.results).map((s) =>
521
+ s.kind === "number"
522
+ ? `${s.name} ${s.avg!.toFixed(2)}`
523
+ : `${s.name} ${s.trueCount}/${s.count}`
524
+ );
525
+ parts.push(`avg ${formatDuration(v.meanLatencyMs)}`);
526
+ return colorize(parts.join(" "), "dim", o);
527
+ }
528
+
529
+ function formatSuiteDetail(
530
+ suite: SuiteSummary,
531
+ v: SuiteVitals,
532
+ o: RenderOptions
533
+ ): string {
534
+ const title = `${colorize(suite.name, statusColor(v.status), o)} ${colorize(
535
+ countsLabel(v),
536
+ "dim",
537
+ o
538
+ )}`;
539
+ const warnings = warningLines(suite, o);
540
+
541
+ // Clean suite with nothing to warn about: one line is the whole story.
542
+ if (v.status === "pass" && warnings.length === 0) {
543
+ return `${title} ${vitalsInline(suite, v, o)}`;
544
+ }
545
+
546
+ const lines = [title, ` ${vitalsInline(suite, v, o)}`, ...warnings];
547
+ // Never hide a hard failure; cap the number of below-bar misses shown.
548
+ const failures = v.problems.filter((r) => r.status === "failed");
549
+ const misses = v.problems.filter((r) => r.status !== "failed");
550
+ const shownMisses = misses.slice(0, Math.max(o.maxRows - failures.length, 0));
551
+ for (const r of [...failures, ...shownMisses]) {
552
+ lines.push(...problemEntry(r, o));
553
+ }
554
+ const hidden = misses.length - shownMisses.length;
555
+ if (hidden > 0) {
556
+ lines.push(
557
+ colorize(` … ${hidden} more miss${hidden === 1 ? "" : "es"}`, "dim", o)
558
+ );
559
+ }
560
+ for (const a of suite.acceptanceResults ?? []) {
561
+ if (!a.passed) {
562
+ lines.push(
563
+ ` ${colorize("✗ acceptance", "red", o)} ${formatAcceptanceResult(
564
+ a
565
+ ).replace(/^FAIL /, "")}`
566
+ );
567
+ }
568
+ }
569
+ for (const link of suite.links) {
570
+ lines.push(` ${link.label}: ${link.url}`);
571
+ }
572
+ return lines.join("\n");
573
+ }
574
+
575
+ /**
576
+ * One failing / missing row: a weighted title with its scores, then the dim
577
+ * detail an agent needs to fix it — rationale, model output, and trace ids.
578
+ */
579
+ function problemEntry(result: TestResult, o: RenderOptions): string[] {
580
+ const mark = colorize("✗", result.status === "failed" ? "red" : "yellow", o);
581
+ const name = truncateEnd(humanizeLabel(result.testName), 72);
582
+ const dry = result.dryRun ? colorize(" (dry run)", "dim", o) : "";
583
+ const lines = [` ${mark} ${colorize(name, "bold", o)}${dry}`];
584
+ const indent = " ";
585
+
586
+ const err = compactError(result.error);
587
+ if (err) lines.push(`${indent}${colorize(err, "red", o)}`);
588
+
589
+ // The sub-perfect evaluators that dragged the row down, worst score first —
590
+ // a clean `1.0` metric isn't what broke it, so it stays out of the way.
591
+ const rationales = [...result.annotations]
592
+ .filter((a) => a.name !== "pass" && annotationScoreValue(a) < 1)
593
+ .sort((a, b) => annotationScoreValue(a) - annotationScoreValue(b))
594
+ .slice(0, 3);
595
+ for (const ann of rationales) {
596
+ const reason = ann.explanation ?? ann.label;
597
+ const tail = reason
598
+ ? ` ${colorize("·", "dim", o)} ${truncateSummary(reason, 160)}`
599
+ : "";
600
+ lines.push(
601
+ `${indent}${colorize(ann.name, "dim", o)} ${formatScore(ann)}${tail}`
602
+ );
603
+ }
604
+
605
+ const output = summarizeValue(result.output, 160);
606
+ if (output !== null) {
607
+ lines.push(`${indent}${colorize("output", "dim", o)} ${output}`);
608
+ }
609
+
610
+ const ids = formatResultIds(result);
611
+ if (ids) lines.push(`${indent}${colorize(ids, "dim", o)}`);
612
+
613
+ return lines;
614
+ }
615
+
616
+ /** The legacy verbose view: every test as a delimited block including output. */
617
+ function formatVerboseSuite(suite: SuiteSummary): string {
618
+ const lines: string[] = [];
619
+ lines.push("");
620
+ lines.push(suite.name);
621
+ if (suite.trackingDisabled) {
622
+ lines.push(` (tracking disabled — ${friendlyTrackingReason(suite)})`);
623
+ }
624
+ const total = suite.results.length;
625
+ const passed = suite.results.filter((r) => r.status === "passed").length;
626
+ const failed = suite.results.filter((r) => r.status === "failed").length;
627
+ lines.push(
628
+ ` ${passed}/${total} passed${failed ? `, ${failed} failed` : ""}`
629
+ );
630
+ if (suite.uploadFailureCount && suite.uploadFailureCount > 0) {
631
+ lines.push(
632
+ ` warning: ${suite.uploadFailureCount} upload${suite.uploadFailureCount === 1 ? "" : "s"} failed (auth or network?)`
633
+ );
634
+ }
635
+
636
+ const aggregated = aggregateAnnotations(suite.results);
637
+ for (const [name, summary] of Object.entries(aggregated)) {
638
+ lines.push(` ${name}: ${summary}`);
639
+ }
640
+ lines.push(...acceptanceLines(suite));
641
+
642
+ for (const result of suite.results) {
643
+ const status =
644
+ result.status === "passed"
645
+ ? "PASS"
646
+ : result.status === "failed"
647
+ ? "FAIL"
648
+ : "SKIP";
649
+ const tag = result.dryRun ? " (dry run — not uploaded)" : "";
650
+ const annotations = formatAnnotationsInline(result.annotations);
651
+ const annotationSuffix = annotations ? ` → ${annotations}` : "";
652
+ lines.push("");
653
+ lines.push(
654
+ ` [${status}] ${humanizeLabel(result.testName)} (${formatDuration(result.durationMs)})${tag}${annotationSuffix}`
655
+ );
656
+ if (result.error) {
657
+ lines.push(` error: ${result.error}`);
658
+ }
659
+ if (result.output !== undefined) {
660
+ lines.push(` output: ${stringifyForLog(result.output)}`);
661
+ }
662
+ for (const ann of result.annotations) {
663
+ if (ann.name !== "pass" && ann.explanation) {
664
+ lines.push(` why (${ann.name}): ${ann.explanation}`);
665
+ }
666
+ }
667
+ const ids = formatResultIds(result);
668
+ if (ids) {
669
+ lines.push(` ids: ${ids}`);
670
+ }
671
+ }
672
+
673
+ lines.push(...linkLines(suite));
674
+ return lines.join("\n");
675
+ }
676
+
677
+ // ---------------------------------------------------------------------------
678
+ // Cross-suite overview
679
+ // ---------------------------------------------------------------------------
680
+
681
+ /** Friendly, env-var-free reason a suite ran locally. */
682
+ function friendlyTrackingReason(suite: SuiteSummary): string {
683
+ return suite.setupError?.message ?? "local only";
684
+ }
685
+
686
+ /** A single tracking note when every suite ran locally for the same reason. */
687
+ function sharedTrackingNote(
688
+ suites: readonly SuiteSummary[]
689
+ ): string | undefined {
690
+ return suites.length > 0 &&
691
+ suites.every((s) => s.trackingDisabled && !s.setupError)
692
+ ? "tracking disabled (local only)"
693
+ : undefined;
694
+ }
695
+
696
+ /** The primary metric stat for a suite's overview row, or `null`. */
697
+ function primaryStat(suite: SuiteSummary): AnnotationStat | null {
698
+ const stats = computeAnnotationStats(suite.results);
699
+ const primary = selectAnnotationColumns(suite)[0];
700
+ return stats.find((s) => s.name === primary) ?? null;
701
+ }
702
+
703
+ /**
704
+ * Render the run header (totals + tracking note) and, for multi-suite runs, an
705
+ * aligned overview table: one row per suite with its pass count, primary metric,
706
+ * acceptance verdict, mean latency, and a miss/fail note. This is the index;
707
+ * only suites with problems are expanded into a detail block below it.
708
+ */
709
+ export function formatScoreboard(
710
+ suites: readonly SuiteSummary[],
711
+ o: RenderOptions = resolveRenderOptions()
712
+ ): string {
713
+ if (suites.length === 0) return "";
714
+
715
+ const vitals = suites.map(computeSuiteVitals);
716
+ const passed = vitals.reduce((a, v) => a + v.passed, 0);
717
+ const total = vitals.reduce((a, v) => a + v.total, 0);
718
+ const failedTests = vitals.reduce((a, v) => a + v.failed, 0);
719
+ const misses = vitals.reduce((a, v) => a + v.missCount, 0);
720
+ const acceptFails = vitals.filter((v) => v.acceptanceFailed).length;
721
+ const uploadFails = suites.reduce(
722
+ (a, s) => a + (s.uploadFailureCount ?? 0),
723
+ 0
724
+ );
725
+
726
+ // Totals are test/row-level so the clauses stay in one unit; the per-suite
727
+ // breakdown lives in the table below.
728
+ const header = [
729
+ "Eval Results",
730
+ `${suites.length} suite${suites.length === 1 ? "" : "s"}`,
731
+ `${passed}/${total} passed`,
732
+ ];
733
+ if (failedTests > 0) header.push(`${failedTests} failed`);
734
+ if (misses > 0) header.push(`${misses} miss${misses === 1 ? "" : "es"}`);
735
+ if (acceptFails > 0) {
736
+ header.push(
737
+ `${acceptFails} acceptance failure${acceptFails === 1 ? "" : "s"}`
738
+ );
739
+ }
740
+ if (uploadFails > 0) header.push(`${uploadFails} uploads failed`);
741
+ const note = sharedTrackingNote(suites);
742
+ if (note) header.push(note);
743
+ const headerLine = colorize(header.join(" · "), "bold", o);
744
+
745
+ // A single suite's own detail block is the overview; just print the header.
746
+ if (suites.length === 1) return headerLine;
747
+
748
+ // Columns appear only when at least one suite has something to put in them.
749
+ const anyScore = suites.some((s) => primaryStat(s) !== null);
750
+ const anyAccept = suites.some((s) => (s.acceptanceResults ?? []).length > 0);
751
+ const anyLink = suites.some((s) => s.links.length > 0);
752
+ const primaries = suites.map((s) => selectAnnotationColumns(s)[0]);
753
+ const shared =
754
+ primaries.every((p) => p && p === primaries[0]) && primaries[0]
755
+ ? primaries[0]
756
+ : undefined;
757
+
758
+ const headers = ["Suite", "Tests"];
759
+ const caps = [34, 7];
760
+ if (anyScore) {
761
+ headers.push(shared ?? "Score");
762
+ caps.push(22);
763
+ }
764
+ if (anyAccept) {
765
+ headers.push("Accept");
766
+ caps.push(7);
767
+ }
768
+ headers.push("Latency", "Result");
769
+ caps.push(8, 12);
770
+ if (anyLink) {
771
+ headers.push("Link");
772
+ caps.push(48);
773
+ }
774
+
775
+ const rows = suites.map((suite, i) => {
776
+ const v = vitals[i]!;
777
+ const stats = computeAnnotationStats(suite.results);
778
+ const stat = primaryStat(suite);
779
+ const row = [suite.name, `${v.passed}/${v.total}`];
780
+ if (anyScore) {
781
+ row.push(
782
+ !stat
783
+ ? "—"
784
+ : shared
785
+ ? aggregateCell(stats, stat.name)
786
+ : `${stat.name} ${aggregateCell(stats, stat.name).replace(/^avg /, "")}`
787
+ );
788
+ }
789
+ if (anyAccept) {
790
+ row.push(
791
+ (suite.acceptanceResults ?? []).length === 0
792
+ ? "—"
793
+ : v.acceptanceFailed
794
+ ? colorize("FAIL", "red", o)
795
+ : colorize("PASS", "green", o)
796
+ );
797
+ }
798
+ row.push(formatDuration(v.meanLatencyMs), resultNote(v, o));
799
+ if (anyLink) row.push(suite.links[0]?.url ?? "—");
800
+ return row;
801
+ });
802
+
803
+ return [headerLine, "", ...renderTable(headers, rows, o, { caps })].join(
804
+ "\n"
805
+ );
806
+ }
807
+
808
+ /** The overview "Result" cell: what went wrong, colored, or blank when clean. */
809
+ function resultNote(v: SuiteVitals, o: RenderOptions): string {
810
+ if (v.failed > 0) return colorize(`${v.failed} failed`, "red", o);
811
+ if (v.acceptanceFailed) return colorize("accept ✗", "red", o);
812
+ if (v.missCount > 0) {
813
+ return colorize(
814
+ `${v.missCount} miss${v.missCount === 1 ? "" : "es"}`,
815
+ "yellow",
816
+ o
817
+ );
818
+ }
819
+ return "";
820
+ }
821
+
822
+ // ---------------------------------------------------------------------------
823
+ // Entry point
824
+ // ---------------------------------------------------------------------------
825
+
826
+ /**
827
+ * Print the run summary: the overview header (and, for multi-suite runs, the
828
+ * index table), then an expanded detail block for every suite that failed or
829
+ * had misses. Clean suites are fully described by their overview row. A
830
+ * single-suite run always prints its block; verbose prints every block.
831
+ */
832
+ export function printSuiteSummaries(suites: readonly SuiteSummary[]): void {
833
+ const o = resolveRenderOptions();
834
+ const overview = formatScoreboard(suites, o);
835
+ // eslint-disable-next-line no-console
836
+ if (overview) console.log(overview);
837
+
838
+ const expand = o.verbose
839
+ ? suites
840
+ : suites.length === 1
841
+ ? suites
842
+ : suites.filter((s) => computeSuiteVitals(s).status !== "pass");
843
+ for (const suite of expand) {
844
+ // eslint-disable-next-line no-console
845
+ console.log(`\n${formatSuiteSummary(suite, o)}`);
846
+ }
847
+ }
848
+
849
+ // ---------------------------------------------------------------------------
850
+ // Shared formatters
851
+ // ---------------------------------------------------------------------------
852
+
853
+ /**
854
+ * Render a test's annotations as a compact, single-line `name=score` list for
855
+ * the verbose per-test header. The implicit `pass` annotation is omitted.
856
+ */
857
+ function formatAnnotationsInline(annotations: readonly Annotation[]): string {
858
+ return annotations
859
+ .filter((ann) => ann.name !== "pass")
860
+ .map((ann) => `${ann.name}=${formatScore(ann)}`)
861
+ .join(", ");
862
+ }
863
+
864
+ function formatScore(ann: Annotation): string {
865
+ if (typeof ann.score === "number") return ann.score.toString();
866
+ if (typeof ann.score === "boolean") return ann.score ? "true" : "false";
867
+ if (ann.label) return ann.label;
868
+ return "(no score)";
869
+ }
870
+
871
+ function formatDuration(ms: number): string {
872
+ if (ms < 1000) return `${Math.round(ms)}ms`;
873
+ return `${(ms / 1000).toFixed(2)}s`;
874
+ }
875
+
876
+ function stringifyForLog(value: unknown): string {
877
+ try {
878
+ const json = JSON.stringify(value);
879
+ return json && json.length > 200
880
+ ? `${json.slice(0, 197)}...`
881
+ : (json ?? "");
882
+ } catch {
883
+ return String(value);
884
+ }
885
+ }
886
+
887
+ // ---------------------------------------------------------------------------
888
+ // Per-failure detail helpers
889
+ //
890
+ // A problem row's block answers the two questions an agent needs to fix it:
891
+ // *why* (the judge's rationale and the model output) and *where* (the Phoenix
892
+ // trace / run / example ids it can pull for the full picture).
893
+ // ---------------------------------------------------------------------------
894
+
895
+ /** A run's numeric score for sorting (booleans as 1/0, missing as +∞). */
896
+ function annotationScoreValue(ann: Annotation): number {
897
+ if (typeof ann.score === "number") return ann.score;
898
+ if (typeof ann.score === "boolean") return ann.score ? 1 : 0;
899
+ return Number.POSITIVE_INFINITY;
900
+ }
901
+
902
+ /** First non-empty line of a multi-line error, truncated for one-line display. */
903
+ function compactError(error: string | undefined): string | null {
904
+ if (!error) return null;
905
+ const firstLine = error
906
+ .split("\n")
907
+ .map((line) => line.trim())
908
+ .find((line) => line.length > 0);
909
+ return firstLine ? truncateSummary(firstLine, 160) : null;
910
+ }
911
+
912
+ /** Phoenix ids for a run as a single `trace=… run=… example=…` string. */
913
+ function formatResultIds(result: TestResult): string | null {
914
+ const parts: string[] = [];
915
+ if (result.traceId) parts.push(`trace=${result.traceId}`);
916
+ if (result.runId) parts.push(`run=${result.runId}`);
917
+ if (result.exampleId) parts.push(`example=${result.exampleId}`);
918
+ return parts.length > 0 ? parts.join(" ") : null;
919
+ }
920
+
921
+ // ---------------------------------------------------------------------------
922
+ // Label humanization
923
+ // ---------------------------------------------------------------------------
924
+
925
+ /**
926
+ * Turn a machine test name into a readable title. A `test.each` row is named by
927
+ * stringifying its input (`{"userQuery":"Show active users"}`); we surface the
928
+ * value of a single-field object directly (`Show active users`) and fold a
929
+ * multi-field object to `key=value` pairs. Non-JSON names pass through.
930
+ */
931
+ function humanizeLabel(name: string): string {
932
+ const trimmed = name.trim();
933
+ if (!(trimmed.startsWith("{") || trimmed.startsWith("["))) return name;
934
+ let parsed: unknown;
935
+ try {
936
+ parsed = JSON.parse(trimmed);
937
+ } catch {
938
+ return name;
939
+ }
940
+ if (Array.isArray(parsed)) return summarizeValue(parsed) ?? name;
941
+ if (!parsed || typeof parsed !== "object") return name;
942
+ const entries = Object.entries(parsed as Record<string, unknown>).filter(
943
+ ([, v]) => v != null && v !== ""
944
+ );
945
+ if (entries.length === 0) return name;
946
+ if (entries.length === 1 && typeof entries[0]![1] === "string") {
947
+ return entries[0]![1] as string;
948
+ }
949
+ return entries
950
+ .map(([k, v]) => `${k}=${summaryPrimitive(v) ?? ""}`)
951
+ .join(" ");
952
+ }
953
+
954
+ /** Truncate keeping the head (most identifying for a title), trailing ellipsis. */
955
+ function truncateEnd(s: string, max: number): string {
956
+ return s.length <= max ? s : `${s.slice(0, max - 1)}…`;
957
+ }
958
+
959
+ // ---------------------------------------------------------------------------
960
+ // Token-efficient value summarization (adapted from the vitest-evals reporter)
961
+ //
962
+ // Replaces a blind JSON truncation with a key-preferring summary: the salient
963
+ // keys of an eval output (`score`, `output`, `error`, …) come first, primitives
964
+ // are rendered compactly, and JSON-encoded strings are parsed so the same
965
+ // summary applies. The result is far denser and more legible per token.
966
+ // ---------------------------------------------------------------------------
967
+
968
+ /** Keys surfaced first when summarizing a record, in priority order. */
969
+ const PREFERRED_SUMMARY_KEYS = [
970
+ "score",
971
+ "label",
972
+ "pass",
973
+ "passed",
974
+ "output",
975
+ "result",
976
+ "answer",
977
+ "response",
978
+ "reason",
979
+ "rationale",
980
+ "explanation",
981
+ "error",
982
+ "message",
983
+ "name",
984
+ "id",
985
+ "status",
986
+ ];
987
+
988
+ function truncateSummary(value: string, maxLength = 96): string {
989
+ return value.length <= maxLength
990
+ ? value
991
+ : `${value.slice(0, maxLength - 1)}…`;
992
+ }
993
+
994
+ /** Render a single value as a short token: scalars inline, containers as counts. */
995
+ function summaryPrimitive(value: unknown): string | null {
996
+ if (value === undefined) return null;
997
+ if (value === null) return "null";
998
+ if (typeof value === "string") {
999
+ const truncated = truncateSummary(value, 48);
1000
+ // Bare-word strings stay unquoted; anything with spaces/punctuation is
1001
+ // quoted so the key=value pairs remain unambiguous.
1002
+ return /^[\w.:/@-]+$/.test(truncated)
1003
+ ? truncated
1004
+ : JSON.stringify(truncated);
1005
+ }
1006
+ if (typeof value === "number" || typeof value === "boolean") {
1007
+ return String(value);
1008
+ }
1009
+ if (Array.isArray(value)) return `array(${value.length})`;
1010
+ if (typeof value === "object") {
1011
+ return `object(${Object.keys(value as Record<string, unknown>).length})`;
1012
+ }
1013
+ return String(value);
1014
+ }
1015
+
1016
+ function summarizeRecord(
1017
+ record: Record<string, unknown>,
1018
+ maxLength: number
1019
+ ): string | null {
1020
+ const keys = Object.keys(record);
1021
+ if (keys.length === 0) return "object(0)";
1022
+ const ordered = [
1023
+ ...PREFERRED_SUMMARY_KEYS.filter((key) => keys.includes(key)),
1024
+ ...keys.filter((key) => !PREFERRED_SUMMARY_KEYS.includes(key)),
1025
+ ].slice(0, 4);
1026
+ const parts = ordered
1027
+ .map((key) => {
1028
+ const formatted = summaryPrimitive(record[key]);
1029
+ return formatted === null ? null : `${key}=${formatted}`;
1030
+ })
1031
+ .filter((part): part is string => part !== null);
1032
+ if (parts.length === 0) return null;
1033
+ const suffix = keys.length > ordered.length ? " …" : "";
1034
+ return truncateSummary(`${parts.join(" ")}${suffix}`, maxLength);
1035
+ }
1036
+
1037
+ /**
1038
+ * Summarize an arbitrary value to a compact, single-line string, or `null` when
1039
+ * there's nothing to show (`undefined`). JSON-encoded strings are parsed first
1040
+ * so the key-preferring record summary still applies.
1041
+ */
1042
+ function summarizeValue(value: unknown, maxLength = 96): string | null {
1043
+ if (value === undefined) return null;
1044
+ if (value === null) return "null";
1045
+ if (typeof value === "string") {
1046
+ const trimmed = value.trim();
1047
+ if (trimmed.startsWith("{") || trimmed.startsWith("[")) {
1048
+ try {
1049
+ return summarizeValue(JSON.parse(trimmed), maxLength);
1050
+ } catch {
1051
+ // Not valid JSON — fall through to plain-string handling.
1052
+ }
1053
+ }
1054
+ return truncateSummary(value, maxLength);
1055
+ }
1056
+ if (typeof value === "number" || typeof value === "boolean") {
1057
+ return String(value);
1058
+ }
1059
+ if (Array.isArray(value)) {
1060
+ if (value.length === 0) return "array(0)";
1061
+ const first = summaryPrimitive(value[0]);
1062
+ const suffix = value.length > 1 ? " …" : "";
1063
+ return truncateSummary(
1064
+ `array(${value.length}) ${first ?? ""}${suffix}`.trim(),
1065
+ maxLength
1066
+ );
1067
+ }
1068
+ if (typeof value === "object") {
1069
+ return summarizeRecord(value as Record<string, unknown>, maxLength);
1070
+ }
1071
+ return truncateSummary(String(value), maxLength);
1072
+ }