niceeval 0.6.1 → 0.6.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (296) hide show
  1. package/dist/agents/types.d.ts +67 -5
  2. package/dist/context/types.d.ts +32 -12
  3. package/dist/i18n/en.d.ts +54 -0
  4. package/dist/i18n/zh-CN.d.ts +55 -1
  5. package/dist/o11y/types.d.ts +16 -2
  6. package/dist/report/aggregate.d.ts +5 -3
  7. package/dist/report/aggregate.js +32 -5
  8. package/dist/report/built-ins/experiment-comparison.d.ts +39 -1
  9. package/dist/report/built-ins/experiment-comparison.js +116 -10
  10. package/dist/report/built-ins/index.d.ts +1 -0
  11. package/dist/report/built-ins/index.js +1 -1
  12. package/dist/report/components.d.ts +8 -2
  13. package/dist/report/components.js +3 -3
  14. package/dist/report/compute.d.ts +11 -18
  15. package/dist/report/compute.js +54 -34
  16. package/dist/report/flag.d.ts +16 -1
  17. package/dist/report/flag.js +19 -1
  18. package/dist/report/format.d.ts +16 -8
  19. package/dist/report/format.js +27 -12
  20. package/dist/report/index.d.ts +4 -3
  21. package/dist/report/index.js +5 -4
  22. package/dist/report/locale.d.ts +11 -2
  23. package/dist/report/locale.js +23 -5
  24. package/dist/report/metrics.d.ts +13 -1
  25. package/dist/report/metrics.js +65 -14
  26. package/dist/report/primitives.d.ts +6 -0
  27. package/dist/report/react/AttemptList.d.ts +2 -2
  28. package/dist/report/react/AttemptList.js +5 -6
  29. package/dist/report/react/EvalList.d.ts +1 -1
  30. package/dist/report/react/EvalList.js +0 -0
  31. package/dist/report/react/ExperimentComparison.d.ts +8 -0
  32. package/dist/report/react/ExperimentComparison.js +11 -0
  33. package/dist/report/react/ExperimentList.d.ts +2 -1
  34. package/dist/report/react/ExperimentList.js +8 -10
  35. package/dist/report/react/MetricScatter.js +5 -11
  36. package/dist/report/react/chart-math.d.ts +23 -6
  37. package/dist/report/react/chart-math.js +71 -19
  38. package/dist/report/react/fixtures.d.ts +3 -3
  39. package/dist/report/react/fixtures.js +21 -14
  40. package/dist/report/report.d.ts +5 -1
  41. package/dist/report/report.js +6 -2
  42. package/dist/report/text/faces.d.ts +1 -1
  43. package/dist/report/text/faces.js +42 -41
  44. package/dist/report/text/table.js +36 -5
  45. package/dist/report/types.d.ts +39 -21
  46. package/dist/results/types.d.ts +11 -0
  47. package/dist/runner/feedback/sink.d.ts +110 -0
  48. package/dist/runner/types.d.ts +513 -22
  49. package/dist/sandbox/docker.d.ts +23 -2
  50. package/dist/sandbox/e2b.d.ts +15 -1
  51. package/dist/sandbox/errors.d.ts +30 -3
  52. package/dist/sandbox/io-retry.d.ts +17 -0
  53. package/dist/sandbox/registry.d.ts +2 -0
  54. package/dist/sandbox/resolve.d.ts +18 -5
  55. package/dist/sandbox/retry.d.ts +11 -1
  56. package/dist/sandbox/types.d.ts +39 -5
  57. package/dist/sandbox/vercel.d.ts +7 -1
  58. package/dist/scoring/coverage.d.ts +30 -0
  59. package/dist/scoring/display.d.ts +21 -0
  60. package/dist/scoring/display.js +120 -0
  61. package/dist/scoring/types.d.ts +103 -20
  62. package/dist/shared/aggregate.d.ts +1 -0
  63. package/dist/shared/aggregate.js +3 -3
  64. package/dist/shared/types.d.ts +28 -0
  65. package/dist/tty-line.d.ts +0 -4
  66. package/dist/util.d.ts +23 -0
  67. package/docs-site/zh/concepts/adapter.mdx +22 -4
  68. package/docs-site/zh/concepts/experiment.mdx +1 -1
  69. package/docs-site/zh/concepts/overview.mdx +6 -6
  70. package/docs-site/zh/guides/agent-feedback-loop.mdx +28 -26
  71. package/docs-site/zh/guides/authoring.mdx +33 -0
  72. package/docs-site/zh/guides/ci-integration.mdx +23 -12
  73. package/docs-site/zh/guides/connect-your-agent.mdx +29 -3
  74. package/docs-site/zh/guides/custom-reports.mdx +29 -34
  75. package/docs-site/zh/guides/dataset-fanout.mdx +25 -3
  76. package/docs-site/zh/guides/debug-sandbox.mdx +57 -0
  77. package/docs-site/zh/guides/debugging.mdx +210 -0
  78. package/docs-site/zh/guides/experiments.mdx +10 -3
  79. package/docs-site/zh/guides/official-adapters.mdx +26 -2
  80. package/docs-site/zh/guides/publish-report.mdx +30 -16
  81. package/docs-site/zh/guides/report-components.mdx +42 -30
  82. package/docs-site/zh/guides/reporters.mdx +2 -2
  83. package/docs-site/zh/guides/results-data.mdx +17 -9
  84. package/docs-site/zh/guides/runner.mdx +17 -7
  85. package/docs-site/zh/guides/sandbox-agent.mdx +56 -7
  86. package/docs-site/zh/guides/sandbox-providers.mdx +257 -9
  87. package/docs-site/zh/guides/scoring-guide.mdx +4 -4
  88. package/docs-site/zh/guides/viewing-results.mdx +79 -36
  89. package/docs-site/zh/guides/write-experiment.mdx +5 -3
  90. package/docs-site/zh/guides/write-send.mdx +17 -1
  91. package/docs-site/zh/index.mdx +1 -1
  92. package/docs-site/zh/reference/builtin-agents.mdx +27 -0
  93. package/docs-site/zh/reference/capabilities.mdx +2 -2
  94. package/docs-site/zh/reference/cli.mdx +33 -7
  95. package/docs-site/zh/reference/define-agent.mdx +57 -4
  96. package/docs-site/zh/reference/define-config.mdx +1 -1
  97. package/docs-site/zh/reference/define-eval.mdx +42 -9
  98. package/docs-site/zh/reference/expect.mdx +26 -1
  99. package/package.json +5 -1
  100. package/src/agents/ai-sdk-otel.test.ts +1 -0
  101. package/src/agents/ai-sdk.test.ts +3 -0
  102. package/src/agents/ai-sdk.ts +3 -0
  103. package/src/agents/bub-install-spec.test.ts +34 -0
  104. package/src/agents/bub-install-spec.ts +32 -0
  105. package/src/agents/bub.ts +31 -32
  106. package/src/agents/claude-code.test.ts +130 -9
  107. package/src/agents/claude-code.ts +76 -4
  108. package/src/agents/codex.test.ts +189 -40
  109. package/src/agents/codex.ts +155 -14
  110. package/src/agents/coding-cli-versions.test.ts +15 -0
  111. package/src/agents/coding-cli-versions.ts +3 -0
  112. package/src/agents/index.ts +11 -0
  113. package/src/agents/langgraph.test.ts +204 -0
  114. package/src/agents/langgraph.ts +495 -0
  115. package/src/agents/marketplace.ts +85 -0
  116. package/src/agents/native-config.test.ts +179 -0
  117. package/src/agents/native-config.ts +267 -0
  118. package/src/agents/openai-compat.test.ts +1 -0
  119. package/src/agents/openclaw.test.ts +31 -0
  120. package/src/agents/openclaw.ts +171 -0
  121. package/src/agents/plugin-config.test.ts +1 -0
  122. package/src/agents/sdk-streams.test.ts +79 -0
  123. package/src/agents/sdk-streams.ts +55 -10
  124. package/src/agents/skills.test.ts +1 -0
  125. package/src/agents/streaming.test.ts +3 -9
  126. package/src/agents/types.ts +68 -5
  127. package/src/agents/ui-message-stream.test.ts +3 -0
  128. package/src/cli.ts +411 -108
  129. package/src/context/context.test.ts +51 -12
  130. package/src/context/context.ts +161 -29
  131. package/src/context/session.test.ts +1 -0
  132. package/src/context/session.ts +114 -6
  133. package/src/context/types.ts +30 -12
  134. package/src/define.test.ts +13 -8
  135. package/src/define.ts +25 -4
  136. package/src/expect/index.ts +53 -23
  137. package/src/i18n/en.ts +64 -2
  138. package/src/i18n/zh-CN.ts +65 -3
  139. package/src/o11y/cost.test.ts +1 -0
  140. package/src/o11y/execution-tree.test.ts +1 -20
  141. package/src/o11y/otlp/mappers/claude-code.test.ts +1 -0
  142. package/src/o11y/otlp/parse.test.ts +1 -0
  143. package/src/o11y/otlp/turn-otel.test.ts +1 -0
  144. package/src/o11y/parsers/bub.test.ts +1 -0
  145. package/src/o11y/parsers/claude-code.test.ts +1 -34
  146. package/src/o11y/parsers/openclaw.test.ts +154 -0
  147. package/src/o11y/parsers/openclaw.ts +310 -0
  148. package/src/o11y/prices.json +746 -311
  149. package/src/o11y/tool-names.test.ts +1 -0
  150. package/src/o11y/types.ts +16 -2
  151. package/src/report/aggregate.ts +34 -5
  152. package/src/report/built-in-user-parity.test.tsx +110 -153
  153. package/src/report/built-ins/experiment-comparison.tsx +173 -13
  154. package/src/report/built-ins/index.ts +6 -1
  155. package/src/report/components.tsx +9 -3
  156. package/src/report/compute.ts +70 -40
  157. package/src/report/dual-render.test.tsx +194 -67
  158. package/src/report/flag.ts +30 -2
  159. package/src/report/format.ts +35 -11
  160. package/src/report/index.ts +22 -4
  161. package/src/report/locale.ts +25 -5
  162. package/src/report/metrics.ts +67 -14
  163. package/src/report/primitives.tsx +6 -0
  164. package/src/report/react/AttemptList.tsx +6 -31
  165. package/src/report/react/EvalList.tsx +0 -0
  166. package/src/report/react/ExperimentComparison.tsx +68 -0
  167. package/src/report/react/ExperimentList.tsx +15 -9
  168. package/src/report/react/MetricScatter.tsx +12 -14
  169. package/src/report/react/chart-math.test.ts +85 -0
  170. package/src/report/react/chart-math.ts +101 -22
  171. package/src/report/react/enhance.js +33 -1
  172. package/src/report/react/fixtures.ts +24 -17
  173. package/src/report/react/render.test.tsx +9 -64
  174. package/src/report/react/styles.css +73 -2
  175. package/src/report/report.test.ts +306 -98
  176. package/src/report/report.ts +6 -2
  177. package/src/report/text/faces.ts +47 -43
  178. package/src/report/text/table.ts +42 -5
  179. package/src/report/types.ts +41 -21
  180. package/src/results/annotated-source.test.ts +62 -9
  181. package/src/results/annotated-source.ts +64 -6
  182. package/src/results/attempt-evidence.test.ts +9 -7
  183. package/src/results/attempt-evidence.ts +15 -8
  184. package/src/results/attempt-source.ts +6 -3
  185. package/src/results/copy.ts +145 -55
  186. package/src/results/host-equivalence.test.ts +8 -6
  187. package/src/results/index.ts +2 -0
  188. package/src/results/locator.test.ts +1 -22
  189. package/src/results/open.ts +7 -1
  190. package/src/results/publish.ts +149 -0
  191. package/src/results/results.test.ts +85 -51
  192. package/src/results/truncate.ts +90 -0
  193. package/src/results/types.ts +7 -0
  194. package/src/results/writer.ts +31 -13
  195. package/src/runner/attempt.test.ts +138 -7
  196. package/src/runner/attempt.ts +603 -104
  197. package/src/runner/discover.test.ts +47 -0
  198. package/src/runner/discover.ts +36 -2
  199. package/src/runner/eval-source.test.ts +1 -27
  200. package/src/runner/feedback/agent.test.ts +504 -0
  201. package/src/runner/feedback/agent.ts +409 -0
  202. package/src/runner/feedback/ci.test.ts +562 -0
  203. package/src/runner/feedback/ci.ts +401 -0
  204. package/src/runner/feedback/coordinator.test.ts +317 -0
  205. package/src/runner/feedback/coordinator.ts +397 -0
  206. package/src/runner/feedback/failure.ts +40 -0
  207. package/src/runner/feedback/human.test.ts +616 -0
  208. package/src/runner/feedback/human.ts +535 -0
  209. package/src/runner/feedback/index.ts +66 -0
  210. package/src/runner/feedback/io.ts +78 -0
  211. package/src/runner/feedback/profile.test.ts +50 -0
  212. package/src/runner/feedback/profile.ts +58 -0
  213. package/src/runner/feedback/reducer.test.ts +395 -0
  214. package/src/runner/feedback/reducer.ts +260 -0
  215. package/src/runner/feedback/renderer.ts +82 -0
  216. package/src/runner/feedback/sink.ts +203 -0
  217. package/src/runner/feedback/testing.ts +106 -0
  218. package/src/runner/ledger.test.ts +230 -0
  219. package/src/runner/ledger.ts +329 -0
  220. package/src/runner/report.test.ts +128 -3
  221. package/src/runner/report.ts +33 -9
  222. package/src/runner/reporters/artifacts.ts +8 -2
  223. package/src/runner/reporters/braintrust.test.ts +8 -7
  224. package/src/runner/reporters/braintrust.ts +9 -2
  225. package/src/runner/reporters/index.ts +2 -2
  226. package/src/runner/reporters/json.test.ts +162 -0
  227. package/src/runner/reporters/json.ts +35 -8
  228. package/src/runner/reporters/shared.ts +1 -5
  229. package/src/runner/run.test.ts +760 -3
  230. package/src/runner/run.ts +242 -36
  231. package/src/runner/sandbox-prep.ts +3 -42
  232. package/src/runner/timing.ts +158 -0
  233. package/src/runner/types.ts +518 -22
  234. package/src/sandbox/checkpoint.test.ts +55 -0
  235. package/src/sandbox/checkpoint.ts +29 -8
  236. package/src/sandbox/cli-commands.ts +407 -0
  237. package/src/sandbox/docker.ts +115 -16
  238. package/src/sandbox/e2b-agent-template.test.ts +56 -0
  239. package/src/sandbox/e2b-agent-template.ts +94 -0
  240. package/src/sandbox/e2b.ts +74 -9
  241. package/src/sandbox/errors.ts +111 -4
  242. package/src/sandbox/index.ts +2 -0
  243. package/src/sandbox/io-retry.test.ts +58 -0
  244. package/src/sandbox/io-retry.ts +45 -0
  245. package/src/sandbox/keep-registry.test.ts +86 -0
  246. package/src/sandbox/keep-registry.ts +142 -0
  247. package/src/sandbox/keep.ts +178 -0
  248. package/src/sandbox/paths.test.ts +1 -0
  249. package/src/sandbox/paths.ts +19 -8
  250. package/src/sandbox/registry.ts +20 -3
  251. package/src/sandbox/resolve.ts +76 -11
  252. package/src/sandbox/retry.test.ts +70 -0
  253. package/src/sandbox/retry.ts +46 -4
  254. package/src/sandbox/types.ts +44 -6
  255. package/src/sandbox/vercel.ts +43 -20
  256. package/src/scoring/collector.ts +60 -17
  257. package/src/scoring/coverage.ts +95 -0
  258. package/src/scoring/diff.ts +81 -0
  259. package/src/scoring/display.test.ts +121 -0
  260. package/src/scoring/display.ts +133 -0
  261. package/src/scoring/evidence.test.ts +189 -0
  262. package/src/scoring/judge.test.ts +142 -0
  263. package/src/scoring/judge.ts +15 -18
  264. package/src/scoring/scoped.ts +217 -50
  265. package/src/scoring/types.ts +117 -20
  266. package/src/scoring/verdict.ts +16 -4
  267. package/src/shared/aggregate.ts +3 -2
  268. package/src/shared/types.ts +31 -0
  269. package/src/show/compose.ts +2 -2
  270. package/src/show/index.ts +21 -1
  271. package/src/show/render.ts +619 -104
  272. package/src/show/show.test.ts +235 -19
  273. package/src/tty-line.ts +8 -26
  274. package/src/util.test.ts +1 -0
  275. package/src/util.ts +41 -0
  276. package/src/view/app/components/AttemptModal.tsx +153 -2
  277. package/src/view/app/components/CodeView.tsx +32 -11
  278. package/src/view/app/components/CopyControls.tsx +2 -2
  279. package/src/view/app/i18n.ts +6 -0
  280. package/src/view/app/lib/attempt-route.test.ts +1 -0
  281. package/src/view/app/lib/verdict.ts +7 -9
  282. package/src/view/artifact-serving.test.ts +2 -1
  283. package/src/view/client-dist/app.css +1 -1
  284. package/src/view/client-dist/app.js +17 -17
  285. package/src/view/data.test.ts +1 -0
  286. package/src/view/data.ts +11 -1
  287. package/src/view/index.ts +11 -0
  288. package/src/view/server.ts +2 -0
  289. package/src/view/styles.css +3 -0
  290. package/src/view/view-report.test.ts +6 -5
  291. package/src/runner/reporters/console.ts +0 -70
  292. package/src/runner/reporters/live.test.ts +0 -56
  293. package/src/runner/reporters/live.ts +0 -247
  294. package/src/runner/reporters/quiet.test.ts +0 -66
  295. package/src/runner/reporters/quiet.ts +0 -49
  296. package/src/runner/reporters/table.ts +0 -277
@@ -1,3 +1,4 @@
1
+ // cases: docs/engineering/unit-tests/reports/cases.md
1
2
  // view 数据层(data.ts)的单测:loader 收编到 openResults、统计整体住进报告槽之后,
2
3
  // 守护三件事——skipped 三种原因如实进 viewData(producer 感知的 npx 提示)、报告槽是
3
4
  // 现刻水位口径(裸跑经 selectCurrentResults 跨快照合成每 experiment × eval 的最新判定,
package/src/view/data.ts CHANGED
@@ -42,6 +42,12 @@ export interface ViewScan {
42
42
  * 界面语言对应的那块 HTML 摆进报告槽位置,不解析。
43
43
  */
44
44
  reportHtml: ReportSlotHtml;
45
+ /**
46
+ * --out 的数据等级(见 docs/feature/reports/view.md「静态导出」):全部选中快照带
47
+ * publish:{redaction:"applied"} 才是 "applied",否则 "sensitive"(含本地事实根与
48
+ * redaction:"none"——上游声明过原文发布也不豁免导出时的确认)。
49
+ */
50
+ publishState: "applied" | "sensitive";
45
51
  }
46
52
 
47
53
  /** view 宿主输入的组合语义(与 show 对齐,docs/feature/reports/architecture.md「Selection 是计算入口」)。 */
@@ -261,7 +267,11 @@ export async function loadViewScan(input?: string, opts: ViewScanOptions = {}):
261
267
  snapshots,
262
268
  skippedRuns: results.skipped.map(toSkippedNotice),
263
269
  };
264
- return { viewData, artifactDirs, attemptsByBase, reportHtml };
270
+ const publishState =
271
+ selection.snapshots.length > 0 && selection.snapshots.every((snap) => snap.publish?.redaction === "applied")
272
+ ? ("applied" as const)
273
+ : ("sensitive" as const);
274
+ return { viewData, artifactDirs, attemptsByBase, reportHtml, publishState };
265
275
  }
266
276
 
267
277
  /**
package/src/view/index.ts CHANGED
@@ -89,6 +89,17 @@ export async function buildView(opts: ViewOptions = {}): Promise<string> {
89
89
  );
90
90
  }
91
91
  const scan = await loadViewScan(opts.input, opts.scan);
92
+ // --out 按数据等级防呆(见 docs/feature/reports/view.md「静态导出」):目标结果根的全部快照
93
+ // 带 publish:{redaction:"applied"}(copySnapshots 补记)时直接导出;redaction:"none"、
94
+ // 无标记结果或本地事实根,都必须显式传 --allow-sensitive-artifacts——静态站原样携带证据文件,
95
+ // 上游声明过原文发布也不豁免这里的确认。
96
+ if (!opts.allowSensitiveArtifacts && scan.publishState !== "applied") {
97
+ throw new ViewInputError(
98
+ `--out refuses to export unsanitized results: not every selected snapshot carries publish: { redaction: "applied" }. ` +
99
+ `Produce a publish root first with copySnapshots({ redact }) and export that (niceeval view --run <publish-root> --out <site>), ` +
100
+ `or pass --allow-sensitive-artifacts to explicitly export raw evidence (prompts, tool args, full outputs, sources).`,
101
+ );
102
+ }
92
103
  await mkdir(out, { recursive: true });
93
104
  await writeFile(join(out, "index.html"), await renderHtml(scan), "utf-8");
94
105
  await copyFetchedArtifacts(scan, join(out, "artifact"));
@@ -12,6 +12,8 @@ export interface ViewOptions {
12
12
  input?: string;
13
13
  out?: string;
14
14
  port?: number;
15
+ /** `--out` 对非发布根(无 publish:applied 标记)导出时的显式确认;静态站原样携带证据文件。 */
16
+ allowSensitiveArtifacts?: boolean;
15
17
  /** 报告槽的组合语义(位置前缀 / --experiment / --report),透传给 loadViewScan。 */
16
18
  scan?: ViewScanOptions;
17
19
  }
@@ -849,6 +849,7 @@ a.brand:hover { opacity: 0.75; }
849
849
  .gstat.good { color: var(--good); }
850
850
  .gstat.bad { color: var(--bad); }
851
851
  .gstat.warn { color: var(--warn); }
852
+ .gstat.na { color: var(--muted, #8a8f98); }
852
853
  .gsend { width: 13px; height: 13px; flex-shrink: 0; color: var(--blue); }
853
854
 
854
855
  .ctext { white-space: pre; color: var(--text); padding-left: 8px; }
@@ -866,6 +867,8 @@ a.brand:hover { opacity: 0.75; }
866
867
  .abadge.good { color: var(--good); border-color: color-mix(in oklch, var(--good), var(--line) 55%); background: color-mix(in oklch, var(--good), transparent 90%); }
867
868
  .abadge.bad { color: var(--bad); border-color: color-mix(in oklch, var(--bad), var(--line) 55%); background: color-mix(in oklch, var(--bad), transparent 90%); }
868
869
  .abadge.warn { color: var(--warn); border-color: color-mix(in oklch, var(--warn), var(--line) 55%); background: color-mix(in oklch, var(--warn), transparent 90%); }
870
+ /* unavailable:独立第三态(非红非绿),证据评不了 ≠ 失败 */
871
+ .abadge.na { color: var(--muted, #8a8f98); border-color: var(--line); background: color-mix(in oklch, var(--muted, #8a8f98), transparent 92%); }
869
872
  .abadge-th { opacity: 0.55; font-weight: 400; }
870
873
 
871
874
  /* 可点击行的显式提示:send 行一个 "reply" 药丸,断言行一个右侧展开箭头;hover 时变亮。 */
@@ -1,3 +1,4 @@
1
+ // cases: docs/engineering/unit-tests/reports/cases.md
1
2
  // niceeval view 的报告槽与宿主组合语义(docs/feature/reports/architecture.md「Selection 是计算入口」
2
3
  // 与裁决记录 6;公开行为准绳 docs-site/zh/guides/viewing-results.mdx / custom-reports.mdx)。
3
4
  // 覆盖:
@@ -84,7 +85,7 @@ async function seedRoot(): Promise<string> {
84
85
  await writeSnapshot(root, "compare_bub", "2026-07-08T10-00-00-000Z", { experimentId: "compare/bub", agent: "bub", startedAt: "2026-07-08T10:00:00.000Z" }, [
85
86
  res("weather/brooklyn", "passed"),
86
87
  res("fixtures/button", "failed", {
87
- assertions: [{ name: 'fileChanged("Button.tsx")', severity: "gate", score: 0, passed: false }],
88
+ assertions: [{ name: 'fileChanged("Button.tsx")', severity: "gate", score: 0, outcome: "failed" as const }],
88
89
  }),
89
90
  ]);
90
91
  await writeSnapshot(root, "compare_codex", "2026-07-09T10-00-00-000Z", { experimentId: "compare/codex", agent: "codex", startedAt: "2026-07-09T10:00:00.000Z" }, [
@@ -185,8 +186,8 @@ describe("loadViewScan · 默认报告槽(裸跑)", () => {
185
186
  it("报告槽双语渲染:同一棵树按 locale 渲染两遍,chrome 文案分语言、数据不分语言", async () => {
186
187
  const root = await seedRoot();
187
188
  const { reportHtml } = await loadViewScan(root);
188
- expect(reportHtml.en).toContain("Pass rate"); // ExperimentList 主行(en)
189
- expect(reportHtml["zh-CN"]).toContain("成功率"); // ExperimentList 主行(zh-CN,passRate 的 zh label)
189
+ expect(reportHtml.en).toContain("End-to-end pass rate"); // ExperimentList 主行(en)
190
+ expect(reportHtml["zh-CN"]).toContain("端到端成功率"); // ExperimentList 主行(zh-CN)
190
191
  for (const html of [reportHtml.en, reportHtml["zh-CN"]]) {
191
192
  expect(html).toContain("compare/bub");
192
193
  // 失败案例深链进证据室:不透明 AttemptLocator 单段路由 `#/attempt/@<locator>`,
@@ -323,7 +324,7 @@ describe("buildView · --out 与 --report", () => {
323
324
  await writeFile(join(artifactDir, "events.json"), "[]", "utf-8");
324
325
 
325
326
  const out = join(root, "site");
326
- await buildView({ input: root, out, scan: { report: { path: EXAM_REPORT, cwd: root } } });
327
+ await buildView({ input: root, out, allowSensitiveArtifacts: true, scan: { report: { path: EXAM_REPORT, cwd: root } } });
327
328
 
328
329
  const html = await readFile(join(out, "index.html"), "utf-8");
329
330
  // 双语两个 <template> 静态块都在,壳按界面语言摆放。
@@ -343,7 +344,7 @@ describe("buildView · --out 与 --report", () => {
343
344
  it("默认导出(无 --report):报告槽填充 ExperimentComparison,双语块与增强 runtime 恒内联", async () => {
344
345
  const root = await seedRoot();
345
346
  const out = join(root, "site");
346
- await buildView({ input: root, out });
347
+ await buildView({ input: root, out, allowSensitiveArtifacts: true });
347
348
  const html = await readFile(join(out, "index.html"), "utf-8");
348
349
  expect(html).toContain('<template id="niceeval-report-en">');
349
350
  expect(html).toContain('<template id="niceeval-report-zh-CN">');
@@ -1,70 +0,0 @@
1
- // 控制台报告器:流式逐行输出,失败断言内联展开,末尾出效率三件套(时间 / token / $)。
2
-
3
- import type { EvalResult, Reporter, RunSummary } from "../../types.ts";
4
- import { t } from "../../i18n/index.ts";
5
- import { formatCost, formatDuration, formatTokens, renderRunReport } from "./table.ts";
6
- import { verdictSymbol } from "./shared.ts";
7
-
8
- export function Console(): Reporter {
9
- return {
10
- onRunStart(evals, _agent, shape) {
11
- // compare(多 agent / 多 model)或 runs>1 时,实际 attempt 数 > eval 数。头部如实报清,
12
- // 否则「本次运行 5 个 eval」会和末尾「5 passed, 5 failed」(按 attempt 计 10)对不上。
13
- const n = shape?.evals ?? evals.length;
14
- const extra =
15
- shape && shape.totalRuns > n
16
- ? t("report.runStartExtra", { configs: shape.configs, totalRuns: shape.totalRuns })
17
- : "";
18
- process.stdout.write(t("report.runStart", { count: n, extra, concurrency: shape?.maxConcurrency ?? "?" }));
19
- },
20
- onEvalComplete(result: EvalResult) {
21
- const sym = verdictSymbol(result.verdict);
22
- const tok = (result.usage?.inputTokens ?? 0) + (result.usage?.outputTokens ?? 0);
23
- // requests > 0 但 tokens = 0 → agent 跑了但不上报用量(如 bub);显示 — 而非误导性的 0
24
- const tokStr = tok > 0 ? `${formatTokens(tok)} tok` : (result.usage?.requests ?? 0) > 0 ? `— tok` : `0 tok`;
25
- const cost = result.estimatedCostUSD !== undefined ? ` ${formatCost(result.estimatedCostUSD)}` : "";
26
- const who = result.model ? `${result.agent}/${result.model}` : result.agent;
27
- const meta = `(${formatDuration(result.durationMs)} ${tokStr}${cost})`;
28
- const label = result.verdict === "passed" ? "" : ` ${formatVerdict(result.verdict)}`;
29
- process.stdout.write(` ${sym} ${result.id}${label} [${who}] ${meta}\n`);
30
-
31
- if (result.skipReason) {
32
- process.stdout.write(` ○ ${t("report.skipped")}: ${result.skipReason}\n`);
33
- }
34
- if (result.error) {
35
- process.stdout.write(` ! ${t("report.error")}: ${truncate(result.error, 400)}\n`);
36
- }
37
- let lastGroup: string | undefined;
38
- for (const a of result.assertions) {
39
- if (a.passed) continue;
40
- if (a.group !== undefined && a.group !== lastGroup) {
41
- process.stdout.write(` ▸ ${a.group}\n`);
42
- }
43
- lastGroup = a.group;
44
- const sev = a.severity === "gate" ? t("report.gate") : t("report.soft");
45
- const thr = a.threshold !== undefined
46
- ? t("report.assertionThreshold", { score: a.score.toFixed(2), threshold: a.threshold })
47
- : "";
48
- const indent = a.group !== undefined ? " " : " ";
49
- process.stdout.write(`${indent}- ${sev}: ${a.name}${thr}${a.detail ? ` — ${truncate(a.detail, 300)}` : ""}\n`);
50
- }
51
- },
52
- onRunComplete(summary: RunSummary) {
53
- process.stdout.write(renderRunReport(summary));
54
- },
55
- };
56
- }
57
-
58
- function formatVerdict(verdict: string): string {
59
- switch (verdict) {
60
- case "passed": return t("report.passed");
61
- case "failed": return t("report.failed");
62
- case "errored": return t("report.errored");
63
- case "skipped": return t("report.skipped");
64
- default: return verdict;
65
- }
66
- }
67
-
68
- function truncate(s: string, n: number): string {
69
- return s.length > n ? s.slice(0, n) + "…" : s;
70
- }
@@ -1,56 +0,0 @@
1
- import { afterEach, describe, expect, it, vi } from "vitest";
2
- import { Live, type LiveRow } from "./live.ts";
3
-
4
- function withMockTty<T>(fn: () => T): { writes: string[]; result: T } {
5
- const writes: string[] = [];
6
- vi.spyOn(process.stderr, "write").mockImplementation((chunk: unknown) => {
7
- writes.push(String(chunk));
8
- return true;
9
- });
10
- Object.defineProperty(process.stderr, "isTTY", { value: true, configurable: true });
11
- Object.defineProperty(process.stderr, "columns", { value: 120, configurable: true });
12
- Object.defineProperty(process.stderr, "rows", { value: 40, configurable: true });
13
- return { writes, result: fn() };
14
- }
15
-
16
- describe("Live carried rows", () => {
17
- afterEach(() => {
18
- vi.restoreAllMocks();
19
- });
20
-
21
- it("renders a carried row as already-done from the first frame, not waiting for a slot", () => {
22
- const rows: LiveRow[] = [
23
- { evalId: "memory/carried-fail", who: "codex-e2b", total: 1, carriedVerdict: "failed" },
24
- { evalId: "memory/fresh", who: "codex-e2b", total: 1 },
25
- ];
26
- const live = Live(rows, 2);
27
-
28
- const { writes } = withMockTty(() => {
29
- live.onRunStart?.([], { name: "codex", kind: "sandbox" } as never, {
30
- evals: 2,
31
- configs: 1,
32
- totalRuns: 1,
33
- maxConcurrency: 5,
34
- });
35
- });
36
- vi.spyOn(process.stdout, "write").mockImplementation(() => true);
37
- live.onRunComplete?.({
38
- startedAt: "",
39
- finishedAt: "",
40
- durationMs: 0,
41
- results: [],
42
- totalRuns: 1,
43
- } as never);
44
-
45
- const out = writes.join("");
46
- const carriedLine = out.split("\n").find((l) => l.includes("memory/carried-fail"));
47
- const freshLine = out.split("\n").find((l) => l.includes("memory/fresh"));
48
-
49
- expect(carriedLine).toBeDefined();
50
- expect(carriedLine).toContain("✗");
51
- expect(carriedLine).not.toContain("waiting");
52
-
53
- expect(freshLine).toBeDefined();
54
- expect(freshLine).toContain("waiting");
55
- });
56
- });
@@ -1,247 +0,0 @@
1
- // Live terminal reporter:在 TTY 终端里渲染实时状态表,每个 (eval, who) 对占一行。
2
- // spinner 每 80ms 刷新;attempt 完成后行内显示 ✓/✗/~ 符号。
3
- // onRunComplete 时清除状态表,打印和网页榜单同口径的表格报告。
4
-
5
- import type { Reporter, ReporterEvent, RunShape, RunSummary } from "../../types.ts";
6
- import { t } from "../../i18n/index.ts";
7
- import { renderRunReport } from "./table.ts";
8
- import { verdictSymbol, WAITING_SYM } from "./shared.ts";
9
- import { runWho } from "../types.ts";
10
- import { onBeforeExternalTerminalWrite } from "../../tty-line.ts";
11
-
12
- const SPINNER = ["⠋", "⠙", "⠹", "⠸", "⠼", "⠴", "⠦", "⠧", "⠇", "⠏"];
13
-
14
- // 进度日志里的基础设施噪声:OTLP 端口、remote-agent 启动提示、trace span 计数。
15
- // 这些在行尾显示毫无意义,直接丢弃。
16
- const NOISE_PATTERNS = [
17
- /^OTLP /,
18
- /^使用 remote agent/,
19
- /^using remote agent/,
20
- /^驱动 agent/,
21
- /^driving agent/,
22
- /^trace:\d/,
23
- /^agent tracing/,
24
- /^agent setup/,
25
- ];
26
-
27
- function isNoise(msg: string): boolean {
28
- return NOISE_PATTERNS.some((p) => p.test(msg));
29
- }
30
-
31
- export interface LiveRow {
32
- evalId: string;
33
- who: string;
34
- /** 该 (evalId, who) 对的总 attempt 数(含 earlyExit 估算值)。 */
35
- total: number;
36
- /**
37
- * 设置时表示这一行从第一帧起就是携入(carry)的结果,不会真的调度 attempt——直接按这个
38
- * verdict 渲染成已完成,不经过 waiting → spinner 的过程。没有它,carry 掉的行会卡在
39
- * "waiting for a slot"直到进程退出:它永远等不到 eval:start,因为 run.ts 压根不会为
40
- * 它派发 attempt。
41
- */
42
- carriedVerdict?: string;
43
- }
44
-
45
- export interface LiveReporter extends Reporter {
46
- /** 被 RunOptions.onProgress 调用,更新行尾的 lastMsg。 */
47
- progress(evalId: string, who: string, msg: string): void;
48
- }
49
-
50
- interface RowState extends LiveRow {
51
- completed: number;
52
- lastMsg: string;
53
- dominantVerdict: string | undefined;
54
- /** true 一旦拿到并发名额、attempt effect 真正开始跑;之前是排队等待,不该转圈误导。 */
55
- started: boolean;
56
- }
57
-
58
- export function Live(rows: LiveRow[], totalAttempts: number): LiveReporter {
59
- const stateMap = new Map<string, RowState>();
60
- const keyOrder: string[] = [];
61
- let totalCompleted = 0;
62
-
63
- for (const r of rows) {
64
- const k = `${r.evalId}|${r.who}`;
65
- if (!stateMap.has(k)) {
66
- const carried = r.carriedVerdict !== undefined;
67
- stateMap.set(k, {
68
- ...r,
69
- completed: carried ? r.total : 0,
70
- lastMsg: "",
71
- dominantVerdict: r.carriedVerdict,
72
- started: carried,
73
- });
74
- keyOrder.push(k);
75
- if (carried) totalCompleted += r.total;
76
- } else {
77
- // 同一 (evalId, who) 可能在多个 agentRun 里出现(不应发生,但做防御)
78
- stateMap.get(k)!.total += r.total;
79
- }
80
- }
81
-
82
- let spinFrame = 0;
83
- let drawnLines = 0; // 上次 draw() 写了多少行,用于 \x1B[nA 回跳
84
- let intervalId: ReturnType<typeof setInterval> | undefined;
85
- let shape: RunShape | undefined;
86
- let unsubscribeExternalWrite: (() => void) | undefined;
87
-
88
- const cols = () => process.stderr.columns || 100;
89
-
90
- function renderRow(state: RowState, frame: number): string {
91
- const done = state.completed >= state.total;
92
- const sym = done
93
- ? verdictSymbol(state.dominantVerdict ?? "")
94
- : state.started
95
- ? SPINNER[frame % SPINNER.length]
96
- : WAITING_SYM;
97
-
98
- const evalCol = state.evalId.slice(0, 24).padEnd(24);
99
- const whoCol = `[${state.who}]`.slice(0, 26).padEnd(26);
100
- const cntCol = `${state.completed}/${state.total}`.padEnd(5);
101
-
102
- const prefix = ` ${sym} ${evalCol} ${whoCol} ${cntCol} `;
103
- const budget = Math.max(0, cols() - prefix.length - 1);
104
- const msg = done ? "" : state.started ? state.lastMsg.slice(0, budget) : t("live.waiting").slice(0, budget);
105
-
106
- return `\x1B[2K${prefix}${msg}`;
107
- }
108
-
109
- function renderHeader(): string {
110
- const hdr = shape
111
- ? t("live.running", {
112
- totalRuns: shape.totalRuns,
113
- evals: shape.evals,
114
- configs: shape.configs,
115
- concurrency: shape.maxConcurrency,
116
- completed: totalCompleted,
117
- total: totalAttempts,
118
- })
119
- : t("live.runningUnknown", { completed: totalCompleted, total: totalAttempts });
120
- return `\x1B[2K${hdr}`;
121
- }
122
-
123
- // 终端放不下全部行时,光标回跳会被屏幕顶端截断,导致每帧往下追加整表。
124
- // 所以按终端高度截断:优先显示运行中的行,放不下的折叠成一行摘要。
125
- function frameLines(frame: number): string[] {
126
- const lines = [renderHeader()];
127
- const termRows = process.stderr.rows || 30;
128
- // 预留:1 行表头 + 1 行防止末尾换行触发滚动
129
- const budget = Math.max(1, termRows - 2);
130
-
131
- if (keyOrder.length <= budget) {
132
- for (const k of keyOrder) lines.push(renderRow(stateMap.get(k)!, frame));
133
- return lines;
134
- }
135
-
136
- const running: string[] = [];
137
- const waiting: string[] = [];
138
- const done: string[] = [];
139
- for (const k of keyOrder) {
140
- const s = stateMap.get(k)!;
141
- if (s.completed >= s.total) done.push(k);
142
- else if (s.started) running.push(k);
143
- else waiting.push(k);
144
- }
145
- // 选出要显示的 key(运行中 > 等待 > 已完成),但按原始顺序渲染,避免行来回跳动
146
- const shown = new Set([...running, ...waiting, ...done].slice(0, budget - 1));
147
- for (const k of keyOrder) {
148
- if (shown.has(k)) lines.push(renderRow(stateMap.get(k)!, frame));
149
- }
150
- lines.push(
151
- `\x1B[2K ${t("live.more", {
152
- hidden: keyOrder.length - shown.size,
153
- running: running.filter((k) => !shown.has(k)).length,
154
- waiting: waiting.filter((k) => !shown.has(k)).length,
155
- done: done.filter((k) => !shown.has(k)).length,
156
- })}`,
157
- );
158
- return lines;
159
- }
160
-
161
- function draw(frame: number) {
162
- if (!process.stderr.isTTY) return;
163
-
164
- const lines = frameLines(frame);
165
- let out = drawnLines > 0 ? `\x1B[${drawnLines}A` : "";
166
- out += lines.join("\n") + "\n";
167
- // 本帧比上帧短(行完成后折叠、终端拉高)时,清掉下方残留的旧行
168
- const extra = drawnLines - lines.length;
169
- if (extra > 0) {
170
- out += "\x1B[2K\n".repeat(extra) + `\x1B[${extra}A`;
171
- }
172
- process.stderr.write(out);
173
- drawnLines = lines.length;
174
- }
175
-
176
- function clearDisplay() {
177
- if (!process.stderr.isTTY || drawnLines === 0) return;
178
- // 回到起点,逐行清空
179
- process.stderr.write(`\x1B[${drawnLines}A`);
180
- for (let i = 0; i < drawnLines; i++) {
181
- process.stderr.write("\x1B[2K\n");
182
- }
183
- process.stderr.write(`\x1B[${drawnLines}A`);
184
- drawnLines = 0;
185
- }
186
-
187
- return {
188
- progress(evalId, who, msg) {
189
- if (isNoise(msg)) return;
190
- const state = stateMap.get(`${evalId}|${who}`);
191
- if (state) state.lastMsg = msg;
192
- // 不在这里 draw();由 interval 驱动,避免每条日志都刷屏
193
- },
194
-
195
- onEvent(event: ReporterEvent) {
196
- if (event.type !== "eval:start") return;
197
- const who = runWho({ agentName: event.agent.name, model: event.model, experimentId: event.experimentId });
198
- const state = stateMap.get(`${event.eval.id}|${who}`);
199
- if (state) state.started = true;
200
- },
201
-
202
- onRunStart(_evals, _agent, s) {
203
- shape = s;
204
- // sandbox teardown 失败、budget 不可执行这类独立诊断行可能绕开 progress()/onEvalComplete
205
- // 直接落地(见 tty-line.ts)。订阅后先把已画的表格清掉、drawnLines 归零,下一帧从新起点
206
- // 重画,不然回跳量和实际光标错位,越滚越多。
207
- unsubscribeExternalWrite = onBeforeExternalTerminalWrite(() => clearDisplay());
208
- // 初始渲染:让用户看到行表
209
- draw(0);
210
- intervalId = setInterval(() => {
211
- spinFrame = (spinFrame + 1) % SPINNER.length;
212
- draw(spinFrame);
213
- }, 80);
214
- },
215
-
216
- onEvalComplete(result) {
217
- const who = runWho({ agentName: result.agent, model: result.model, experimentId: result.experimentId });
218
- const state = stateMap.get(`${result.id}|${who}`);
219
- if (state) {
220
- state.completed += 1;
221
- totalCompleted += 1;
222
- const prev = state.dominantVerdict;
223
- if (!prev || (prev === "passed" && result.verdict !== "passed")) {
224
- state.dominantVerdict = result.verdict;
225
- }
226
- } else {
227
- totalCompleted += 1;
228
- }
229
- },
230
-
231
- onRunComplete(summary) {
232
- if (intervalId) {
233
- clearInterval(intervalId);
234
- intervalId = undefined;
235
- }
236
- if (unsubscribeExternalWrite) {
237
- unsubscribeExternalWrite();
238
- unsubscribeExternalWrite = undefined;
239
- }
240
- // 最后刷一帧(所有行都 done)
241
- draw(spinFrame);
242
- clearDisplay();
243
-
244
- process.stdout.write(renderRunReport(summary));
245
- },
246
- };
247
- }
@@ -1,66 +0,0 @@
1
- import { describe, expect, it } from "vitest";
2
- import { quietLine } from "./quiet.ts";
3
- import type { EvalResult } from "../../types.ts";
4
-
5
- function baseResult(overrides: Partial<EvalResult> = {}): EvalResult {
6
- return {
7
- id: "algebra/quadratic",
8
- agent: "codex",
9
- verdict: "passed",
10
- attempt: 0,
11
- durationMs: 42_000,
12
- assertions: [],
13
- ...overrides,
14
- };
15
- }
16
-
17
- describe("quietLine", () => {
18
- it("passed / skipped 静默(返回 undefined)", () => {
19
- expect(quietLine(baseResult())).toBeUndefined();
20
- expect(quietLine(baseResult({ verdict: "skipped", skipReason: "no api key" }))).toBeUndefined();
21
- });
22
-
23
- it("errored 带 eval id、[who] 与截断后的 error", () => {
24
- const line = quietLine(
25
- baseResult({
26
- verdict: "errored",
27
- model: "gpt-5",
28
- error: "sandbox create failed: e2b timeout after 3.9s",
29
- }),
30
- );
31
- expect(line).toContain("algebra/quadratic");
32
- expect(line).toContain("[codex/gpt-5]");
33
- expect(line).toContain("errored");
34
- expect(line).toContain("e2b timeout after 3.9s");
35
- });
36
-
37
- it("[who] 与进度行同源:有 experimentId 时用其 basename", () => {
38
- const line = quietLine(
39
- baseResult({ verdict: "errored", experimentId: "compare/xxx--agents-md", error: "boom" }),
40
- );
41
- expect(line).toContain("[xxx--agents-md]");
42
- });
43
-
44
- it("failed 无 error 时取首个失败断言(severity、阈值、detail)", () => {
45
- const line = quietLine(
46
- baseResult({
47
- verdict: "failed",
48
- assertions: [
49
- { name: "compiles", severity: "gate", score: 1, passed: true },
50
- { name: "closedQA", severity: "gate", score: 0.2, passed: false, threshold: 0.7, detail: "wrong city" },
51
- ],
52
- }),
53
- );
54
- expect(line).toContain("failed");
55
- expect(line).toContain("closedQA");
56
- expect(line).toContain("0.20");
57
- expect(line).toContain("wrong city");
58
- expect(line).not.toContain("compiles");
59
- });
60
-
61
- it("超长 error 截断到 200 字符并加省略号", () => {
62
- const line = quietLine(baseResult({ verdict: "errored", error: "x".repeat(500) }))!;
63
- expect(line).toContain("…");
64
- expect(line.length).toBeLessThan(300);
65
- });
66
- });
@@ -1,49 +0,0 @@
1
- // Quiet 报告器:--quiet 下的最小结果流。进度流照旧(attempt 无 onProgress 时直写 stderr),
2
- // 这里只补「坏结果」:verdict 为 errored / failed 的结果各写一行 stderr,passed / skipped
3
- // 静默 —— 保证 --quiet 下起沙箱失败这类执行错不会全程无声、只能事后读 summary.json。
4
- // 结果流仍走统一的 Reporter 管线,不在 attempt 里散写。
5
-
6
- import type { EvalResult, Reporter } from "../../types.ts";
7
- import { t } from "../../i18n/index.ts";
8
- import { runWho } from "../types.ts";
9
- import { verdictSymbol } from "./shared.ts";
10
-
11
- /** error / 断言 detail 的截断上限;比 Console 的 400 更紧,--quiet 只要能定位问题。 */
12
- const DETAIL_MAX = 200;
13
-
14
- /** 纯函数:一条结果 → 该写 stderr 的行;passed / skipped 返回 undefined(静默)。 */
15
- export function quietLine(result: EvalResult): string | undefined {
16
- if (result.verdict !== "errored" && result.verdict !== "failed") return undefined;
17
- // [who] 与 attempt 进度行同源(runWho),两条流才能对上同一个运行配置。
18
- const who = runWho({ agentName: result.agent, model: result.model, experimentId: result.experimentId });
19
- const verdict = result.verdict === "errored" ? t("report.errored") : t("report.failed");
20
- const detail =
21
- result.error !== undefined
22
- ? `${t("report.error")}: ${truncate(result.error, DETAIL_MAX)}`
23
- : firstFailedAssertion(result);
24
- return ` ${verdictSymbol(result.verdict)} ${result.id} ${verdict} [${who}]${detail ? ` ${detail}` : ""}\n`;
25
- }
26
-
27
- function firstFailedAssertion(result: EvalResult): string | undefined {
28
- const a = result.assertions.find((x) => !x.passed);
29
- if (!a) return undefined;
30
- const sev = a.severity === "gate" ? t("report.gate") : t("report.soft");
31
- const thr =
32
- a.threshold !== undefined
33
- ? t("report.assertionThreshold", { score: a.score.toFixed(2), threshold: a.threshold })
34
- : "";
35
- return `${sev}: ${a.name}${thr}${a.detail ? ` — ${truncate(a.detail, DETAIL_MAX)}` : ""}`;
36
- }
37
-
38
- function truncate(s: string, n: number): string {
39
- return s.length > n ? s.slice(0, n) + "…" : s;
40
- }
41
-
42
- export function Quiet(): Reporter {
43
- return {
44
- onEvalComplete(result: EvalResult) {
45
- const line = quietLine(result);
46
- if (line !== undefined) process.stderr.write(line);
47
- },
48
- };
49
- }