niceeval 0.10.0 → 0.10.3-canary.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (71) hide show
  1. package/INDEX.md +4 -1
  2. package/README.md +1 -1
  3. package/README.zh.md +1 -1
  4. package/dist/report/components/attempt-detail/AttemptAssertions.js +2 -1
  5. package/dist/report/components/attempt-detail/AttemptConversation.d.ts +5 -1
  6. package/dist/report/components/attempt-detail/AttemptConversation.js +42 -2
  7. package/dist/report/components/attempt-detail/AttemptSource.js +84 -13
  8. package/dist/report/components/attempt-detail/compute.js +44 -1
  9. package/dist/report/components/attempt-detail/faces.js +4 -0
  10. package/dist/report/components/attempt-detail/index.d.ts +2 -2
  11. package/dist/report/components/attempt-detail/index.js +24 -4
  12. package/dist/report/components/entity-lists/ExperimentList.d.ts +1 -2
  13. package/dist/report/components/entity-lists/ExperimentList.js +6 -5
  14. package/dist/report/components/entity-lists/faces.d.ts +1 -1
  15. package/dist/report/components/entity-lists/faces.js +10 -9
  16. package/dist/report/components/entity-lists/index.d.ts +5 -7
  17. package/dist/report/components/entity-lists/index.js +7 -3
  18. package/dist/report/components/metric-views/MetricScatter.js +2 -38
  19. package/dist/report/index.d.ts +1 -1
  20. package/dist/report/model/format.d.ts +5 -5
  21. package/dist/report/model/format.js +34 -8
  22. package/dist/report/model/types.d.ts +16 -2
  23. package/dist/report/react/index.d.ts +1 -1
  24. package/dist/runner/types.d.ts +2 -1
  25. package/dist/sandbox/e2b.d.ts +0 -1
  26. package/dist/scoring/display.d.ts +7 -1
  27. package/dist/scoring/display.js +22 -2
  28. package/docs-site/zh/examples/integrations/ai-sdk-v7.mdx +5 -5
  29. package/docs-site/zh/explanation/runner.mdx +10 -2
  30. package/docs-site/zh/index.mdx +1 -1
  31. package/docs-site/zh/reference/cli.mdx +2 -2
  32. package/docs-site/zh/reference/expect.mdx +21 -1
  33. package/docs-site/zh/reference/report-components.mdx +3 -1
  34. package/docs-site/zh/troubleshooting/debugging.mdx +8 -8
  35. package/docs-site/zh/troubleshooting/recover-after-kill.mdx +62 -0
  36. package/docs-site/zh/tutorials/agent-onboarding.mdx +14 -2
  37. package/docs-site/zh/tutorials/experiments.mdx +1 -2
  38. package/docs-site/zh/tutorials/quickstart.mdx +2 -2
  39. package/docs-site/zh/tutorials/sandbox-providers.mdx +27 -0
  40. package/docs-site/zh/tutorials/viewing-results.mdx +13 -10
  41. package/docs-site/zh/tutorials/write-experiment.mdx +1 -3
  42. package/package.json +3 -4
  43. package/src/cli.ts +10 -2
  44. package/src/expect/index.test.ts +39 -0
  45. package/src/expect/index.ts +33 -0
  46. package/src/o11y/prices.json +176 -99
  47. package/src/report/assets/styles.css +822 -0
  48. package/src/report/components/attempt-detail/AttemptAssertions.tsx +4 -3
  49. package/src/report/components/attempt-detail/AttemptConversation.tsx +55 -7
  50. package/src/report/components/attempt-detail/AttemptSource.tsx +187 -42
  51. package/src/report/components/attempt-detail/attempt-components.test.tsx +82 -4
  52. package/src/report/components/attempt-detail/compute.ts +50 -1
  53. package/src/report/components/attempt-detail/faces.ts +3 -0
  54. package/src/report/components/attempt-detail/index.tsx +33 -16
  55. package/src/report/components/attempt-detail/validate.test.ts +19 -2
  56. package/src/report/components/entity-lists/ExperimentList.tsx +13 -10
  57. package/src/report/components/entity-lists/faces.ts +10 -9
  58. package/src/report/components/entity-lists/index.tsx +7 -10
  59. package/src/report/components/metric-views/MetricScatter.tsx +2 -38
  60. package/src/report/components/render.test.tsx +19 -9
  61. package/src/report/index.ts +2 -0
  62. package/src/report/model/format.ts +33 -8
  63. package/src/report/model/types.ts +18 -2
  64. package/src/report/react/index.tsx +2 -0
  65. package/src/report/runtime/dual-render.test.tsx +43 -10
  66. package/src/runner/types.ts +2 -1
  67. package/src/sandbox/e2b-reconcile.test.ts +126 -0
  68. package/src/sandbox/e2b.ts +38 -2
  69. package/src/scoring/display.test.ts +26 -0
  70. package/src/scoring/display.ts +24 -2
  71. package/src/view/view-report.test.ts +3 -1
@@ -1,5 +1,5 @@
1
1
  import { jsx as _jsx, jsxs as _jsxs } from "react/jsx-runtime";
2
- import { formatMetricValue } from "../../model/format.js";
2
+ import { formatMetricValue, shortestUniqueLabels } from "../../model/format.js";
3
3
  import { DEFAULT_REPORT_LOCALE, countText, localeText, resolveLocalizedText, resolveMetricLabel } from "../../model/locale.js";
4
4
  import { niceTicks, placePointLabels } from "./chart-math.js";
5
5
  import { colorIndicesForKeys } from "../../assets/colors.js";
@@ -9,42 +9,6 @@ const HEIGHT = 400;
9
9
  const MARGIN = { top: 28, right: 32, bottom: 48, left: 64 };
10
10
  const PLOT_W = WIDTH - MARGIN.left - MARGIN.right;
11
11
  const PLOT_H = HEIGHT - MARGIN.top - MARGIN.bottom;
12
- /**
13
- * 点的直接标签:末段在当前 data 中唯一才缩成末段;重名时逐步加长为能区分它们的最短
14
- * 路径后缀(完整 id 与两轴值仍进 <title>)。
15
- */
16
- function pointLabels(keys) {
17
- const segsOf = (key) => key.split("/").filter(Boolean);
18
- const depth = new Map(keys.map((key) => [key, 1]));
19
- for (;;) {
20
- const byLabel = new Map();
21
- for (const key of keys) {
22
- const segs = segsOf(key);
23
- const label = segs.slice(-Math.min(depth.get(key), segs.length)).join("/") || key;
24
- byLabel.set(label, [...(byLabel.get(label) ?? []), key]);
25
- }
26
- let grew = false;
27
- for (const group of byLabel.values()) {
28
- if (group.length < 2)
29
- continue;
30
- for (const key of group) {
31
- const segs = segsOf(key);
32
- if (depth.get(key) < segs.length) {
33
- depth.set(key, depth.get(key) + 1);
34
- grew = true;
35
- }
36
- }
37
- }
38
- if (!grew) {
39
- const out = new Map();
40
- for (const key of keys) {
41
- const segs = segsOf(key);
42
- out.set(key, segs.slice(-Math.min(depth.get(key), segs.length)).join("/") || key);
43
- }
44
- return out;
45
- }
46
- }
47
- }
48
12
  /**
49
13
  * 一根轴:niceTicks 撑出整齐的值域,值 → 像素做线性映射。轴方向跟随指标的 `better`:
50
14
  * `better: "lower"` 的轴反向渲染(值大的一端在左 / 下),「更好」因此恒指向右与上;
@@ -78,7 +42,7 @@ export function MetricScatter({ data, connect = false, pointHref, className, loc
78
42
  const xScale = axisScale(drawableRows.map((r) => r.x.value), MARGIN.left, MARGIN.left + PLOT_W, data.x.better === "lower");
79
43
  // y 像素轴向下增长:正向 = 高值在上 → 映射到 [bottom, top];lower 反向 = 高值在下。
80
44
  const yScale = axisScale(drawableRows.map((r) => r.y.value), MARGIN.top + PLOT_H, MARGIN.top, data.y.better === "lower");
81
- const labelByKey = pointLabels(drawableRows.map((r) => r.key));
45
+ const labelByKey = shortestUniqueLabels(drawableRows.map((r) => r.key));
82
46
  // 方向提示恒为「越靠右上越好」,仅当两轴都声明 better 时显示——任一轴未声明,
83
47
  // 组件不猜「更好」朝哪边,整图无提示。
84
48
  const showBetterHint = data.x.better !== undefined && data.y.better !== undefined;
@@ -30,5 +30,5 @@ export type { DeltaTableOptions, MetricLineOptions, MetricMatrixOptions, MetricS
30
30
  export { copyFixPromptData, heroData, scopeWarningsData, traceWaterfallData, } from "./components/site-components/compute.ts";
31
31
  export { attemptAssertionsData, attemptConversationData, attemptDiagnosticsData, attemptDiffData, attemptErrorData, attemptFixPromptData, attemptSourceData, attemptSummaryData, attemptTimelineData, attemptTraceData, attemptUsageData, } from "./components/attempt-detail/compute.ts";
32
32
  export type { Aggregator, AttemptListItem, AttemptLocator, BuiltInDimension, CopyFixPromptData, CustomDimension, DeltaData, DeltaPair, DimensionInput, DimensionOptions, DimensionRef, EvalListItem, ExperimentListEvalRow, ExperimentListItem, FlagPairs, HeroData, LineData, MatrixData, Metric, MetricAggregate, MetricCell, MetricColumn, NumericAxis, NumericAxisOptions, NumericRunConfigAxisOptions, ReportInput, RunConfigKey, ScatterData, ScopeSummaryData, ScopeWarning, ScoreboardData, SeriesInput, TableData, TraceSpanSummary, TraceWaterfallRow, VerdictTally, } from "./model/types.ts";
33
- export type { AttemptAssertionsData, AttemptConversationData, AttemptConversationReply, AttemptConversationRound, AttemptDiagnosticsData, AttemptDiffData, AttemptDiffFileEntry, AttemptErrorData, AttemptFixPromptData, AttemptSourceData, AttemptSummaryData, AttemptTimelineData, AttemptTraceData, AttemptUsageData, } from "./model/types.ts";
33
+ export type { AttemptAssertionsData, AttemptConversationData, AttemptConversationReply, AttemptConversationRound, AttemptDiagnosticsData, AttemptDiffData, AttemptDiffFileEntry, AttemptErrorData, AttemptFixPromptData, AttemptSourceData, AttemptSourceLineData, AttemptSourceTurn, AttemptSummaryData, AttemptTimelineData, AttemptTraceData, AttemptUsageData, } from "./model/types.ts";
34
34
  export type { AttemptHandle, Results, Scope, Snapshot } from "../results/types.ts";
@@ -1,12 +1,12 @@
1
1
  import type { Verdict } from "../../types.ts";
2
2
  import { type LocalizedText, type ReportLocale } from "./locale.ts";
3
3
  /**
4
- * experiment 行的显示名:给了父路径 `relativeTo` 且它确是前缀,就去掉 `relativeTo + "/"`,
5
- * 只留 id 末段——供自定义报告在已知父路径的上下文里显式调用,避免每行重复文件夹名。默认
6
- * `ExperimentComparison` 不传 `relativeTo`,完整 id 始终可见。不给 `relativeTo`、或它不是
7
- * 前缀时原样返回完整 id。完整 id 仍是排序 / 着色 / 折叠的键,调用方不要拿这个显示名当身份用。
4
+ * 一组 id 的显示名:每个 id 缩成在这组里唯一的最短路径后缀,重名逐步加长到能区分为止
5
+ * (与 `MetricScatter` 点标签同一算法,两处共用本函数以保证同一份 experiment id 在散点和
6
+ * 列表里缩成同一个显示名)。单个 id、或所有 id 深度不同时也照常缩到各自的最短唯一后缀。
7
+ * 完整 id 不受影响,调用方仍用它做排序 / 过滤 / 折叠的身份键,这里只产出显示名。
8
8
  */
9
- export declare function experimentDisplayName(experimentId: string, relativeTo?: string): string;
9
+ export declare function shortestUniqueLabels(ids: readonly string[]): Map<string, string>;
10
10
  export declare function formatMetricValue(value: number, unit?: string): string;
11
11
  /**
12
12
  * MetricCell.display 的生成:为官方生成面覆盖的每个 locale(DISPLAY_LOCALES)各生成一份;
@@ -3,16 +3,42 @@
3
3
  // metric.display 可整体覆盖;这里只负责默认。
4
4
  import { DISPLAY_LOCALES } from "./locale.js";
5
5
  /**
6
- * experiment 行的显示名:给了父路径 `relativeTo` 且它确是前缀,就去掉 `relativeTo + "/"`,
7
- * 只留 id 末段——供自定义报告在已知父路径的上下文里显式调用,避免每行重复文件夹名。默认
8
- * `ExperimentComparison` 不传 `relativeTo`,完整 id 始终可见。不给 `relativeTo`、或它不是
9
- * 前缀时原样返回完整 id。完整 id 仍是排序 / 着色 / 折叠的键,调用方不要拿这个显示名当身份用。
6
+ * 一组 id 的显示名:每个 id 缩成在这组里唯一的最短路径后缀,重名逐步加长到能区分为止
7
+ * (与 `MetricScatter` 点标签同一算法,两处共用本函数以保证同一份 experiment id 在散点和
8
+ * 列表里缩成同一个显示名)。单个 id、或所有 id 深度不同时也照常缩到各自的最短唯一后缀。
9
+ * 完整 id 不受影响,调用方仍用它做排序 / 过滤 / 折叠的身份键,这里只产出显示名。
10
10
  */
11
- export function experimentDisplayName(experimentId, relativeTo) {
12
- if (relativeTo && experimentId.startsWith(`${relativeTo}/`)) {
13
- return experimentId.slice(relativeTo.length + 1);
11
+ export function shortestUniqueLabels(ids) {
12
+ const segsOf = (id) => id.split("/").filter(Boolean);
13
+ const depth = new Map(ids.map((id) => [id, 1]));
14
+ for (;;) {
15
+ const byLabel = new Map();
16
+ for (const id of ids) {
17
+ const segs = segsOf(id);
18
+ const label = segs.slice(-Math.min(depth.get(id), segs.length)).join("/") || id;
19
+ byLabel.set(label, [...(byLabel.get(label) ?? []), id]);
20
+ }
21
+ let grew = false;
22
+ for (const group of byLabel.values()) {
23
+ if (group.length < 2)
24
+ continue;
25
+ for (const id of group) {
26
+ const segs = segsOf(id);
27
+ if (depth.get(id) < segs.length) {
28
+ depth.set(id, depth.get(id) + 1);
29
+ grew = true;
30
+ }
31
+ }
32
+ }
33
+ if (!grew) {
34
+ const out = new Map();
35
+ for (const id of ids) {
36
+ const segs = segsOf(id);
37
+ out.set(id, segs.slice(-Math.min(depth.get(id), segs.length)).join("/") || id);
38
+ }
39
+ return out;
40
+ }
14
41
  }
15
- return experimentId;
16
42
  }
17
43
  /** 一位小数、去掉无意义的 ".0" 尾巴。 */
18
44
  function trimmed(n) {
@@ -423,13 +423,27 @@ export interface AttemptAssertionsData {
423
423
  items: AssertionResult[];
424
424
  }[];
425
425
  }
426
- /** `AttemptSource` 的 data:AnnotatedEvalSource 的展示投影;没有 source 时 null。 */
426
+ /** `AttemptSource` 源码行内的一轮执行:send 头事实 + 标准事件流归并出的完整回复。 */
427
+ export interface AttemptSourceTurn {
428
+ label: string;
429
+ status: "completed" | "failed" | "waiting";
430
+ durationMs?: number;
431
+ sentText: string;
432
+ replies: AttemptConversationReply[];
433
+ }
434
+ /** AnnotatedSourceLine 加上 web 源码视图需要的行内执行轮。 */
435
+ export interface AttemptSourceLineData extends AnnotatedSourceLine {
436
+ turns: AttemptSourceTurn[];
437
+ }
438
+ /** `AttemptSource` 的 data:AnnotatedEvalSource + 按 loc 投影的标准事件流;没有 source 时 null。 */
427
439
  export interface AttemptSourceData {
428
440
  /** text 面拼 `niceeval show <locator> --source` 下钻命令用;web 面不需要。 */
429
441
  locator: AttemptLocator;
430
442
  sourcePath: string;
431
- lines: AnnotatedSourceLine[];
443
+ lines: AttemptSourceLineData[];
432
444
  unmapped: AssertionResult[];
445
+ /** 没有 loc、指向其它文件或越界的轮次;不能静默丢弃,放在源码块末尾。 */
446
+ unlocatedTurns: AttemptSourceTurn[];
433
447
  summary: AnnotatedEvalSourceSummary;
434
448
  }
435
449
  /** `AttemptFixPrompt` 的 data:单条 attempt 的复制修复 prompt;passed/skipped 或无可操作失败时 null。 */
@@ -25,7 +25,7 @@ export { AttemptDiagnostics } from "../components/attempt-detail/AttemptDiagnost
25
25
  export { AttemptUsage } from "../components/attempt-detail/AttemptUsage.tsx";
26
26
  export { AttemptTrace } from "../components/attempt-detail/AttemptTrace.tsx";
27
27
  export { AttemptDiff } from "../components/attempt-detail/AttemptDiff.tsx";
28
- export type { AttemptAssertionsData, AttemptConversationData, AttemptConversationReply, AttemptConversationRound, AttemptDiagnosticsData, AttemptDiffData, AttemptDiffFileEntry, AttemptErrorData, AttemptFixPromptData, AttemptListItem, AttemptLocator, AttemptSourceData, AttemptSummaryData, AttemptTimelineData, AttemptTraceData, AttemptUsageData, CopyFixPromptData, DeltaData, EvalListItem, ExperimentListEvalRow, ExperimentListItem, HeroData, LineData, MatrixData, MetricCell, MetricColumn, ScatterData, ScopeSummaryData, ScopeWarning, ScoreboardData, TableData, TraceSpanSummary, TraceWaterfallRow, VerdictTally, } from "../model/types.ts";
28
+ export type { AttemptAssertionsData, AttemptConversationData, AttemptConversationReply, AttemptConversationRound, AttemptDiagnosticsData, AttemptDiffData, AttemptDiffFileEntry, AttemptErrorData, AttemptFixPromptData, AttemptListItem, AttemptLocator, AttemptSourceData, AttemptSourceLineData, AttemptSourceTurn, AttemptSummaryData, AttemptTimelineData, AttemptTraceData, AttemptUsageData, CopyFixPromptData, DeltaData, EvalListItem, ExperimentListEvalRow, ExperimentListItem, HeroData, LineData, MatrixData, MetricCell, MetricColumn, ScatterData, ScopeSummaryData, ScopeWarning, ScoreboardData, TableData, TraceSpanSummary, TraceWaterfallRow, VerdictTally, } from "../model/types.ts";
29
29
  export type { AttemptEvidence, AttemptEvidenceCapabilities } from "../../results/attempt-evidence.ts";
30
30
  export { DEFAULT_REPORT_LOCALE, resolveLocalizedText, resolveMetricLabel } from "../model/locale.ts";
31
31
  export type { LocalizedText, ReportLocale } from "../model/locale.ts";
@@ -410,7 +410,8 @@ export interface ExperimentDef {
410
410
  labels?: Record<string, string | number>;
411
411
  /** 同一 eval 重复跑几次(结果各计一条 attempt);省略/CLI `--runs` 覆盖时默认 1。 */
412
412
  runs?: number;
413
- /** 一次重复(runs > 1)里某次 attempt 失败后是否跳过剩余重复;省略默认 true(提前退出省钱)。 */
413
+ /** 一次重复(runs > 1)里某次 attempt 通过后是否跳过剩余重复;省略默认 false(`runs` 跑满、测完整通过率),
414
+ * 显式打开用于「只想知道能不能过」的省钱场景。 */
414
415
  earlyExit?: boolean;
415
416
  /**
416
417
  * 这个实验覆盖哪些 eval:`"*"` 全部、字符串数组按 id 前缀、或自定义谓词(逐条收到发现并扇出后的
@@ -1,6 +1,5 @@
1
1
  import type { Sandbox, CommandResult, CommandOptions, SandboxFile, SourceFiles, ReadSourceFilesOptions } from "../types.ts";
2
2
  import { type SandboxProvisionErrorKind } from "./errors.ts";
3
- /** e2b 的限流错误是 SDK 原生的 RateLimitError(HTTP 429 映射而来);见 resolve.ts 的 withProvisionRetry。 */
4
3
  /**
5
4
  * Provisioning 重试前的对账:按 metadata 里的 provision token 检索远端实例,查到即 kill。
6
5
  * 检索或销毁失败必须抛出——对账是重试的硬前置,静默放行等于盲重试,会复制计费实例
@@ -1,5 +1,11 @@
1
1
  import type { AssertionResult, PrimaryAssertionSummary, Verdict } from "./types.ts";
2
- /** 摘要面的单值收口:折单行 + 240 字符上限。任何把断言事实放进「行」里的面共用这一条。 */
2
+ /**
3
+ * 剥离 ANSI 转义与其余不可打印控制字节,保留可打印字符与结构性空白(换行 / 制表)。给需要
4
+ * 完整多行值的面(报告详情)直接用;`summaryText` 在此基础上再折单行 + 截断。jest 合法打印的
5
+ * `✕ ✓ › ❯ ↓ │`(均 ≥ U+2020)在保留范围内,不误删。
6
+ */
7
+ export declare function stripControl(value: string): string;
8
+ /** 摘要面的单值收口:剥控制字节 + 折单行 + 240 字符上限。任何把断言事实放进「行」里的面共用这一条。 */
3
9
  export declare function summaryText(value: string): string;
4
10
  /**
5
11
  * 按公开展示契约选择主失败断言:failed gate 优先;只有 soft 促成 failed verdict 时才取 soft;
@@ -11,9 +11,29 @@ const SUMMARY_TEXT_MAX_CHARS = 240;
11
11
  * 终端 columns——agent profile 的 handoff 不是 TTY,不能按运行时宽度变化。
12
12
  */
13
13
  const DETAIL_LINE_MAX_CHARS = 100;
14
- /** 摘要面的单值收口:折单行 + 240 字符上限。任何把断言事实放进「行」里的面共用这一条。 */
14
+ // 捕获内容(received=命令输出 / expected=源码 / evidence)常带被测工具的着色:jest/vitest 的
15
+ // 代码帧、行号、✕ 都由 ANSI 转义(ESC[…m 等)上色。这些 ESC(U+001B)不是 \s,若原样落进任何
16
+ // 面,终端会重新解释它们(被单行截断从序列中间切开时尤其乱),HTML 报告则把 ESC[2m28|ESC[22m
17
+ // 当字面文本渲染。所以任何展示面在渲染捕获内容前先剥控制字节;剥的是展示投影,不改存进
18
+ // AssertionResult / artifact 的原始字节(完整证据仍在 events.json / diff.json)。
19
+ // CSI(ESC[…,含 SGR 着色 / 光标控制)与 OSC(ESC]…,以 BEL 或 ST 收尾);OSC 的 payload 一并吃掉,
20
+ // 不让它作为裸文本泄漏。没配成序列的裸 ESC 由 OTHER_CONTROL 兜底。
21
+ // eslint-disable-next-line no-control-regex
22
+ const ANSI_ESCAPE = /\u001B(?:\[[0-9;:?]*[ -/]*[@-~]|\][^\u0007\u001B]*(?:\u0007|\u001B\\))/g;
23
+ // 其余不可打印 C0/C1(含裸 ESC);保留 \t\n\f\r 交给下游折空白规则,不在这里塌成空。
24
+ // eslint-disable-next-line no-control-regex
25
+ const OTHER_CONTROL = /[\u0000-\u0008\u000B\u000E-\u001F\u007F-\u009F]/g;
26
+ /**
27
+ * 剥离 ANSI 转义与其余不可打印控制字节,保留可打印字符与结构性空白(换行 / 制表)。给需要
28
+ * 完整多行值的面(报告详情)直接用;`summaryText` 在此基础上再折单行 + 截断。jest 合法打印的
29
+ * `✕ ✓ › ❯ ↓ │`(均 ≥ U+2020)在保留范围内,不误删。
30
+ */
31
+ export function stripControl(value) {
32
+ return value.replace(ANSI_ESCAPE, "").replace(OTHER_CONTROL, "");
33
+ }
34
+ /** 摘要面的单值收口:剥控制字节 + 折单行 + 240 字符上限。任何把断言事实放进「行」里的面共用这一条。 */
15
35
  export function summaryText(value) {
16
- const singleLine = value.replace(/\s+/g, " ").trim();
36
+ const singleLine = stripControl(value).replace(/\s+/g, " ").trim();
17
37
  return singleLine.length <= SUMMARY_TEXT_MAX_CHARS
18
38
  ? singleLine
19
39
  : `${singleLine.slice(0, SUMMARY_TEXT_MAX_CHARS - 1)}…`;
@@ -19,7 +19,7 @@ description: "一个 AI SDK v7 聊天应用,对着它的 HTTP 接口无侵入
19
19
 
20
20
  接入的全部代码变更(生成时从两个目录实测统计):
21
21
 
22
- <table className="gd-summary"><tbody><tr><th>{"类别"}</th><th>{"文件数"}</th><th>{"行数"}</th></tr><tr><td>{"应用侧配置(必要:依赖声明)"}</td><td>{"3"}</td><td>{"+5 −1"}</td></tr><tr><td>{"adapter(必要:传输粘合,协议映射在官方包里)"}</td><td>{"2"}</td><td>{"+27"}</td></tr><tr><td>{"evals 与 experiments(评测内容,按需增长)"}</td><td>{"8"}</td><td>{"+141"}</td></tr><tr className="gd-total"><td>{"合计"}</td><td>{"13"}</td><td>{"+173 −1"}</td></tr></tbody></table>
22
+ <table className="gd-summary"><tbody><tr><th>{"类别"}</th><th>{"文件数"}</th><th>{"行数"}</th></tr><tr><td>{"应用侧配置(必要:依赖声明)"}</td><td>{"3"}</td><td>{"+5 −1"}</td></tr><tr><td>{"adapter(必要:传输粘合,协议映射在官方包里)"}</td><td>{"2"}</td><td>{"+27"}</td></tr><tr><td>{"evals 与 experiments(评测内容,按需增长)"}</td><td>{"8"}</td><td>{"+139"}</td></tr><tr className="gd-total"><td>{"合计"}</td><td>{"13"}</td><td>{"+171 −1"}</td></tr></tbody></table>
23
23
 
24
24
  ## 文件清单
25
25
 
@@ -126,15 +126,15 @@ ai-sdk-v7/
126
126
  </div>
127
127
 
128
128
  <div className="gd-file">
129
- <div className="gd-head"><span className="gd-name">{"experiments/compare-models/deepseek-v4-flash.ts"}</span><span className="gd-stats"><span className="gd-plus">{"+13"}</span></span></div>
129
+ <div className="gd-head"><span className="gd-name">{"experiments/compare-models/deepseek-v4-flash.ts"}</span><span className="gd-stats"><span className="gd-plus">{"+12"}</span></span></div>
130
130
  <div className="gd-body">
131
- <table className="gd-table"><tbody><tr className="gd-add"><td className="gd-ln"></td><td className="gd-ln">{"1"}</td><td className="gd-sign">{"+"}</td><td className="gd-code"><span className="gdt4">{"import"}</span><span className="gdt0">{" { defineExperiment } "}</span><span className="gdt4">{"from"}</span><span className="gdt0">{" "}</span><span className="gdt2">{"\"niceeval\""}</span><span className="gdt0">{";"}</span></td></tr><tr className="gd-add"><td className="gd-ln"></td><td className="gd-ln">{"2"}</td><td className="gd-sign">{"+"}</td><td className="gd-code"><span className="gdt4">{"import"}</span><span className="gdt0">{" agent "}</span><span className="gdt4">{"from"}</span><span className="gdt0">{" "}</span><span className="gdt2">{"\"../../agents/ai-sdk-v7.ts\""}</span><span className="gdt0">{";"}</span></td></tr><tr className="gd-add"><td className="gd-ln"></td><td className="gd-ln">{"3"}</td><td className="gd-sign">{"+"}</td><td className="gd-code">{" "}</td></tr><tr className="gd-add"><td className="gd-ln"></td><td className="gd-ln">{"4"}</td><td className="gd-sign">{"+"}</td><td className="gd-code"><span className="gdt6">{"// compare-models 组的一格:deepseek-v4-flash。一文件一配置(单 model),model 经 ctx.model"}</span></td></tr><tr className="gd-add"><td className="gd-ln"></td><td className="gd-ln">{"5"}</td><td className="gd-sign">{"+"}</td><td className="gd-code"><span className="gdt6">{"// 走请求体传给应用,同一个 server 实例服务所有 model,不用重启进程。"}</span></td></tr><tr className="gd-add"><td className="gd-ln"></td><td className="gd-ln">{"6"}</td><td className="gd-sign">{"+"}</td><td className="gd-code"><span className="gdt4">{"export"}</span><span className="gdt0">{" "}</span><span className="gdt4">{"default"}</span><span className="gdt0">{" "}</span><span className="gdt5">{"defineExperiment"}</span><span className="gdt0">{"({"}</span></td></tr><tr className="gd-add"><td className="gd-ln"></td><td className="gd-ln">{"7"}</td><td className="gd-sign">{"+"}</td><td className="gd-code"><span className="gdt0">{" description: "}</span><span className="gdt2">{"\"deepseek-v4-flash: 对比模型\""}</span><span className="gdt0">{","}</span></td></tr><tr className="gd-add"><td className="gd-ln"></td><td className="gd-ln">{"8"}</td><td className="gd-sign">{"+"}</td><td className="gd-code"><span className="gdt0">{" agent,"}</span></td></tr><tr className="gd-add"><td className="gd-ln"></td><td className="gd-ln">{"9"}</td><td className="gd-sign">{"+"}</td><td className="gd-code"><span className="gdt0">{" model: "}</span><span className="gdt2">{"\"deepseek-v4-flash\""}</span><span className="gdt0">{","}</span></td></tr><tr className="gd-add"><td className="gd-ln"></td><td className="gd-ln">{"10"}</td><td className="gd-sign">{"+"}</td><td className="gd-code"><span className="gdt0">{" runs: "}</span><span className="gdt1">{"2"}</span><span className="gdt0">{","}</span></td></tr><tr className="gd-add"><td className="gd-ln"></td><td className="gd-ln">{"11"}</td><td className="gd-sign">{"+"}</td><td className="gd-code"><span className="gdt0">{" earlyExit: "}</span><span className="gdt1">{"true"}</span><span className="gdt0">{","}</span></td></tr><tr className="gd-add"><td className="gd-ln"></td><td className="gd-ln">{"12"}</td><td className="gd-sign">{"+"}</td><td className="gd-code"><span className="gdt0">{" budget: "}</span><span className="gdt1">{"2"}</span><span className="gdt0">{","}</span></td></tr><tr className="gd-add"><td className="gd-ln"></td><td className="gd-ln">{"13"}</td><td className="gd-sign">{"+"}</td><td className="gd-code"><span className="gdt0">{"});"}</span></td></tr></tbody></table>
131
+ <table className="gd-table"><tbody><tr className="gd-add"><td className="gd-ln"></td><td className="gd-ln">{"1"}</td><td className="gd-sign">{"+"}</td><td className="gd-code"><span className="gdt4">{"import"}</span><span className="gdt0">{" { defineExperiment } "}</span><span className="gdt4">{"from"}</span><span className="gdt0">{" "}</span><span className="gdt2">{"\"niceeval\""}</span><span className="gdt0">{";"}</span></td></tr><tr className="gd-add"><td className="gd-ln"></td><td className="gd-ln">{"2"}</td><td className="gd-sign">{"+"}</td><td className="gd-code"><span className="gdt4">{"import"}</span><span className="gdt0">{" agent "}</span><span className="gdt4">{"from"}</span><span className="gdt0">{" "}</span><span className="gdt2">{"\"../../agents/ai-sdk-v7.ts\""}</span><span className="gdt0">{";"}</span></td></tr><tr className="gd-add"><td className="gd-ln"></td><td className="gd-ln">{"3"}</td><td className="gd-sign">{"+"}</td><td className="gd-code">{" "}</td></tr><tr className="gd-add"><td className="gd-ln"></td><td className="gd-ln">{"4"}</td><td className="gd-sign">{"+"}</td><td className="gd-code"><span className="gdt6">{"// compare-models 组的一格:deepseek-v4-flash。一文件一配置(单 model),model 经 ctx.model"}</span></td></tr><tr className="gd-add"><td className="gd-ln"></td><td className="gd-ln">{"5"}</td><td className="gd-sign">{"+"}</td><td className="gd-code"><span className="gdt6">{"// 走请求体传给应用,同一个 server 实例服务所有 model,不用重启进程。"}</span></td></tr><tr className="gd-add"><td className="gd-ln"></td><td className="gd-ln">{"6"}</td><td className="gd-sign">{"+"}</td><td className="gd-code"><span className="gdt4">{"export"}</span><span className="gdt0">{" "}</span><span className="gdt4">{"default"}</span><span className="gdt0">{" "}</span><span className="gdt5">{"defineExperiment"}</span><span className="gdt0">{"({"}</span></td></tr><tr className="gd-add"><td className="gd-ln"></td><td className="gd-ln">{"7"}</td><td className="gd-sign">{"+"}</td><td className="gd-code"><span className="gdt0">{" description: "}</span><span className="gdt2">{"\"deepseek-v4-flash: 对比模型\""}</span><span className="gdt0">{","}</span></td></tr><tr className="gd-add"><td className="gd-ln"></td><td className="gd-ln">{"8"}</td><td className="gd-sign">{"+"}</td><td className="gd-code"><span className="gdt0">{" agent,"}</span></td></tr><tr className="gd-add"><td className="gd-ln"></td><td className="gd-ln">{"9"}</td><td className="gd-sign">{"+"}</td><td className="gd-code"><span className="gdt0">{" model: "}</span><span className="gdt2">{"\"deepseek-v4-flash\""}</span><span className="gdt0">{","}</span></td></tr><tr className="gd-add"><td className="gd-ln"></td><td className="gd-ln">{"10"}</td><td className="gd-sign">{"+"}</td><td className="gd-code"><span className="gdt0">{" runs: "}</span><span className="gdt1">{"2"}</span><span className="gdt0">{", "}</span><span className="gdt6">{"// 跑满 2 次,才能比较 model 间的通过率"}</span></td></tr><tr className="gd-add"><td className="gd-ln"></td><td className="gd-ln">{"11"}</td><td className="gd-sign">{"+"}</td><td className="gd-code"><span className="gdt0">{" budget: "}</span><span className="gdt1">{"2"}</span><span className="gdt0">{","}</span></td></tr><tr className="gd-add"><td className="gd-ln"></td><td className="gd-ln">{"12"}</td><td className="gd-sign">{"+"}</td><td className="gd-code"><span className="gdt0">{"});"}</span></td></tr></tbody></table>
132
132
  </div>
133
133
  </div>
134
134
 
135
135
  <div className="gd-file">
136
- <div className="gd-head"><span className="gd-name">{"experiments/compare-models/deepseek-v4-pro.ts"}</span><span className="gd-stats"><span className="gd-plus">{"+12"}</span></span></div>
136
+ <div className="gd-head"><span className="gd-name">{"experiments/compare-models/deepseek-v4-pro.ts"}</span><span className="gd-stats"><span className="gd-plus">{"+11"}</span></span></div>
137
137
  <div className="gd-body">
138
- <table className="gd-table"><tbody><tr className="gd-add"><td className="gd-ln"></td><td className="gd-ln">{"1"}</td><td className="gd-sign">{"+"}</td><td className="gd-code"><span className="gdt4">{"import"}</span><span className="gdt0">{" { defineExperiment } "}</span><span className="gdt4">{"from"}</span><span className="gdt0">{" "}</span><span className="gdt2">{"\"niceeval\""}</span><span className="gdt0">{";"}</span></td></tr><tr className="gd-add"><td className="gd-ln"></td><td className="gd-ln">{"2"}</td><td className="gd-sign">{"+"}</td><td className="gd-code"><span className="gdt4">{"import"}</span><span className="gdt0">{" agent "}</span><span className="gdt4">{"from"}</span><span className="gdt0">{" "}</span><span className="gdt2">{"\"../../agents/ai-sdk-v7.ts\""}</span><span className="gdt0">{";"}</span></td></tr><tr className="gd-add"><td className="gd-ln"></td><td className="gd-ln">{"3"}</td><td className="gd-sign">{"+"}</td><td className="gd-code">{" "}</td></tr><tr className="gd-add"><td className="gd-ln"></td><td className="gd-ln">{"4"}</td><td className="gd-sign">{"+"}</td><td className="gd-code"><span className="gdt6">{"// compare-models 组的一格:deepseek-v4-pro。与 deepseek-v4-flash.ts 钉住一切、只差 model。"}</span></td></tr><tr className="gd-add"><td className="gd-ln"></td><td className="gd-ln">{"5"}</td><td className="gd-sign">{"+"}</td><td className="gd-code"><span className="gdt4">{"export"}</span><span className="gdt0">{" "}</span><span className="gdt4">{"default"}</span><span className="gdt0">{" "}</span><span className="gdt5">{"defineExperiment"}</span><span className="gdt0">{"({"}</span></td></tr><tr className="gd-add"><td className="gd-ln"></td><td className="gd-ln">{"6"}</td><td className="gd-sign">{"+"}</td><td className="gd-code"><span className="gdt0">{" description: "}</span><span className="gdt2">{"\"deepseek-v4-pro: 对比模型\""}</span><span className="gdt0">{","}</span></td></tr><tr className="gd-add"><td className="gd-ln"></td><td className="gd-ln">{"7"}</td><td className="gd-sign">{"+"}</td><td className="gd-code"><span className="gdt0">{" agent,"}</span></td></tr><tr className="gd-add"><td className="gd-ln"></td><td className="gd-ln">{"8"}</td><td className="gd-sign">{"+"}</td><td className="gd-code"><span className="gdt0">{" model: "}</span><span className="gdt2">{"\"deepseek-v4-pro\""}</span><span className="gdt0">{","}</span></td></tr><tr className="gd-add"><td className="gd-ln"></td><td className="gd-ln">{"9"}</td><td className="gd-sign">{"+"}</td><td className="gd-code"><span className="gdt0">{" runs: "}</span><span className="gdt1">{"2"}</span><span className="gdt0">{","}</span></td></tr><tr className="gd-add"><td className="gd-ln"></td><td className="gd-ln">{"10"}</td><td className="gd-sign">{"+"}</td><td className="gd-code"><span className="gdt0">{" earlyExit: "}</span><span className="gdt1">{"true"}</span><span className="gdt0">{","}</span></td></tr><tr className="gd-add"><td className="gd-ln"></td><td className="gd-ln">{"11"}</td><td className="gd-sign">{"+"}</td><td className="gd-code"><span className="gdt0">{" budget: "}</span><span className="gdt1">{"2"}</span><span className="gdt0">{","}</span></td></tr><tr className="gd-add"><td className="gd-ln"></td><td className="gd-ln">{"12"}</td><td className="gd-sign">{"+"}</td><td className="gd-code"><span className="gdt0">{"});"}</span></td></tr></tbody></table>
138
+ <table className="gd-table"><tbody><tr className="gd-add"><td className="gd-ln"></td><td className="gd-ln">{"1"}</td><td className="gd-sign">{"+"}</td><td className="gd-code"><span className="gdt4">{"import"}</span><span className="gdt0">{" { defineExperiment } "}</span><span className="gdt4">{"from"}</span><span className="gdt0">{" "}</span><span className="gdt2">{"\"niceeval\""}</span><span className="gdt0">{";"}</span></td></tr><tr className="gd-add"><td className="gd-ln"></td><td className="gd-ln">{"2"}</td><td className="gd-sign">{"+"}</td><td className="gd-code"><span className="gdt4">{"import"}</span><span className="gdt0">{" agent "}</span><span className="gdt4">{"from"}</span><span className="gdt0">{" "}</span><span className="gdt2">{"\"../../agents/ai-sdk-v7.ts\""}</span><span className="gdt0">{";"}</span></td></tr><tr className="gd-add"><td className="gd-ln"></td><td className="gd-ln">{"3"}</td><td className="gd-sign">{"+"}</td><td className="gd-code">{" "}</td></tr><tr className="gd-add"><td className="gd-ln"></td><td className="gd-ln">{"4"}</td><td className="gd-sign">{"+"}</td><td className="gd-code"><span className="gdt6">{"// compare-models 组的一格:deepseek-v4-pro。与 deepseek-v4-flash.ts 钉住一切、只差 model。"}</span></td></tr><tr className="gd-add"><td className="gd-ln"></td><td className="gd-ln">{"5"}</td><td className="gd-sign">{"+"}</td><td className="gd-code"><span className="gdt4">{"export"}</span><span className="gdt0">{" "}</span><span className="gdt4">{"default"}</span><span className="gdt0">{" "}</span><span className="gdt5">{"defineExperiment"}</span><span className="gdt0">{"({"}</span></td></tr><tr className="gd-add"><td className="gd-ln"></td><td className="gd-ln">{"6"}</td><td className="gd-sign">{"+"}</td><td className="gd-code"><span className="gdt0">{" description: "}</span><span className="gdt2">{"\"deepseek-v4-pro: 对比模型\""}</span><span className="gdt0">{","}</span></td></tr><tr className="gd-add"><td className="gd-ln"></td><td className="gd-ln">{"7"}</td><td className="gd-sign">{"+"}</td><td className="gd-code"><span className="gdt0">{" agent,"}</span></td></tr><tr className="gd-add"><td className="gd-ln"></td><td className="gd-ln">{"8"}</td><td className="gd-sign">{"+"}</td><td className="gd-code"><span className="gdt0">{" model: "}</span><span className="gdt2">{"\"deepseek-v4-pro\""}</span><span className="gdt0">{","}</span></td></tr><tr className="gd-add"><td className="gd-ln"></td><td className="gd-ln">{"9"}</td><td className="gd-sign">{"+"}</td><td className="gd-code"><span className="gdt0">{" runs: "}</span><span className="gdt1">{"2"}</span><span className="gdt0">{", "}</span><span className="gdt6">{"// 跑满 2 次,才能比较 model 间的通过率"}</span></td></tr><tr className="gd-add"><td className="gd-ln"></td><td className="gd-ln">{"10"}</td><td className="gd-sign">{"+"}</td><td className="gd-code"><span className="gdt0">{" budget: "}</span><span className="gdt1">{"2"}</span><span className="gdt0">{","}</span></td></tr><tr className="gd-add"><td className="gd-ln"></td><td className="gd-ln">{"11"}</td><td className="gd-sign">{"+"}</td><td className="gd-code"><span className="gdt0">{"});"}</span></td></tr></tbody></table>
139
139
  </div>
140
140
  </div>
@@ -39,19 +39,27 @@ npx niceeval exp local --max-concurrency 8
39
39
 
40
40
  远程 HTTP agent 可以用更高并发;本地 Docker Sandbox 通常需要更低并发,避免 CPU、内存和磁盘竞争。
41
41
 
42
+ 把跑得慢的实验和跑得快的实验混在一次命令里跑是安全的:并发名额优先给「要跑最多轮才能跑完」的实验(比如 `maxConcurrency: 1` 又有几十个 Attempt 的那个),快的实验见缝插针补空位。慢实验有一段 `setup`(起隧道、起共享服务)时也一样——它等 `setup` 的这段时间不占名额,别的实验照常跑,`setup` 一好它就拿下一个空出来的名额,不用为它单开一次命令。
43
+
42
44
  ## runs 与 early-exit
43
45
 
44
46
  ```bash
45
47
  npx niceeval exp local fixtures/button --runs 5
46
- npx niceeval exp local fixtures/button --runs 5 --no-early-exit
48
+ npx niceeval exp local fixtures/button --runs 5 --early-exit
47
49
  ```
48
50
 
49
- `runs` 用于测 pass rate。首过即停默认开启:某个 Attempt 通过后,同一 eval 的剩余 Attempt 会被停止。想拿完整的通过率分布时,用 `--no-early-exit` 关闭,让每个 eval 跑满 `runs` 次。
51
+ `runs` 用于测 pass rate,默认跑满 `runs` 次,给出完整的通过率分布。首过即停默认关闭;只想知道"这题能不能过"、不在乎完整分布时,用 `--early-exit` 打开——某个 Attempt 通过后,同一 eval 的剩余 Attempt 会被停止。
52
+
53
+ `runs` 的多次 Attempt 默认并发派发,并发上限由 `--max-concurrency`(或该实验的 `maxConcurrency`)决定——不会等上一次的结果出来再决定要不要派发下一次。首过即停能省下的,只是还没抢到并发名额、原本要排队的那些 Attempt;已经在跑的不受影响。想要"跑一次,过了就停、没过才跑下一次"这种一个接一个的效果,把该实验的 `maxConcurrency` 设成 1 并开启 `earlyExit`:并发名额只有一个时,同一个 eval 的 Attempt 只能排队依次跑,首过即停自然就能在下一次派发前生效。
50
54
 
51
55
  ## 缓存
52
56
 
53
57
  [NiceEval](https://niceeval.com/) 可以根据输入、配置和相关文件 fingerprint 跳过已判定为 `passed` 或 `failed` 的结果——两者都是判定确定的终态。`errored`(超时、Sandbox 异常等框架/环境层面的不确定失败)永远重试。缓存适合加速迭代,但如果你在调试非确定性行为,应该明确关闭(`--force`)或清理相关缓存。
54
58
 
59
+ ## Turn 瞬时错误重试
60
+
61
+ 限流、连接建立失败这类瞬时错误,NiceEval 会在同一个 Turn 里自动做有限次数的指数退避重试,不需要你重跑整个实验;等待重试的 Attempt 会让出并发名额给别的 Attempt。只有能确认 Agent 还没开始处理这次输入的错误才会重试——请求已经开始、中途断流的情况不重试,直接记为 `errored`,避免 Agent 把已经做过的操作再做一遍。重试用尽仍失败时,该 Attempt 记为 `errored`,下次运行照常重试(见上面的缓存规则)。
62
+
55
63
  ## 超时和预算
56
64
 
57
65
  ```bash
@@ -145,7 +145,7 @@ npx niceeval view # 网页交互浏览
145
145
  如果你想让 Coding Agent 直接帮项目接入 [NiceEval](https://niceeval.com/),让它先读安装入口:
146
146
 
147
147
  ```text
148
- READ https://niceeval.com/INIT.md and install niceeval for this repo.
148
+ READ https://niceeval.com/INIT.md and set up niceeval for this repo: install it, integrate it with this project, and run the first eval end to end.
149
149
  ```
150
150
 
151
151
  <CardGroup cols={3}>
@@ -105,7 +105,7 @@ npx niceeval exp models weather
105
105
  | `--output` | string | 反馈 profile:`auto`(默认)按环境自动选择,`human` / `agent` / `ci` 强制指定;只改变终端展示,不改变选择、调度、判定、artifact 或退出码。`auto` 依次判定:stderr 是 TTY → human;否则 `CI`(或其它常见 CI 平台环境变量)存在 → ci;否则 → agent。 |
106
106
  | `--force` | boolean | 忽略上次运行结果,不跳过已通过的 (experiment, eval) 组合,强制全部重跑。 |
107
107
  | `--strict` | boolean | CI 中推荐使用:让软阈值(`soft`)失败也计入整条 eval 的 verdict。 |
108
- | `--early-exit` / `--no-early-exit` | boolean | 某个 eval 的一次 attempt 通过后,停止该 eval 剩余的 attempts。 |
108
+ | `--early-exit` / `--no-early-exit` | boolean | 某个 eval 的一次 attempt 通过后,停止该 eval 剩余的 attempts;省略默认关(`runs` 默认跑满、测完整通过率)。 |
109
109
  | `--open` / `--no-open` | boolean | `view` 命令专用:启动后自动打开浏览器(默认行为)。 |
110
110
  | `--help` | boolean | 打印用法说明并退出。 |
111
111
  | `--version` | boolean | 打印 niceeval 的版本号并退出。 |
@@ -174,7 +174,7 @@ npx niceeval show weather/brooklyn --history
174
174
 
175
175
  ## `--early-exit` 与 `--strict`
176
176
 
177
- `--early-exit` 默认就是 `true`:某个评估用例的一次 attempt 通过后,会自动停止该评估用例剩余的 attempts(省钱)。所以显式传 `--early-exit` 没有效果,真正有用的是反过来关掉它——`--no-early-exit`,让 `--runs` > 1 时把每次 attempt 都跑完(测真实通过率而非提前收尾)。
177
+ `--early-exit` 默认关闭:`--runs` > 1 时默认把每次 attempt 都跑完,给出真实通过率——这是 NiceEval 衡量 agent 稳不稳的核心指标,默认不该被无声截断。只想知道"这题能不能过"、不在乎完整分布时,显式加 `--early-exit`:某个评估用例的一次 attempt 通过后,自动停止该评估用例剩余的 attempts(省钱)。实验文件里写了 `earlyExit: true` 时,用 `--no-early-exit` 强制关掉它。
178
178
 
179
179
  `--strict` 不是"更严格地报错",而是改变软阈值断言的判定:`.atLeast(n)` 这类软阈值(`soft` severity)平时失败不会把整条评估用例判为 `failed`(只是记一条不达标的断言),加了 `--strict` 之后,软阈值没达标也会让整条评估用例的 verdict 计为 `failed`。CI 中推荐加上,避免"断言分数不够但评估用例显示通过"的情况被放过。
180
180
 
@@ -1,7 +1,7 @@
1
1
  ---
2
2
  title: "niceeval/expect matchers 与自定义断言参考"
3
3
  sidebarTitle: "断言 Matchers"
4
- description: "niceeval/expect 参考:includes、equals、matches、similarity、satisfies。可链式调用 .gate() 或 .atLeast(0.7),也可以用 makeAssertion 构建自定义 matcher。"
4
+ description: "niceeval/expect 参考:includes、equals、matches、similarity、satisfies,以及形状断言 includesUrl、hasSections。可链式调用 .gate() 或 .atLeast(0.7),也可以用 makeAssertion 构建自定义 matcher。"
5
5
  ---
6
6
 
7
7
  `niceeval/expect` 提供一组可组合 matcher,传给 `t.check()` 或 `t.require()`。matcher 会返回一个 `Assertion`,并带有默认严重性:`gate` 或 `soft`。
@@ -74,6 +74,26 @@ export function similarity(expected: string): ValueAssertion { ... }
74
74
  纯字符串编辑距离,不是语义相似度——归一化 Levenshtein 距离 [0,1](1 - 编辑距离 / 较长串长度),
75
75
  不理解含义,同义改写 / 语序调整会被判低分。默认软分,阈值 0.6。
76
76
 
77
+ #### `includesUrl`
78
+
79
+ ```ts
80
+ export function includesUrl(min = 1): ValueAssertion { ... }
81
+ ```
82
+
83
+ 文本含至少 min 条(默认 1)去重后的 http(s) 链接则 1,否则 0。默认硬门槛。
84
+ 「回答有没有引用真实来源」的形状断言:没有 Judge key 时,它是「至少引用一条来源链接」
85
+ 的最低成本兜底——被测方复读题目糊弄不过去,编造的链接则留给负例或 Judge 去抓。
86
+
87
+ #### `hasSections`
88
+
89
+ ```ts
90
+ export function hasSections(min = 2): ValueAssertion { ... }
91
+ ```
92
+
93
+ 文本含至少 min 个(默认 2)Markdown 标题(行首 # 到 ######)则 1,否则 0。默认硬门槛。
94
+ 「回答是不是一份有结构的文档」的形状断言,适合研究报告、方案文档这类产出型回答——
95
+ 一段没有任何小标题的流水文本过不了。
96
+
77
97
  #### `satisfies`
78
98
 
79
99
  ```ts
@@ -152,6 +152,8 @@ description: "报告文件里能摆的全部官方双面组件:每个组件回
152
152
 
153
153
  每个 experiment 的评估用例集合来自快照记录的 `selectedEvalIds`。网页面与终端面都显示完整 Scope;需要子集时用 `--exp` 收窄,或在自定义报告里对 Scope 先 `.filter()` 再传给 `input`。
154
154
 
155
+ 实验列表行的显示名不需要额外配置:experiment id 共享公共目录前缀时(比如都在 `compare/` 下),行标签自动缩成各自的最短唯一后缀,不显示重复的前缀;末段撞名的 id 会自动加长到能互相区分为止。完整 id 始终是排序、过滤和折叠展开用的身份键,只是显示文字变短了。
156
+
155
157
  组卡使用 `Pass rate / 通过率`、`Experiments / 实验`、`Evals / Eval`、`Attempts / Attempt`、`Eval results / Eval 结果`、`Total cost / 总成本` 这套字段标签,不在标签里重复“数”“次”或“计票”。时间显示为本地化到分钟的 `Last run / 最近运行` 或 `Run range / 运行范围`,不直接暴露 ISO 字符串;成本数据覆盖不全时写明“`63/72 次有成本数据`”,不显示没有上下文的 `63/72`。六项 KPI 在宽卡片保持同一行,空间不足时按三项或两项一组换行,避免总成本单独掉到下一行。
156
158
 
157
159
  下面的 `MetricScatter`、`ScopeSummary` 和 `ExperimentList` 同样忠实消费调用方传入的数据,不推导隐藏范围。它等价于把三个组件按下面这样手工摆放——想自定义顺序或搭配其它组件时,照这个形状写:
@@ -194,7 +196,7 @@ const CompareSummary = defineComponent((_props, ctx) => (
194
196
 
195
197
  每项固定代表一个 experiment。主行显示 experiment id、agent、model、flags、评估用例判定构成、通过率、Tokens、成本和耗时。默认 `ExperimentComparison` 把当前 Scope 的全部条目交给它;组件本身不猜边界。
196
198
 
197
- 默认比较显示完整 experiment id;排序、过滤和身份也使用完整 id。中文副行用“`8 个 Eval`”而不是“`8 道题`”。
199
+ 行标签默认缩成 experiment id 在当前列表里的最短唯一后缀——末段唯一就只显示末段,撞名的 id 各自向前多取一段直到能区分为止(与 `MetricScatter` 散点的点标签同一算法)。排序、过滤和折叠展开始终用完整 id,不受显示名影响。中文副行用“`8 个 Eval`”而不是“`8 道题`”。
198
200
 
199
201
  ```tsx
200
202
  <ExperimentList filter />
@@ -46,12 +46,12 @@ available:
46
46
 
47
47
  排查顺序是「哪条断言挂了 → agent 当时做了什么 → 它到底改了什么」。
48
48
 
49
- **1. 把断言放回源码。** `--source` 显示运行时保存的那份评估用例源码(不是你工作区里可能已经改过的版本),失败的断言直接标在对应行上;`t.send(...)` 的调用行标出它产生的那一轮——轮标签(`s1/t1`,与 `--execution` / `--timing` 用同一套)、这轮成没成、花了多久:
49
+ **1. 把断言放回源码。** `--source` 显示运行时保存的那份评估用例源码(不是你工作区里可能已经改过的版本),失败的断言直接标在对应行上;`t.send(...)` 的调用行标出它产生的那一轮——轮标签(`turn1`,与 `--execution` / `--timing` 用同一套;`t.newSession()` 开的会话记 `session2/turn1`)、这轮成没成、花了多久:
50
50
 
51
51
  ```text
52
52
  $ niceeval show @1qrdcfq8 --source
53
53
  21✓ await t.send("Review the proposals and record your decision…");
54
- s1/t1 · completed · 22.4s
54
+ turn1 · completed · 22.4s
55
55
  38 for (const [issue, label] of Object.entries(expected)) {
56
56
  39 await t.group(`Issue ${issue}: selected proposal matches…`, async () => {
57
57
  40✗ t.check(Number(decisions[issue]?.selected_proposal_id), equals(label.selected_proposal_id));
@@ -64,7 +64,7 @@ $ niceeval show @1qrdcfq8 --source
64
64
 
65
65
  ```text
66
66
  $ niceeval show @1qrdcfq8 --execution
67
- TURN s1/t1 · completed · 22.4s · 12.4k tok · $0.02
67
+ turn1 · completed · 22.4s · 12.4k tok · $0.02
68
68
  USER
69
69
  Review the proposals and record your decision for each issue…
70
70
 
@@ -78,7 +78,7 @@ TURN s1/t1 · completed · 22.4s · 12.4k tok · $0.02
78
78
  Proposal 1: …
79
79
  ```
80
80
 
81
- 对话按轮分段,每轮头行给出编号(`s1/t1`)、状态、耗时和用量——这个编号和 `--diff`、`--timing` 里的轮次标签是同一套,能互相对照。
81
+ 对话按轮分段,每轮头行给出轮标签(`turn1`)、状态、耗时和用量——这个标签和 `--diff`、`--timing` 里的轮次标签是同一套,能互相对照。
82
82
 
83
83
  不想通读全文时,接 `grep` 定向查。列出这次 Attempt 用过哪些工具、各多少次:
84
84
 
@@ -99,17 +99,17 @@ niceeval show @1qrdcfq8 --execution | grep proposals
99
99
  ```text
100
100
  $ niceeval show @1qrdcfq8 --diff
101
101
  2 files changed by agent
102
- M manager_decisions.json +6 -2 s1/t1, s1/t2
103
- A notes/decision-log.md +18 s1/t2
102
+ M manager_decisions.json +6 -2 turn1, turn2
103
+ A notes/decision-log.md +18 turn2
104
104
 
105
105
  single file: niceeval show @1qrdcfq8 --diff=manager_decisions.json
106
106
  ```
107
107
 
108
- 行尾的 `s1/t1` 表示这个文件是在第几轮对话里被改的,能和 `--execution` 的轮次对上。要看单个文件的逐行改动,用 `=` 连写文件路径:
108
+ 行尾的 `turn1` 表示这个文件是在第几轮对话里被改的,能和 `--execution` 的轮次对上。要看单个文件的逐行改动,用 `=` 连写文件路径:
109
109
 
110
110
  ```text
111
111
  $ niceeval show @1qrdcfq8 --diff=manager_decisions.json
112
- M manager_decisions.json · changed in s1/t1, s1/t2
112
+ M manager_decisions.json · changed in turn1, turn2
113
113
  @@ -1,5 +1,7 @@
114
114
  {
115
115
  - "15193": { "selected_proposal_id": 1 },
@@ -0,0 +1,62 @@
1
+ ---
2
+ title: "运行被强杀后恢复"
3
+ description: "进程被 kill -9、CI 超时或断电杀掉后,重跑同一条命令续跑没跑完的部分,用 niceeval sandbox prune 收回没清理的容器,用 --teardown 补齐实验收尾。"
4
+ ---
5
+
6
+ `niceeval exp` 跑到一半被 `kill -9`、CI 时限或断电直接杀掉时,进程没有机会做任何清理。你会碰到三种残留:没跑完的评估、还在 Provider 侧占资源的 Sandbox 实例、实验 `setup` 起过但没关掉的外部服务(隧道、共享服务、license 席位)。三种各有一个恢复入口,都不需要手工翻 Docker 或云控制台。
7
+
8
+ 正常的 Ctrl+C 或 SIGTERM 不在本页范围:那些路径 NiceEval 会自己走完全部收尾,不留残留。
9
+
10
+ ## 重跑同一条命令,续跑没跑完的部分
11
+
12
+ 已经跑完并落盘的 Attempt 是可信结果,重跑时自动带入,不再花一次 Agent 和 Sandbox 的成本:
13
+
14
+ ```bash
15
+ niceeval exp compare/bub-e2b memory/commit0
16
+ ```
17
+
18
+ - 只补跑缺的部分:`--runs 5` 已经落盘 3 次,就只再跑 2 次。
19
+ - 被强杀的实验如果留了没做完的收尾,重跑会先补一次实验级 `teardown` 再开始,泄漏不会越积越多。
20
+ - 判定为 `errored` 的 Attempt 不复用,照常重跑。
21
+ - 想全部重来,加 `--force`。
22
+
23
+ ## 收回没清理的 Sandbox 实例
24
+
25
+ 先核对有哪些实例属于已经死掉的运行:
26
+
27
+ ```bash
28
+ niceeval sandbox list --orphans
29
+ ```
30
+
31
+ ```text
32
+ ID PROVIDER OWNER STARTED STATE
33
+ f31b9a02 docker pid 4242@mbp dead 2026-07-20 14:02 orphan
34
+ ```
35
+
36
+ 确认后一条命令收回:
37
+
38
+ ```bash
39
+ niceeval sandbox prune
40
+ ```
41
+
42
+ - `orphan` 表示属主进程已确认死亡,可以安全销毁;正在跑的运行的实例不会出现在列表里。
43
+ - 从别的机器创建、无法核对的实例标为 `unverified`,默认不动;确认后用 `niceeval sandbox prune --force`。
44
+ - Vercel Sandbox 无法按元数据核对,到 Provider 的保留期限后自动回收,不需要处理。
45
+ - `--keep-sandbox` 留存的现场不受 `prune` 影响,仍用 `niceeval sandbox stop` 管理。
46
+
47
+ ## 补齐实验收尾
48
+
49
+ 实验 `setup` 起的外部服务要靠实验 `teardown` 关掉。被强杀的运行如果你暂时不想重跑,单独补一次收尾:
50
+
51
+ ```bash
52
+ niceeval exp compare/bub-e2b --teardown
53
+ ```
54
+
55
+ - 只执行选中实验的 `teardown`,不跑任何评估、不跑 `setup`。
56
+ - 随时可以执行,不依赖上次运行留下的记录;`teardown` 自身要能容忍重复执行。
57
+ - 强杀后原进程的内存已经丢失,`teardown` 里要从容器名、pid 文件或幂等的关停脚本这类持久信息找回要关的资源,不要依赖 `setup` 存在内存里的对象。
58
+
59
+ ## 预防:让下一次强杀无害
60
+
61
+ - 长任务放在有时限的环境(CI、外部看门狗)里跑时,把时限内跑不完当成常态:靠上面的续跑机制分多次跑完,不必强求单次完成。
62
+ - 实验 `teardown` 写成幂等的:重复执行不报错、目标已经关掉也算成功。这是补收尾机制正确工作的前提。