niceeval 0.10.3-canary.2 → 0.10.3-canary.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (182) hide show
  1. package/INDEX.md +1 -0
  2. package/dist/agents/types.d.ts +12 -0
  3. package/dist/context/turn-errors.d.ts +49 -0
  4. package/dist/context/types.d.ts +32 -16
  5. package/dist/i18n/en.d.ts +3 -0
  6. package/dist/i18n/zh-CN.d.ts +3 -0
  7. package/dist/o11y/execution-tree.d.ts +11 -1
  8. package/dist/o11y/types.d.ts +20 -3
  9. package/dist/report/components/entity-lists/AttemptList.d.ts +17 -0
  10. package/dist/report/components/entity-lists/AttemptList.js +23 -2
  11. package/dist/report/components/entity-lists/EvalList.js +0 -0
  12. package/dist/report/components/entity-lists/ExperimentList.js +20 -4
  13. package/dist/report/components/entity-lists/compute.js +10 -2
  14. package/dist/report/components/entity-lists/faces.js +51 -7
  15. package/dist/report/components/entity-lists/index.js +9 -0
  16. package/dist/report/components/fixtures.d.ts +1 -2
  17. package/dist/report/components/fixtures.js +12 -22
  18. package/dist/report/components/site-components/ScopeWarnings.js +1 -1
  19. package/dist/report/components/site-components/index.d.ts +3 -3
  20. package/dist/report/components/site-components/index.js +13 -26
  21. package/dist/report/components/site-components/scope-warnings.d.ts +7 -3
  22. package/dist/report/components/site-components/scope-warnings.js +16 -32
  23. package/dist/report/model/aggregate.d.ts +9 -1
  24. package/dist/report/model/aggregate.js +11 -2
  25. package/dist/report/model/format.d.ts +7 -0
  26. package/dist/report/model/format.js +12 -0
  27. package/dist/report/model/locale.d.ts +4 -10
  28. package/dist/report/model/locale.js +7 -20
  29. package/dist/report/model/types.d.ts +10 -2
  30. package/dist/results/select.d.ts +16 -11
  31. package/dist/results/select.js +87 -89
  32. package/dist/results/types.d.ts +45 -29
  33. package/dist/scoring/types.d.ts +1 -1
  34. package/docs-site/zh/reference/cli.mdx +1 -0
  35. package/docs-site/zh/reference/define-agent.mdx +16 -0
  36. package/docs-site/zh/reference/define-eval.mdx +4 -2
  37. package/docs-site/zh/reference/events.mdx +6 -0
  38. package/docs-site/zh/reference/report-components.mdx +6 -6
  39. package/docs-site/zh/reference/results-data.mdx +14 -11
  40. package/docs-site/zh/troubleshooting/debugging.mdx +1 -2
  41. package/docs-site/zh/tutorials/agent-feedback-loop.mdx +4 -10
  42. package/docs-site/zh/tutorials/custom-reports.mdx +2 -2
  43. package/docs-site/zh/tutorials/local-iteration.mdx +91 -0
  44. package/docs-site/zh/tutorials/sandbox-providers.mdx +14 -0
  45. package/docs-site/zh/tutorials/viewing-results.mdx +21 -20
  46. package/package.json +1 -1
  47. package/src/agents/index.ts +6 -0
  48. package/src/agents/types.ts +12 -0
  49. package/src/agents/ui-message-stream.ts +1 -1
  50. package/src/cli.ts +7 -0
  51. package/src/context/context.test.ts +63 -5
  52. package/src/context/context.ts +14 -6
  53. package/src/context/send-retry.test.ts +292 -0
  54. package/src/context/send-retry.ts +169 -0
  55. package/src/context/session.test.ts +105 -1
  56. package/src/context/session.ts +40 -2
  57. package/src/context/turn-errors.test.ts +160 -0
  58. package/src/context/turn-errors.ts +122 -0
  59. package/src/context/types.ts +32 -16
  60. package/src/define.test.ts +1 -1
  61. package/src/define.ts +2 -0
  62. package/src/expect/index.test.ts +1 -1
  63. package/src/i18n/en.ts +6 -1
  64. package/src/i18n/zh-CN.ts +6 -1
  65. package/src/o11y/cost.test.ts +1 -1
  66. package/src/o11y/derive.test.ts +78 -0
  67. package/src/o11y/derive.ts +10 -2
  68. package/src/o11y/execution-tree.test.ts +1 -1
  69. package/src/o11y/execution-tree.ts +19 -3
  70. package/src/o11y/types.ts +16 -3
  71. package/src/report/assets/styles.css +6 -6
  72. package/src/report/components/attempt-detail/attempt-components.test.tsx +12 -274
  73. package/src/report/components/attempt-detail/validate.test.ts +1 -1
  74. package/src/report/components/compute.test.ts +36 -52
  75. package/src/report/components/entity-lists/AttemptList.tsx +37 -1
  76. package/src/report/components/entity-lists/EvalList.tsx +0 -0
  77. package/src/report/components/entity-lists/ExperimentList.tsx +43 -2
  78. package/src/report/components/entity-lists/compute.ts +10 -1
  79. package/src/report/components/entity-lists/faces.ts +55 -10
  80. package/src/report/components/entity-lists/index.tsx +7 -0
  81. package/src/report/components/entity-lists/validate.test.ts +16 -1
  82. package/src/report/components/fixtures.ts +12 -24
  83. package/src/report/components/metric-views/chart-math.test.ts +1 -1
  84. package/src/report/components/metric-views/validate.test.ts +1 -1
  85. package/src/report/components/site-components/ScopeWarnings.tsx +1 -1
  86. package/src/report/components/site-components/index.tsx +11 -18
  87. package/src/report/components/site-components/scope-warnings.ts +16 -39
  88. package/src/report/components/site-components/site-components.test.tsx +116 -279
  89. package/src/report/components/site-components/validate.test.ts +8 -11
  90. package/src/report/components/summaries/validate.test.ts +1 -1
  91. package/src/report/definition/grid-layout.test.ts +1 -1
  92. package/src/report/definition/shell-head.test.ts +1 -1
  93. package/src/report/model/aggregate.ts +14 -3
  94. package/src/report/model/format.ts +14 -0
  95. package/src/report/model/locale.ts +9 -20
  96. package/src/report/model/types.ts +10 -2
  97. package/src/report/runtime/dual-render.test.tsx +104 -784
  98. package/src/report/runtime/host.test.ts +1 -1
  99. package/src/results/annotated-source.test.ts +1 -1
  100. package/src/results/attempt-evidence.test.ts +1 -1
  101. package/src/results/host-equivalence.test.ts +78 -243
  102. package/src/results/index.ts +1 -0
  103. package/src/results/locator.test.ts +1 -1
  104. package/src/results/open.ts +6 -3
  105. package/src/results/results.test.ts +104 -29
  106. package/src/results/select.ts +96 -90
  107. package/src/results/types.ts +46 -34
  108. package/src/runner/attempt.test.ts +1 -1
  109. package/src/runner/attempt.ts +11 -0
  110. package/src/runner/cleanup-timeout.test.ts +1 -1
  111. package/src/runner/discover.test.ts +1 -1
  112. package/src/runner/eval-selection.test.ts +1 -1
  113. package/src/runner/eval-source.test.ts +1 -1
  114. package/src/runner/experiment-cleanup-registry.test.ts +1 -1
  115. package/src/runner/experiment-labels.test.ts +1 -1
  116. package/src/runner/feedback/ci.test.ts +8 -528
  117. package/src/runner/feedback/coordinator.test.ts +1 -1
  118. package/src/runner/feedback/profile.test.ts +1 -1
  119. package/src/runner/feedback/reducer.test.ts +1 -1
  120. package/src/runner/ledger.test.ts +1 -1
  121. package/src/runner/report.test.ts +6 -8
  122. package/src/runner/reporters/braintrust.test.ts +1 -1
  123. package/src/runner/reporters/json.test.ts +1 -1
  124. package/src/runner/run.test.ts +1 -1
  125. package/src/runner/run.ts +19 -1
  126. package/src/runner/sandbox-selection.test.ts +1 -1
  127. package/src/sandbox/checkpoint.test.ts +1 -1
  128. package/src/sandbox/e2b-agent-template.test.ts +1 -1
  129. package/src/sandbox/e2b-reconcile.test.ts +1 -1
  130. package/src/sandbox/io-retry.test.ts +1 -1
  131. package/src/sandbox/keep-registry.test.ts +1 -1
  132. package/src/sandbox/paths.test.ts +1 -1
  133. package/src/sandbox/retry.test.ts +1 -1
  134. package/src/scoring/display.test.ts +1 -1
  135. package/src/scoring/evidence.test.ts +1 -1
  136. package/src/scoring/judge.test.ts +2 -2
  137. package/src/scoring/judge.ts +4 -2
  138. package/src/scoring/scoped.test.ts +239 -0
  139. package/src/scoring/scoped.ts +84 -16
  140. package/src/scoring/types.ts +1 -1
  141. package/src/shared/aggregate.test.ts +1 -1
  142. package/src/show/command.test.ts +4 -2
  143. package/src/show/index.ts +4 -2
  144. package/src/show/render.ts +4 -2
  145. package/src/show/show.test.ts +34 -879
  146. package/src/util.test.ts +1 -1
  147. package/src/view/app/App.tsx +1 -1
  148. package/src/view/app/lib/attempt-dialog.test.ts +1 -1
  149. package/src/view/data.test.ts +23 -37
  150. package/src/view/data.ts +3 -1
  151. package/src/view/server.ts +1 -1
  152. package/src/view/site-head.test.ts +30 -62
  153. package/src/view/site.ts +1 -1
  154. package/src/view/view-report.test.ts +39 -378
  155. package/src/agents/ai-sdk-otel.test.ts +0 -23
  156. package/src/agents/ai-sdk.test.ts +0 -465
  157. package/src/agents/bub-install-spec.test.ts +0 -34
  158. package/src/agents/claude-code.test.ts +0 -315
  159. package/src/agents/codex.test.ts +0 -523
  160. package/src/agents/coding-cli-versions.test.ts +0 -15
  161. package/src/agents/langgraph.test.ts +0 -204
  162. package/src/agents/native-config.test.ts +0 -179
  163. package/src/agents/openai-compat.test.ts +0 -57
  164. package/src/agents/openclaw.test.ts +0 -31
  165. package/src/agents/plugin-config.test.ts +0 -95
  166. package/src/agents/sdk-streams.test.ts +0 -224
  167. package/src/agents/skills.test.ts +0 -215
  168. package/src/agents/streaming.test.ts +0 -146
  169. package/src/agents/ui-message-stream.test.ts +0 -254
  170. package/src/o11y/otlp/mappers/claude-code.test.ts +0 -32
  171. package/src/o11y/otlp/parse.test.ts +0 -128
  172. package/src/o11y/otlp/turn-otel.test.ts +0 -111
  173. package/src/o11y/parsers/bub.test.ts +0 -72
  174. package/src/o11y/parsers/claude-code.test.ts +0 -142
  175. package/src/o11y/parsers/openclaw.test.ts +0 -154
  176. package/src/o11y/tool-names.test.ts +0 -62
  177. package/src/report/components/render.test.tsx +0 -472
  178. package/src/runner/feedback/agent.test.ts +0 -536
  179. package/src/runner/feedback/human.test.ts +0 -723
  180. package/src/view/app/App.test.tsx +0 -131
  181. package/src/view/artifact-serving.test.ts +0 -141
  182. package/src/view/site-parity.test.ts +0 -140
package/INDEX.md CHANGED
@@ -26,6 +26,7 @@
26
26
  - `docs-site/zh/tutorials/dataset-fanout.mdx` — 数据驱动测试(dataset fan-out):用多份数据运行同一套评估用例:从 .eval.ts 文件导出数组或 keyed record,将一套评估逻辑展开为多个 case。用 loadYaml 或 loadJson 读取外部数据集,并获得稳定 ID。
27
27
  - `docs-site/zh/tutorials/experiments.mdx` — 实验矩阵:用运行矩阵比较 agents 和 models:使用 NiceEval experiments 让同一批评估用例横跨多个 agents、models 和 flags,比较 pass rate、成本和延迟。
28
28
  - `docs-site/zh/tutorials/fixtures.mdx` — Sandbox Fixture:用任务评估 coding agents:用 .eval.ts 给 coding agent 准备隔离 workspace、发送真实任务,并用 Sandbox 文件、命令、diff 和 judge 验证结果。
29
+ - `docs-site/zh/tutorials/local-iteration.mdx` — 本地批跑提速:复用一个热 Sandbox:用 --reuse-sandbox 让一批评估共用一个装好的 Sandbox 串行跑,把重复安装折成一次;并按生命周期把安装写对层,让每道题几乎立即开跑。
29
30
  - `docs-site/zh/tutorials/publish-report.mdx` — 通过 CI 发布报告:把经过 copySnapshots 大小预检的结果目录提交进仓库,CI 用一行 view --results 导出报告站;超大文件在 commit 前就会得到可执行错误。
30
31
  - `docs-site/zh/tutorials/quickstart.mdx` — 为你的 Agent 项目设置评估:安装 NiceEval,写三个文件,10 分钟内对你自己的应用跑通第一条评估用例。
31
32
  - `docs-site/zh/tutorials/reporters.mdx` — 把结果上报到 Braintrust 与其它目的地:用内置 reporters 把评估用例结果送到 Braintrust 实验、JUnit XML 或自定义目的地。
@@ -1,6 +1,7 @@
1
1
  import type { DiagnosticInput, ProgressUpdate } from "../shared/types.ts";
2
2
  import type { StreamEvent, TraceSpan, Usage } from "../o11y/types.ts";
3
3
  import type { Sandbox } from "../sandbox/types.ts";
4
+ import type { TurnErrorClassifier } from "../context/turn-errors.ts";
4
5
  /**
5
6
  * 本地 stdio 形态的 MCP server:沙箱内起子进程,按 stdio 说 MCP 协议。
6
7
  * 与 {@link McpHttpServer} 按形状判别(有 `command` 的是 stdio,有 `url` 的是 HTTP)。
@@ -360,6 +361,13 @@ export interface Agent {
360
361
  /** 原生 span → canonical 的薄 mapper;省略走通用 heuristic。只影响瀑布图。 */
361
362
  spanMapper?: SpanMapper;
362
363
  send(input: TurnInput, ctx: AgentContext): Promise<Turn>;
364
+ /**
365
+ * 可选 turn 失败分类器:按重试安全性归类一次 send 失败(抛出或返回 `status: "failed"` 的
366
+ * Turn),返回 `undefined` 回落保守兜底。分类器只声明决策与诊断词,不影响重试策略(次数、
367
+ * 退避对所有 agent 一致);抛错按不可重试处理并被吞掉。形状与分类链、执行体时序见
368
+ * docs/feature/error-classification/architecture.md。
369
+ */
370
+ classifyTurnError?: TurnErrorClassifier;
363
371
  teardown?: AgentTeardown;
364
372
  }
365
373
  /** `defineSandboxAgent()` 的入参形状(见 src/define.ts)——`kind: "sandbox"` 由 define 固定填入,不由用户声明。 */
@@ -380,6 +388,8 @@ export interface SandboxAgentDef {
380
388
  spanMapper?: SpanMapper;
381
389
  /** 每轮一次:跑 prompt(fresh / resume)+ 解析成 events。 */
382
390
  send(input: TurnInput, ctx: AgentContext): Promise<Turn>;
391
+ /** 可选 turn 失败分类器:见 `Agent.classifyTurnError`。 */
392
+ classifyTurnError?: TurnErrorClassifier;
383
393
  /** Sandbox 销毁前的清理,当且仅当本 attempt 走到过 `setup` 时点才执行(`setup` 抛错不豁免),
384
394
  * 在 finally 里跑一次。 */
385
395
  teardown?: AgentTeardown;
@@ -402,6 +412,8 @@ export interface RemoteAgentDef {
402
412
  spanMapper?: SpanMapper;
403
413
  /** 每轮一次:把一轮 prompt 发给远程被测对象(HTTP/SDK 等),解析响应成 events。 */
404
414
  send(input: TurnInput, ctx: AgentContext): Promise<Turn>;
415
+ /** 可选 turn 失败分类器:见 `Agent.classifyTurnError`。 */
416
+ classifyTurnError?: TurnErrorClassifier;
405
417
  /** 运行结束前的清理,当且仅当本 attempt 走到过 `setup` 时点才执行(`setup` 抛错不豁免),
406
418
  * 在 finally 里跑一次。 */
407
419
  teardown?: AgentTeardown;
@@ -0,0 +1,49 @@
1
+ import type { Turn } from "../types.ts";
2
+ /**
3
+ * 一次 send 失败的分类结果:`retryable` 是执行体唯一消费的决策轴;`reason` 是开放词表的
4
+ * 细分诊断,只进 activity 与耗尽摘要,不参与策略。内建兜底产出 reason `"rate_limit"` /
5
+ * `"network"`;adapter 分类器可自造词。`retryable: true` 时 `reason` 必填——可重试的失败
6
+ * 一定会出现在 activity 行与可能的耗尽摘要里,那里需要一个给人读的词。
7
+ */
8
+ export type TurnErrorClass = {
9
+ readonly retryable: true;
10
+ readonly reason: string;
11
+ } | {
12
+ readonly retryable: false;
13
+ readonly reason?: string;
14
+ };
15
+ /** 一次 send 失败的两种浮出形态:`send()` 抛出异常,或返回 `status: "failed"` 的 Turn。 */
16
+ export type TurnFailure = {
17
+ readonly type: "thrown";
18
+ readonly error: unknown;
19
+ } | {
20
+ readonly type: "turn-failed";
21
+ readonly turn: Turn;
22
+ };
23
+ /**
24
+ * adapter 可选分类器:返回 `undefined` 表示「不认识,交给保守兜底」。分类器必须快、纯、
25
+ * 不抛错——执行体按「抛错等价于不可重试」处理,自身错误被吞掉,不会掩盖原始失败。
26
+ */
27
+ export type TurnErrorClassifier = (failure: TurnFailure) => TurnErrorClass | undefined;
28
+ /**
29
+ * 失败 Turn 的错误摘要:取 `events` 里最后一个 `type: "error"` 事件的 message。
30
+ * 与 `context.turnFailed` 报错文案、保守兜底分类器读的同一段文本同源——不出现
31
+ * 「报错说 A、分类看 B」。没有 error 事件(status: "failed" 但 adapter 没吐错误事件)时
32
+ * 返回 `undefined`。
33
+ */
34
+ export declare function turnErrorText(turn: Turn): string | undefined;
35
+ /** 两种 `TurnFailure` 形态统一取「给人读也给分类器看」的那段文本。 */
36
+ export declare function turnFailureText(failure: TurnFailure): string;
37
+ /**
38
+ * 保守兜底分类器:三道分类链里的第二道。对失败文本做正则匹配,认不出的一律 `{ retryable: false }`
39
+ * ——宁可判死一个 attempt,不产出不可信的 verdict(判据见 README「分类」)。
40
+ */
41
+ export declare function classifyTurnError(failure: TurnFailure): TurnErrorClass;
42
+ /** 受理证据门:失败 Turn 的 events 里已出现任何 agent 侧产出,即证明 agent 已受理并开始工作。 */
43
+ export declare function hasAgentEvidence(turn: Turn): boolean;
44
+ /**
45
+ * 三道分类链的完整决议:adapter 分类器(可选,抛错按不可重试处理并吞掉)→ 保守兜底 →
46
+ * 受理证据门(否决权,失败 Turn 带 agent 产出事件时强制降级)。执行体只需要调这一个函数,
47
+ * 不必自己拼三道链的顺序。
48
+ */
49
+ export declare function resolveTurnErrorClass(failure: TurnFailure, adapterClassifier?: TurnErrorClassifier): TurnErrorClass;
@@ -94,26 +94,40 @@ export interface DiffView {
94
94
  /** 正则是否命中 diff 里任意文件的路径或内容。 */
95
95
  matches(re: RegExp): boolean;
96
96
  }
97
- /** 工具匹配小语言。 */
97
+ /**
98
+ * 工具匹配小语言。一条调用的全部可断面——入参、次数、输出、状态——都在这一个对象里表达:
99
+ * `input` / `output` / `status` 之间是 AND,且作用在同一笔调用上;`count` 数的是满足这些
100
+ * 条件的调用笔数,不存在「一笔满足 input、另一笔满足 output」也算命中的读法。
101
+ */
98
102
  export interface ToolMatch {
99
103
  /**
100
- * 只匹配入参包含这些键值的调用:**深度部分匹配**——嵌套对象逐键下钻,数组按值比较;
101
- * 值可以是 RegExp(对字符串字段测试,不命中时再对整个 input 的序列化串兜底测一次)
102
- * 或谓词函数。不要求深度相等,多余的入参键不影响命中。
104
+ * 入参匹配:对象做**深度部分匹配**(写出的键值要求出现且相等,未写的忽略,嵌套递归比较;
105
+ * 值位置可以放 RegExp 匹配该字段的字符串值,不命中时再对整个 input 的序列化串兜底测一次,
106
+ * 或放谓词函数拿该字段原始值判断);顶层直接给 RegExp 则匹配序列化后的**完整输入**;
107
+ * 顶层给谓词函数 `(input) => boolean` 拿原始输入值自行判断。三种顶层形态互斥,不会退化
108
+ * 成深比对——RegExp / 函数不是"键值对象",不会被当成 plain object 逐键枚举。
103
109
  */
104
- input?: Record<string, unknown>;
105
- /** 精确匹配调用次数,省略则只要求「至少一次」。 */
106
- count?: number;
107
- /** 只匹配处于该状态的调用(如 HITL 场景下的 rejected)。 */
108
- status?: "completed" | "failed" | "rejected";
110
+ input?: Record<string, unknown> | RegExp | ((input: unknown) => boolean);
111
+ /** 数字精确匹配调用次数;谓词对命中次数自行判定(如 `(n) => n >= 2`);省略则只要求「至少一次」。 */
112
+ count?: number | ((n: number) => boolean);
113
+ /**
114
+ * 输出匹配,值语义同 `input` 的值位置:RegExp 对字符串输出测试(非字符串先序列化再测);
115
+ * 谓词函数拿原始输出自行判断;对象做深度部分匹配;其余值严格相等。
116
+ */
117
+ output?: unknown;
118
+ /** 只匹配处于该状态的调用。`pending` 是已发起、尚无结果的调用——典型是 HITL 停在审批上的那一笔。 */
119
+ status?: "pending" | "completed" | "failed" | "rejected";
109
120
  }
110
121
  /** calledSubagent 的匹配小语言,语义同 ToolMatch。 */
111
122
  export interface SubagentMatch {
112
- /** 精确匹配调用次数,省略则只要求「至少一次」。 */
113
- count?: number;
114
- status?: "completed" | "failed";
115
- /** 只匹配指向该远程地址的子 agent 调用。 */
116
- remoteUrl?: string | RegExp;
123
+ /** 数字精确匹配调用次数;谓词对命中次数自行判定;省略则只要求「至少一次」。 */
124
+ count?: number | ((n: number) => boolean);
125
+ /** 子 agent 委派没有 rejected 状态(subagent.completed 只报 completed / failed)。 */
126
+ status?: "pending" | "completed" | "failed";
127
+ /** 只匹配指向该远程地址的子 agent 调用:字符串精确匹配、RegExp 测试、或谓词函数自行判断。 */
128
+ remoteUrl?: string | RegExp | ((url: string) => boolean);
129
+ /** 匹配子 agent 的返回,值语义同 ToolMatch.output。 */
130
+ output?: unknown;
117
131
  }
118
132
  /** requireInputRequest 的过滤条件;多个字段之间是 AND 关系。 */
119
133
  export interface InputRequestFilter {
@@ -252,8 +266,10 @@ export interface TestContext {
252
266
  /** 取默认会话里等待中的 HITL 输入请求;不传 filter 要求恰好一条,拿不到就抛。 */
253
267
  requireInputRequest(filter?: InputRequestFilter): InputRequest;
254
268
  /**
255
- * 回答默认会话里等待中的输入请求,返回续接的 TurnHandle。字符串形式按顺序对应各请求;
256
- * 多个请求并停、需要指名回答哪一条时用 RespondAnswer 对象形式(见其类型注释)。
269
+ * 回答默认会话里等待中的输入请求,返回续接的 TurnHandle。字符串形式只在恰好一条待处理请求时
270
+ * 才能自动对位——命中该请求 `options` 里的某个 id 就是 `optionId`,否则整句落自由文本;
271
+ * 多个请求并停时字符串形式无法消歧,直接抛 `hitl.stringAmbiguous`,要求改用
272
+ * `{ request, optionId }` / `{ request, text }` 对象形式(RespondAnswer,见其类型注释)显式指名。
257
273
  */
258
274
  respond(...responses: (string | RespondAnswer)[]): Promise<TurnHandle>;
259
275
  /** 用同一个 optionId 批量回答默认会话里全部等待中的输入请求。 */
package/dist/i18n/en.d.ts CHANGED
@@ -238,6 +238,9 @@ export declare const en: {
238
238
  "session.tools": string;
239
239
  "session.turn.primary": string;
240
240
  "session.turn.secondary": string;
241
+ "session.turnRetry": string;
242
+ "session.turnRetryBudgetExhausted": string;
243
+ "session.turnRetrySendExhausted": string;
241
244
  "util.requiredEnv": string;
242
245
  "vercel.fileNotFound": string;
243
246
  "vercel.rotateFailed": string;
@@ -238,6 +238,9 @@ export declare const zhCN: {
238
238
  readonly "session.tools": "{{count}} 工具";
239
239
  readonly "session.turn.primary": "第{{turn}}轮";
240
240
  readonly "session.turn.secondary": "会话{{session}}·第{{turn}}轮";
241
+ readonly "session.turnRetry": "turn 重试 {{attempt}}/{{maxAttempts}}({{reason}})——等待 {{seconds}}s";
242
+ readonly "session.turnRetryBudgetExhausted": " · attempt 重试预算已耗尽({{maxRetries}} 次重试,{{reason}})";
243
+ readonly "session.turnRetrySendExhausted": " · 重试已耗尽({{maxAttempts}} 次尝试,{{reason}})";
241
244
  readonly "util.requiredEnv": "缺少必需的环境变量 {{name}}(请在 .env 里配置)。";
242
245
  readonly "vercel.fileNotFound": "File not found: {{path}}";
243
246
  readonly "vercel.rotateFailed": "[VercelSandbox] session rotate failed ({{seconds}}s): {{error}}";
@@ -18,6 +18,16 @@ export interface ExecutionThinkingNode extends ExecutionNodeBase {
18
18
  kind: "thinking";
19
19
  text: string;
20
20
  }
21
+ /**
22
+ * 被测系统内部机制注入进上下文的文本,不属于任何一方"说的话",不并进 `message`
23
+ * (见 docs/feature/adapters/architecture/events.md「不变量 9」)。与 thinking / compaction
24
+ * 同一档次的直通节点,不参与 callId 关联。
25
+ */
26
+ export interface ExecutionContextInjectedNode extends ExecutionNodeBase {
27
+ kind: "context.injected";
28
+ text: string;
29
+ source?: string;
30
+ }
21
31
  /** Skill 加载节点——一等,直接来自 StreamEvent 的 "skill.loaded",不靠工具名/文本猜。 */
22
32
  export interface ExecutionSkillNode extends ExecutionNodeBase {
23
33
  kind: "skill.loaded";
@@ -79,7 +89,7 @@ export interface ExecutionTelemetryNode {
79
89
  id: string;
80
90
  span: TraceSpan;
81
91
  }
82
- export type ExecutionNode = ExecutionMessageNode | ExecutionThinkingNode | ExecutionSkillNode | ExecutionActionNode | ExecutionSubagentNode | ExecutionInputRequestedNode | ExecutionCompactionNode | ExecutionErrorNode | ExecutionTelemetryNode;
92
+ export type ExecutionNode = ExecutionMessageNode | ExecutionThinkingNode | ExecutionContextInjectedNode | ExecutionSkillNode | ExecutionActionNode | ExecutionSubagentNode | ExecutionInputRequestedNode | ExecutionCompactionNode | ExecutionErrorNode | ExecutionTelemetryNode;
83
93
  export interface ExecutionTree {
84
94
  /**
85
95
  * 骨架节点在前,顺序 = 事件出现顺序;telemetry-only 节点(未能唯一关联的 span)
@@ -110,6 +110,16 @@ export type StreamEvent = {
110
110
  type: "thinking";
111
111
  text: string;
112
112
  }
113
+ /**
114
+ * 被测系统内部注入的、不披着 `message` 外衣的上下文文本(如 Claude Code 的 SessionStart /
115
+ * UserPromptSubmit hook 在下一轮开始前前置进模型上下文的文本)。只承载带实际文本内容的注入;
116
+ * `source` 是可选的原始来源标记(如 hook 名),adapter 按各自协议原样透传,不强行归一到封闭枚举。
117
+ */
118
+ | {
119
+ type: "context.injected";
120
+ text: string;
121
+ source?: string;
122
+ }
113
123
  /** 上下文被压缩/摘要(如超长会话截断历史);`reason` 是可选的压缩原因说明。 */
114
124
  | {
115
125
  type: "compaction";
@@ -120,21 +130,26 @@ export type StreamEvent = {
120
130
  type: "error";
121
131
  message: string;
122
132
  });
123
- /** core 从事件流折叠出的结构化事实(deriveRunFacts)。 */
133
+ /**
134
+ * core 从事件流折叠出的结构化事实(deriveRunFacts)。折叠按 callId 把 called 与 result 对成一条
135
+ * 调用:配上 result 的取 result 的状态;只有 called、尚未等到 result 的调用状态是 `pending`——
136
+ * HITL 停在审批上的调用就以这个状态被断言,不是容错分支(见 docs/feature/adapters/architecture/events.md)。
137
+ */
124
138
  export interface ToolCall {
125
139
  callId: string;
126
140
  name: ToolName;
127
141
  originalName?: string;
128
142
  input: JsonValue;
129
143
  output?: JsonValue;
130
- status: "completed" | "failed" | "rejected";
144
+ status: "pending" | "completed" | "failed" | "rejected";
131
145
  }
132
146
  export interface SubagentCall {
133
147
  callId: string;
134
148
  name: string;
135
149
  remoteUrl?: string;
136
150
  output?: JsonValue;
137
- status: "completed" | "failed";
151
+ /** 子 agent 委派没有 rejected 状态(subagent.completed 只报 completed / failed)。 */
152
+ status: "pending" | "completed" | "failed";
138
153
  }
139
154
  export interface DerivedFacts {
140
155
  readonly toolCalls: readonly ToolCall[];
@@ -143,6 +158,8 @@ export interface DerivedFacts {
143
158
  readonly parked: boolean;
144
159
  readonly messageCount: number;
145
160
  readonly compactions: number;
161
+ /** 事件流里 `context.injected` 事件的次数;只回答存在性问题,不替代逐条读取原文。 */
162
+ readonly contextInjections: number;
146
163
  }
147
164
  /**
148
165
  * span 的【语义角色】,从 OTel GenAI 语义约定的 gen_ai.operation.name 归一而来
@@ -2,6 +2,23 @@ import type { ReactElement } from "react";
2
2
  import type { AttemptListItem } from "../../model/types.ts";
3
3
  import type { AttemptLocator } from "../../../results/locator.ts";
4
4
  import { type ReportLocale } from "../../model/locale.ts";
5
+ /**
6
+ * 时效标注(`↩` + 紧凑时距):历史执行(携带,或跨快照拼入)才渲染,新执行不标——subdued
7
+ * 行内事实,不占框、不用警示色(docs/feature/reports/library/entity-lists.md「时效标注」)。
8
+ * web 面 hover 显示完整执行时刻;ExperimentList / EvalList / AttemptList 三处共用。
9
+ */
10
+ export declare function HistoricalMark({ item, locale, }: {
11
+ item: Pick<AttemptListItem, "startedAt" | "historical">;
12
+ locale?: ReportLocale;
13
+ }): ReactElement | null;
14
+ /**
15
+ * Eval 父行的时效标注:全部 attempt 均为历史执行时,标最近一次执行(startedAt 最大)的时距;
16
+ * 新旧混合时父行不标,子行各自可见(docs/feature/reports/library/entity-lists.md「时效标注」)。
17
+ * EvalList 与 ExperimentList 的 Eval 父行共用。
18
+ */
19
+ export declare function EvalHistoricalMark({ attempts, }: {
20
+ attempts: readonly Pick<AttemptListItem, "startedAt" | "historical">[];
21
+ }): ReactElement | null;
5
22
  /**
6
23
  * locator + 判定符,AttemptList/EvalList/ExperimentList 共用。没有 target(当前报告没有
7
24
  * declare attempt-input page,也没有显式 attemptHref)时是纯文本,不生成空 href 或假链接
@@ -1,8 +1,29 @@
1
1
  import { Fragment as _Fragment, jsx as _jsx, jsxs as _jsxs } from "react/jsx-runtime";
2
2
  import { DEFAULT_REPORT_LOCALE, countText, localeText } from "../../model/locale.js";
3
3
  import { colorClassForKey } from "../../assets/colors.js";
4
- import { formatDurationMs, formatUSD, verdictMark } from "../../model/format.js";
4
+ import { formatDurationMs, formatHistoricalGap, formatReportDateTime, formatUSD, verdictMark } from "../../model/format.js";
5
5
  import { cx } from "../shared.js";
6
+ /**
7
+ * 时效标注(`↩` + 紧凑时距):历史执行(携带,或跨快照拼入)才渲染,新执行不标——subdued
8
+ * 行内事实,不占框、不用警示色(docs/feature/reports/library/entity-lists.md「时效标注」)。
9
+ * web 面 hover 显示完整执行时刻;ExperimentList / EvalList / AttemptList 三处共用。
10
+ */
11
+ export function HistoricalMark({ item, locale = DEFAULT_REPORT_LOCALE, }) {
12
+ if (!item.historical)
13
+ return null;
14
+ return (_jsxs("span", { className: "nre-historical", title: formatReportDateTime(item.startedAt, locale), children: ["\u21A9 ", formatHistoricalGap(item.startedAt)] }));
15
+ }
16
+ /**
17
+ * Eval 父行的时效标注:全部 attempt 均为历史执行时,标最近一次执行(startedAt 最大)的时距;
18
+ * 新旧混合时父行不标,子行各自可见(docs/feature/reports/library/entity-lists.md「时效标注」)。
19
+ * EvalList 与 ExperimentList 的 Eval 父行共用。
20
+ */
21
+ export function EvalHistoricalMark({ attempts, }) {
22
+ if (attempts.length === 0 || !attempts.every((a) => a.historical))
23
+ return null;
24
+ const mostRecent = attempts.reduce((a, b) => (b.startedAt > a.startedAt ? b : a));
25
+ return _jsxs("span", { className: "nre-historical", children: ["\u21A9 ", formatHistoricalGap(mostRecent.startedAt)] });
26
+ }
6
27
  /**
7
28
  * locator + 判定符,AttemptList/EvalList/ExperimentList 共用。没有 target(当前报告没有
8
29
  * declare attempt-input page,也没有显式 attemptHref)时是纯文本,不生成空 href 或假链接
@@ -24,7 +45,7 @@ export function failureSummaryText(item, locale) {
24
45
  /** 一条 Attempt 的比较卡片;完整 assertions 通过 locator 下钻,不在列表内展开。 */
25
46
  export function AttemptRow({ item, attemptHref, locale = DEFAULT_REPORT_LOCALE, }) {
26
47
  const reason = failureSummaryText(item, locale);
27
- return (_jsxs("li", { className: cx("nre-attempt", `nre-attempt-${item.verdict}`), "data-nre-verdict": item.verdict, children: [_jsxs("div", { className: "nre-attempt-head", children: [_jsx(AttemptLocatorBadge, { item: item, attemptHref: attemptHref }), _jsx("span", { className: "nre-attempt-eval", children: item.evalId }), _jsx("span", { className: "nre-attempt-experiment", children: item.experimentId }), _jsx("span", { className: cx("nre-attempt-agent", "nre-key", colorClassForKey(item.agent)), children: item.agent }), _jsx("span", { className: "nre-attempt-duration", children: formatDurationMs(item.durationMs) }), item.costUSD !== null && _jsx("span", { className: "nre-attempt-cost", children: formatUSD(item.costUSD) })] }), reason && _jsx("p", { className: "nre-attempt-result", children: reason })] }));
48
+ return (_jsxs("li", { className: cx("nre-attempt", `nre-attempt-${item.verdict}`), "data-nre-verdict": item.verdict, children: [_jsxs("div", { className: "nre-attempt-head", children: [_jsx(AttemptLocatorBadge, { item: item, attemptHref: attemptHref }), _jsx(HistoricalMark, { item: item, locale: locale }), _jsx("span", { className: "nre-attempt-eval", children: item.evalId }), _jsx("span", { className: "nre-attempt-experiment", children: item.experimentId }), _jsx("span", { className: cx("nre-attempt-agent", "nre-key", colorClassForKey(item.agent)), children: item.agent }), _jsx("span", { className: "nre-attempt-duration", children: formatDurationMs(item.durationMs) }), item.costUSD !== null && _jsx("span", { className: "nre-attempt-cost", children: formatUSD(item.costUSD) })] }), reason && _jsx("p", { className: "nre-attempt-result", children: reason })] }));
28
49
  }
29
50
  export function AttemptList({ data, total, filter = false, attemptHref, className, locale = DEFAULT_REPORT_LOCALE, }) {
30
51
  const remaining = (total ?? data.length) - data.length;
@@ -1,7 +1,7 @@
1
1
  import { Fragment as _Fragment, jsx as _jsx, jsxs as _jsxs } from "react/jsx-runtime";
2
2
  import { shortestUniqueLabels } from "../../model/format.js";
3
3
  import { DEFAULT_REPORT_LOCALE, localeText } from "../../model/locale.js";
4
- import { AttemptLocatorBadge, failureSummaryText } from "./AttemptList.js";
4
+ import { AttemptLocatorBadge, EvalHistoricalMark, HistoricalMark, failureSummaryText } from "./AttemptList.js";
5
5
  import { MetricCellView } from "../cell.js";
6
6
  import { colorClassForKey } from "../../assets/colors.js";
7
7
  import { formatDurationMs, formatUSD, verdictMark } from "../../model/format.js";
@@ -35,12 +35,21 @@ function VerdictSummary({ item, locale }) {
35
35
  }
36
36
  function ExperimentAttemptRow({ attempt, last, attemptHref, locale, }) {
37
37
  const reason = failureSummaryText(attempt, locale);
38
- return (_jsxs("li", { className: cx("nre-experiment-attempt-row", `nre-eval-${attempt.verdict}`), children: [_jsx("span", { className: "nre-attempt-branch", "aria-hidden": "true", children: last ? "└─" : "├─" }), _jsx("span", { className: "nre-eval-attempt-badges", children: _jsx(AttemptLocatorBadge, { item: attempt, attemptHref: attemptHref }) }), _jsxs("span", { className: "nre-eval-attempt-metrics", children: [formatDurationMs(attempt.durationMs), attempt.costUSD !== null && _jsxs(_Fragment, { children: [" \u00B7 ", formatUSD(attempt.costUSD)] })] }), _jsx("span", { className: "nre-eval-reason", children: reason ?? "—" })] }));
38
+ return (_jsxs("li", { className: cx("nre-experiment-attempt-row", `nre-eval-${attempt.verdict}`), children: [_jsx("span", { className: "nre-attempt-branch", "aria-hidden": "true", children: last ? "└─" : "├─" }), _jsxs("span", { className: "nre-eval-attempt-badges", children: [_jsx(AttemptLocatorBadge, { item: attempt, attemptHref: attemptHref }), _jsx(HistoricalMark, { item: attempt, locale: locale })] }), _jsxs("span", { className: "nre-eval-attempt-metrics", children: [formatDurationMs(attempt.durationMs), attempt.costUSD !== null && _jsxs(_Fragment, { children: [" \u00B7 ", formatUSD(attempt.costUSD)] })] }), _jsx("span", { className: "nre-eval-reason", children: reason ?? "—" })] }));
39
39
  }
40
40
  function EvalAttempts({ row, attemptHref, locale, }) {
41
41
  const duration = row.durationMs.value === null ? localeText(locale, "cell.missing") : formatDurationMs(row.durationMs.value);
42
42
  const cost = row.costUSD.value === null ? localeText(locale, "cell.missing") : formatUSD(row.costUSD.value);
43
- return (_jsxs("li", { className: "nre-experiment-eval", children: [_jsxs("div", { className: cx("nre-experiment-eval-header", `nre-eval-${row.verdict}`), children: [_jsx("span", { className: cx("nre-eval-verdict", `nre-verdict-${row.verdict}`), children: verdictMark(row.verdict) }), _jsx("span", { className: "nre-eval-id", children: row.evalId }), _jsx("span", { className: "nre-eval-attempt-count", children: localeText(locale, "overview.attemptsCount", { n: row.attempts.length }) }), _jsxs("span", { className: "nre-eval-rollup", children: [localeText(locale, "entityList.average", { value: duration }), " · ", localeText(locale, "entityList.average", { value: cost })] })] }), _jsx("ul", { className: "nre-experiment-attempts", children: row.attempts.map((attempt, index) => (_jsx(ExperimentAttemptRow, { attempt: attempt, last: index === row.attempts.length - 1, attemptHref: attemptHref, locale: locale }, attempt.locator))) })] }));
43
+ return (_jsxs("li", { className: "nre-experiment-eval", children: [_jsxs("div", { className: cx("nre-experiment-eval-header", `nre-eval-${row.verdict}`), children: [_jsx("span", { className: cx("nre-eval-verdict", `nre-verdict-${row.verdict}`), children: verdictMark(row.verdict) }), _jsx("span", { className: "nre-eval-id", children: row.evalId }), _jsx(EvalHistoricalMark, { attempts: row.attempts }), _jsx("span", { className: "nre-eval-attempt-count", children: localeText(locale, "overview.attemptsCount", { n: row.attempts.length }) }), _jsxs("span", { className: "nre-eval-rollup", children: [localeText(locale, "entityList.average", { value: duration }), " · ", localeText(locale, "entityList.average", { value: cost })] })] }), _jsx("ul", { className: "nre-experiment-attempts", children: row.attempts.map((attempt, index) => (_jsx(ExperimentAttemptRow, { attempt: attempt, last: index === row.attempts.length - 1, attemptHref: attemptHref, locale: locale }, attempt.locator))) })] }));
44
+ }
45
+ /**
46
+ * 覆盖缺口的占位行:状态列为 —,结果列为「当前配置下无结果」+ 可复制的补跑命令,无 attempt
47
+ * 子行,不参与任何指标聚合——只把分母缺口摆进读者正在看的表里
48
+ * (docs/feature/reports/library/entity-lists.md「ExperimentList」)。
49
+ */
50
+ function MissingEvalRow({ evalId, experimentId, locale, }) {
51
+ const command = `niceeval exp ${experimentId}`;
52
+ return (_jsx("li", { className: "nre-experiment-eval nre-experiment-eval-missing", children: _jsxs("div", { className: "nre-experiment-eval-header", children: [_jsx("span", { className: "nre-eval-verdict", children: "\u2014" }), _jsx("span", { className: "nre-eval-id", children: evalId }), _jsxs("span", { className: "nre-eval-rollup", children: [localeText(locale, "experimentList.noResultsForConfig"), " \u00B7 ", _jsx("code", { children: command })] })] }) }));
44
53
  }
45
54
  function Flags({ flags, locale }) {
46
55
  if (!flags || Object.keys(flags).length === 0)
@@ -48,7 +57,14 @@ function Flags({ flags, locale }) {
48
57
  return (_jsxs("div", { className: "nre-experiment-flags", children: [_jsx("span", { children: localeText(locale, "experimentList.flags") }), Object.entries(flags).map(([key, value]) => (_jsxs("b", { children: [key, "=", typeof value === "string" ? value : JSON.stringify(value)] }, key)))] }));
49
58
  }
50
59
  function ExperimentRow({ item, label, attemptHref, locale, }) {
51
- return (_jsxs("details", { className: "nre-experiment-entry", children: [_jsxs("summary", { className: "nre-experiment-summary", children: [_jsxs("span", { className: "nre-experiment-name", "data-sort-value": item.experimentId, children: [_jsx("b", { className: cx("nre-experiment-id", "nre-key"), children: label }), _jsxs("small", { children: [localeText(locale, "overview.evalsCount", { n: item.evals }), item.attempts > item.evals ? ` · ${localeText(locale, "overview.attemptsCount", { n: item.attempts })}` : "", ` · ${formatDate(item.lastRunAt, locale)}`] })] }), _jsx("span", { "data-sort-value": item.model ?? "", children: item.model ?? localeText(locale, "experimentList.defaultModel") }), _jsx("span", { "data-sort-value": item.agent, children: _jsx("span", { className: cx("nre-experiment-agent", "nre-key", colorClassForKey(item.agent)), children: item.agent }) }), _jsx("span", { className: "nre-num", "data-sort-value": item.durationMs.value ?? "", children: _jsx(MetricCellView, { cell: item.durationMs, locale: locale }) }), _jsx("span", { className: cx("nre-num", passRateTone(item.endToEndPassRate.value)), "data-sort-value": item.endToEndPassRate.value ?? "", children: _jsx(MetricCellView, { cell: item.endToEndPassRate, locale: locale }) }), _jsx("span", { className: "nre-num", "data-sort-value": item.tokens.value ?? "", children: _jsx(MetricCellView, { cell: item.tokens, locale: locale }) }), _jsx("span", { className: "nre-num", "data-sort-value": item.costUSD.value ?? "", children: _jsx(MetricCellView, { cell: item.costUSD, locale: locale }) }), _jsx("span", { "data-sort-value": item.evalVerdicts.passed, children: _jsx(VerdictSummary, { item: item, locale: locale }) })] }), _jsxs("div", { className: "nre-experiment-detail", children: [_jsx(Flags, { flags: item.flags, locale: locale }), _jsx("ul", { className: "nre-experiment-evals", children: item.evalRows.map((row) => (_jsx(EvalAttempts, { row: row, attemptHref: attemptHref, locale: locale }, row.evalId))) })] })] }));
60
+ return (_jsxs("details", { className: "nre-experiment-entry", children: [_jsxs("summary", { className: "nre-experiment-summary", children: [_jsxs("span", { className: "nre-experiment-name", "data-sort-value": item.experimentId, children: [_jsx("b", { className: cx("nre-experiment-id", "nre-key"), children: label }), _jsxs("small", { children: [item.missingEvalIds.length > 0
61
+ ? localeText(locale, "overview.evalsCountPartial", {
62
+ covered: item.evals,
63
+ total: item.evals + item.missingEvalIds.length,
64
+ })
65
+ : localeText(locale, "overview.evalsCount", { n: item.evals }), item.attempts > item.evals ? ` · ${localeText(locale, "overview.attemptsCount", { n: item.attempts })}` : "", item.historicalAttempts > 0
66
+ ? ` · ${localeText(locale, "experimentList.historicalAttempts", { n: item.historicalAttempts, m: item.attempts })}`
67
+ : "", ` · ${formatDate(item.lastRunAt, locale)}`] })] }), _jsx("span", { "data-sort-value": item.model ?? "", children: item.model ?? localeText(locale, "experimentList.defaultModel") }), _jsx("span", { "data-sort-value": item.agent, children: _jsx("span", { className: cx("nre-experiment-agent", "nre-key", colorClassForKey(item.agent)), children: item.agent }) }), _jsx("span", { className: "nre-num", "data-sort-value": item.durationMs.value ?? "", children: _jsx(MetricCellView, { cell: item.durationMs, locale: locale }) }), _jsx("span", { className: cx("nre-num", passRateTone(item.endToEndPassRate.value)), "data-sort-value": item.endToEndPassRate.value ?? "", children: _jsx(MetricCellView, { cell: item.endToEndPassRate, locale: locale }) }), _jsx("span", { className: "nre-num", "data-sort-value": item.tokens.value ?? "", children: _jsx(MetricCellView, { cell: item.tokens, locale: locale }) }), _jsx("span", { className: "nre-num", "data-sort-value": item.costUSD.value ?? "", children: _jsx(MetricCellView, { cell: item.costUSD, locale: locale }) }), _jsx("span", { "data-sort-value": item.evalVerdicts.passed, children: _jsx(VerdictSummary, { item: item, locale: locale }) })] }), _jsxs("div", { className: "nre-experiment-detail", children: [_jsx(Flags, { flags: item.flags, locale: locale }), _jsxs("ul", { className: "nre-experiment-evals", children: [item.evalRows.map((row) => (_jsx(EvalAttempts, { row: row, attemptHref: attemptHref, locale: locale }, row.evalId))), item.missingEvalIds.map((evalId) => (_jsx(MissingEvalRow, { evalId: evalId, experimentId: item.experimentId, locale: locale }, evalId)))] })] })] }));
52
68
  }
53
69
  export function ExperimentList({ data, attemptHref, filter = false, className, locale = DEFAULT_REPORT_LOCALE, }) {
54
70
  const experimentLabels = shortestUniqueLabels(data.map((item) => item.experimentId));
@@ -8,7 +8,7 @@
8
8
  // - core 中立:只认 Metric / Dimension 接口,不出现具体 agent 名的分支。
9
9
  import { comparabilityConfigOf, deepEqualJson } from "../../../results/select.js";
10
10
  import { foldEvalVerdict } from "../../../shared/verdict.js";
11
- import { collectItems, computeCell, evalIdOf, experimentIdOf, fullEvalKey, groupItems, locatorOf, resolveInput, } from "../../model/aggregate.js";
11
+ import { collectItems, computeCell, evalIdOf, experimentIdOf, fullEvalKey, groupItems, historicalOf, locatorOf, resolveInput, } from "../../model/aggregate.js";
12
12
  import { attemptCostUSD, costUSD, durationMs, endToEndPassRate, examScore, tokens } from "../../model/metrics.js";
13
13
  import { compactAssertionSummary, primaryAssertionSummary, summaryText } from "../../../scoring/display.js";
14
14
  import { selectedEvalsOnly, summarizeItems } from "../shared-compute.js";
@@ -52,6 +52,11 @@ async function attemptListItemOf(item) {
52
52
  examScore: await computeCell(examScore, [item]),
53
53
  durationMs: result.durationMs,
54
54
  costUSD: attemptCostUSD(result),
55
+ // 缺 startedAt(legacy / 第三方落盘)时退化到所属快照的 startedAt——时效标注宁可粗一档
56
+ // 时距,不留空字段(与 dedupeAttempts「缺才不去重」同一条「不伪造」纪律,这里伪造的只是
57
+ // 展示粒度,不影响身份判定)。
58
+ startedAt: result.startedAt ?? item.snapshot.startedAt,
59
+ historical: historicalOf(item),
55
60
  locator: locatorOf(item),
56
61
  };
57
62
  }
@@ -100,8 +105,9 @@ export async function evalListData(input) {
100
105
  * 看跨配置演化用 snapshot 维度或 MetricLine,不把两套配置拼成一行冒充单一配置。
101
106
  */
102
107
  export async function experimentListData(input) {
103
- const { snapshots: rawSnapshots } = resolveInput(input);
108
+ const { snapshots: rawSnapshots, coverage } = resolveInput(input);
104
109
  const snapshots = selectedEvalsOnly(rawSnapshots);
110
+ const coverageByExperiment = new Map(coverage.map((c) => [c.experimentId, c]));
105
111
  // 可比性配置单义检查:同一 experiment 的输入快照必须共享一套可比性配置。
106
112
  const configByExperiment = new Map();
107
113
  for (const snapshot of snapshots) {
@@ -151,6 +157,8 @@ export async function experimentListData(input) {
151
157
  tokens: await computeCell(tokens, group),
152
158
  evals: stats.evals,
153
159
  attempts: stats.attempts,
160
+ historicalAttempts: group.filter(historicalOf).length,
161
+ missingEvalIds: coverageByExperiment.get(experimentId)?.missingEvalIds ?? [],
154
162
  lastRunAt: stats.lastRunAt,
155
163
  evalRows,
156
164
  });
@@ -3,7 +3,7 @@
3
3
  // locator,中间不留空格)。ExperimentList / EvalList 逐 attempt 只列这一个标记 + 各自的
4
4
  // 摘要,不重复整段 niceeval show 命令;要看某个 attempt 的完整证据,agent 自己拼
5
5
  // `niceeval show <locator>`。零 react、零 IO、纯同步。
6
- import { fitFailureSummary, formatDurationMs, formatUSD, shortestUniqueLabels, verdictMark, } from "../../model/format.js";
6
+ import { fitFailureSummary, formatDurationMs, formatHistoricalGap, formatUSD, shortestUniqueLabels, verdictMark, } from "../../model/format.js";
7
7
  import { countText, localeText } from "../../model/locale.js";
8
8
  import { stringWidth, wrapDisplay } from "../../model/text-layout.js";
9
9
  import { renderTableText } from "../../definition/table-text.js";
@@ -16,6 +16,21 @@ import { cellText, missingText, verdictTallyText, MISSING_MARK } from "../shared
16
16
  function locatorBadge(item) {
17
17
  return `${item.locator}${verdictMark(item.verdict)}`;
18
18
  }
19
+ /**
20
+ * 时效标注(`↩` + 紧凑时距)的 text 面:历史执行(携带,或跨快照拼入)才输出,新执行为
21
+ * 空串;三面(ExperimentList / EvalList / AttemptList)共用
22
+ * (docs/feature/reports/library/entity-lists.md「时效标注」)。
23
+ */
24
+ function historicalSuffix(item) {
25
+ return item.historical ? ` ↩ ${formatHistoricalGap(item.startedAt)}` : "";
26
+ }
27
+ /** Eval 父行的时效标注:全部 attempt 均为历史执行时,标最近一次执行的时距;新旧混合不标。 */
28
+ function evalHistoricalSuffix(attempts) {
29
+ if (attempts.length === 0 || !attempts.every((a) => a.historical))
30
+ return "";
31
+ const mostRecent = attempts.reduce((a, b) => (b.startedAt > a.startedAt ? b : a));
32
+ return ` ↩ ${formatHistoricalGap(mostRecent.startedAt)}`;
33
+ }
19
34
  /**
20
35
  * failureSummary + moreFailures 的展示形态:摘要在计算侧已按 Scoring display 契约折好,
21
36
  * 这里只加 "+N more failures" 计数与宽度收口,不重算摘要。
@@ -55,7 +70,23 @@ function experimentSummaryTable(items, ctx, labels) {
55
70
  cost: cellText(item.costUSD, locale),
56
71
  },
57
72
  }));
58
- const metadata = items.flatMap((item) => wrapDisplay(`${labels.get(item.experimentId) ?? item.experimentId}: ${localeText(locale, "overview.evalsCount", { n: item.evals })} · ${localeText(locale, "overview.attemptsCount", { n: item.attempts })} · ${item.lastRunAt}`, Math.max(8, ctx.width - 2)).map((line) => ` ${line}`));
73
+ const metadata = items.flatMap((item) => {
74
+ const evalsText = item.missingEvalIds.length > 0
75
+ ? localeText(locale, "overview.evalsCountPartial", {
76
+ covered: item.evals,
77
+ total: item.evals + item.missingEvalIds.length,
78
+ })
79
+ : localeText(locale, "overview.evalsCount", { n: item.evals });
80
+ const parts = [
81
+ evalsText,
82
+ localeText(locale, "overview.attemptsCount", { n: item.attempts }),
83
+ ...(item.historicalAttempts > 0
84
+ ? [localeText(locale, "experimentList.historicalAttempts", { n: item.historicalAttempts, m: item.attempts })]
85
+ : []),
86
+ item.lastRunAt,
87
+ ];
88
+ return wrapDisplay(`${labels.get(item.experimentId) ?? item.experimentId}: ${parts.join(" · ")}`, Math.max(8, ctx.width - 2)).map((line) => ` ${line}`);
89
+ });
59
90
  return [renderTableText({ columns: columns, rows, locale }, ctx), metadata.join("\n")].join("\n");
60
91
  }
61
92
  function experimentDetailTable(item, ctx, label) {
@@ -80,7 +111,7 @@ function experimentDetailTable(item, ctx, label) {
80
111
  key: row.evalId,
81
112
  cells: {
82
113
  status: `${verdictMark(row.verdict)} ${localeText(locale, `verdict.${row.verdict}`)}`,
83
- entity: row.evalId,
114
+ entity: `${row.evalId}${evalHistoricalSuffix(row.attempts)}`,
84
115
  result: "",
85
116
  duration: localeText(locale, "entityList.average", { value: cellText(row.durationMs, locale) }),
86
117
  cost: localeText(locale, "entityList.average", { value: cellText(row.costUSD, locale) }),
@@ -90,7 +121,7 @@ function experimentDetailTable(item, ctx, label) {
90
121
  key: attempt.locator,
91
122
  cells: {
92
123
  status: ` ${verdictMark(attempt.verdict)}`,
93
- entity: `${index === row.attempts.length - 1 ? "└─" : "├─"} ${attempt.locator}`,
124
+ entity: `${index === row.attempts.length - 1 ? "└─" : "├─"} ${attempt.locator}${historicalSuffix(attempt)}`,
94
125
  result: attemptReasonText(attempt, locale, resultBudget) ?? MISSING_MARK,
95
126
  duration: attempt.verdict === "skipped" && attempt.durationMs === 0 ? null : formatDurationMs(attempt.durationMs),
96
127
  cost: attempt.costUSD === null ? null : formatUSD(attempt.costUSD),
@@ -98,6 +129,19 @@ function experimentDetailTable(item, ctx, label) {
98
129
  }));
99
130
  return [parent, ...attempts];
100
131
  });
132
+ // 覆盖缺口的占位行:状态列为 —,结果列为「当前配置下无结果」+ 可复制的补跑命令,
133
+ // 无 attempt 子行,duration/cost 留空(不参与任何指标聚合)。
134
+ const missingRows = item.missingEvalIds.map((evalId) => ({
135
+ key: evalId,
136
+ cells: {
137
+ status: MISSING_MARK,
138
+ entity: evalId,
139
+ result: `${localeText(locale, "experimentList.noResultsForConfig")} · niceeval exp ${item.experimentId}`,
140
+ duration: null,
141
+ cost: null,
142
+ },
143
+ }));
144
+ rows.push(...missingRows);
101
145
  const flags = item.flags && Object.keys(item.flags).length > 0
102
146
  ? `${localeText(locale, "experimentList.flags")} ${Object.entries(item.flags)
103
147
  .map(([key, value]) => `${key}=${typeof value === "string" ? value : JSON.stringify(value)}`)
@@ -124,14 +168,14 @@ export function experimentListText(items, ctx) {
124
168
  function evalListAttemptLine(item, ctx) {
125
169
  // 行式列表同守「Result 最多两行」:预算 = 两行终端宽,超出按尾截收口。
126
170
  const reason = attemptReasonText(item, ctx.locale, ctx.width * 2 - stringWidth(locatorBadge(item)) - 6);
127
- return ` ${locatorBadge(item)}${reason ? ` · ${reason}` : ""}`;
171
+ return ` ${locatorBadge(item)}${historicalSuffix(item)}${reason ? ` · ${reason}` : ""}`;
128
172
  }
129
173
  export function evalListText(items, ctx) {
130
174
  const locale = ctx.locale;
131
175
  if (items.length === 0)
132
176
  return localeText(locale, "attemptList.empty");
133
177
  const blocks = items.map((item) => {
134
- const identity = `${item.evalId} · ${item.experimentId} · ${localeText(locale, `verdict.${item.verdict}`)}`;
178
+ const identity = `${item.evalId}${evalHistoricalSuffix(item.attempts)} · ${item.experimentId} · ${localeText(locale, `verdict.${item.verdict}`)}`;
135
179
  const summary = [
136
180
  localeText(locale, "attemptList.score", { score: cellText(item.examScore, locale) }),
137
181
  localeText(locale, "overview.attemptsCount", { n: item.attempts.length }),
@@ -151,7 +195,7 @@ export function evalListText(items, ctx) {
151
195
  /** Attempt 比较卡片:只显示一条主失败摘要(至多两行终端宽);完整 assertions 走 locator 下钻。 */
152
196
  function attemptListItemText(item, ctx) {
153
197
  const head = [
154
- `${verdictMark(item.verdict)} ${item.locator}`,
198
+ `${verdictMark(item.verdict)} ${item.locator}${historicalSuffix(item)}`,
155
199
  item.evalId,
156
200
  item.experimentId,
157
201
  formatDurationMs(item.durationMs),
@@ -37,6 +37,10 @@ function attemptListItemProblem(value, path) {
37
37
  return `"${path}.durationMs" must be a number`;
38
38
  if (!(value.costUSD === null || typeof value.costUSD === "number"))
39
39
  return `"${path}.costUSD" must be a number or null`;
40
+ if (typeof value.startedAt !== "string")
41
+ return `"${path}.startedAt" must be a string`;
42
+ if (typeof value.historical !== "boolean")
43
+ return `"${path}.historical" must be a boolean`;
40
44
  if (typeof value.locator !== "string")
41
45
  return `"${path}.locator" must be a string`;
42
46
  return null;
@@ -67,6 +71,11 @@ export const validateExperimentListData = (data) => arrayProblem(data, "data", (
67
71
  return `"${path}.evals" must be a number`;
68
72
  if (typeof item.attempts !== "number")
69
73
  return `"${path}.attempts" must be a number`;
74
+ if (typeof item.historicalAttempts !== "number")
75
+ return `"${path}.historicalAttempts" must be a number`;
76
+ const missingProblem = arrayProblem(item.missingEvalIds, `${path}.missingEvalIds`, (id, idPath) => typeof id === "string" ? null : `"${idPath}" must be a string`);
77
+ if (missingProblem !== null)
78
+ return missingProblem;
70
79
  if (typeof item.lastRunAt !== "string")
71
80
  return `"${path}.lastRunAt" must be a string`;
72
81
  return arrayProblem(item.evalRows, `${path}.evalRows`, (row, rowPath) => {
@@ -1,4 +1,4 @@
1
- import type { AttemptListItem, DeltaData, EvalListItem, ExperimentListItem, LineData, MatrixData, MetricColumn, ScatterData, ScopeSummaryData, ScoreboardData, TableData } from "../model/types.ts";
1
+ import type { AttemptListItem, DeltaData, ExperimentListItem, LineData, MatrixData, MetricColumn, ScatterData, ScopeSummaryData, ScoreboardData, TableData } from "../model/types.ts";
2
2
  export declare const passRateColumn: MetricColumn;
3
3
  export declare const codeLinesColumn: MetricColumn;
4
4
  export declare const costColumn: MetricColumn;
@@ -11,5 +11,4 @@ export declare const scatterData: ScatterData;
11
11
  export declare const lineData: LineData;
12
12
  export declare const deltaData: DeltaData;
13
13
  export declare const attemptListItems: AttemptListItem[];
14
- export declare const evalListItems: EvalListItem[];
15
14
  export declare const experimentListItems: ExperimentListItem[];