niceeval 0.10.3-canary.2 → 0.10.3-canary.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/INDEX.md +1 -0
- package/dist/agents/types.d.ts +12 -0
- package/dist/context/turn-errors.d.ts +49 -0
- package/dist/context/types.d.ts +32 -16
- package/dist/i18n/en.d.ts +3 -0
- package/dist/i18n/zh-CN.d.ts +3 -0
- package/dist/o11y/execution-tree.d.ts +11 -1
- package/dist/o11y/types.d.ts +20 -3
- package/dist/report/components/entity-lists/AttemptList.d.ts +17 -0
- package/dist/report/components/entity-lists/AttemptList.js +23 -2
- package/dist/report/components/entity-lists/EvalList.js +0 -0
- package/dist/report/components/entity-lists/ExperimentList.js +20 -4
- package/dist/report/components/entity-lists/compute.js +10 -2
- package/dist/report/components/entity-lists/faces.js +51 -7
- package/dist/report/components/entity-lists/index.js +9 -0
- package/dist/report/components/fixtures.d.ts +1 -2
- package/dist/report/components/fixtures.js +12 -22
- package/dist/report/components/site-components/ScopeWarnings.js +1 -1
- package/dist/report/components/site-components/index.d.ts +3 -3
- package/dist/report/components/site-components/index.js +13 -26
- package/dist/report/components/site-components/scope-warnings.d.ts +7 -3
- package/dist/report/components/site-components/scope-warnings.js +16 -32
- package/dist/report/model/aggregate.d.ts +9 -1
- package/dist/report/model/aggregate.js +11 -2
- package/dist/report/model/format.d.ts +7 -0
- package/dist/report/model/format.js +12 -0
- package/dist/report/model/locale.d.ts +4 -10
- package/dist/report/model/locale.js +7 -20
- package/dist/report/model/types.d.ts +10 -2
- package/dist/results/select.d.ts +16 -11
- package/dist/results/select.js +87 -89
- package/dist/results/types.d.ts +45 -29
- package/dist/scoring/types.d.ts +1 -1
- package/docs-site/zh/reference/cli.mdx +1 -0
- package/docs-site/zh/reference/define-agent.mdx +16 -0
- package/docs-site/zh/reference/define-eval.mdx +4 -2
- package/docs-site/zh/reference/events.mdx +6 -0
- package/docs-site/zh/reference/report-components.mdx +6 -6
- package/docs-site/zh/reference/results-data.mdx +14 -11
- package/docs-site/zh/troubleshooting/debugging.mdx +1 -2
- package/docs-site/zh/tutorials/agent-feedback-loop.mdx +4 -10
- package/docs-site/zh/tutorials/custom-reports.mdx +2 -2
- package/docs-site/zh/tutorials/local-iteration.mdx +91 -0
- package/docs-site/zh/tutorials/sandbox-providers.mdx +14 -0
- package/docs-site/zh/tutorials/viewing-results.mdx +21 -20
- package/package.json +1 -1
- package/src/agents/index.ts +6 -0
- package/src/agents/types.ts +12 -0
- package/src/agents/ui-message-stream.ts +1 -1
- package/src/cli.ts +7 -0
- package/src/context/context.test.ts +63 -5
- package/src/context/context.ts +14 -6
- package/src/context/send-retry.test.ts +292 -0
- package/src/context/send-retry.ts +169 -0
- package/src/context/session.test.ts +105 -1
- package/src/context/session.ts +40 -2
- package/src/context/turn-errors.test.ts +160 -0
- package/src/context/turn-errors.ts +122 -0
- package/src/context/types.ts +32 -16
- package/src/define.test.ts +1 -1
- package/src/define.ts +2 -0
- package/src/expect/index.test.ts +1 -1
- package/src/i18n/en.ts +6 -1
- package/src/i18n/zh-CN.ts +6 -1
- package/src/o11y/cost.test.ts +1 -1
- package/src/o11y/derive.test.ts +78 -0
- package/src/o11y/derive.ts +10 -2
- package/src/o11y/execution-tree.test.ts +1 -1
- package/src/o11y/execution-tree.ts +19 -3
- package/src/o11y/types.ts +16 -3
- package/src/report/assets/styles.css +6 -6
- package/src/report/components/attempt-detail/attempt-components.test.tsx +12 -274
- package/src/report/components/attempt-detail/validate.test.ts +1 -1
- package/src/report/components/compute.test.ts +36 -52
- package/src/report/components/entity-lists/AttemptList.tsx +37 -1
- package/src/report/components/entity-lists/EvalList.tsx +0 -0
- package/src/report/components/entity-lists/ExperimentList.tsx +43 -2
- package/src/report/components/entity-lists/compute.ts +10 -1
- package/src/report/components/entity-lists/faces.ts +55 -10
- package/src/report/components/entity-lists/index.tsx +7 -0
- package/src/report/components/entity-lists/validate.test.ts +16 -1
- package/src/report/components/fixtures.ts +12 -24
- package/src/report/components/metric-views/chart-math.test.ts +1 -1
- package/src/report/components/metric-views/validate.test.ts +1 -1
- package/src/report/components/site-components/ScopeWarnings.tsx +1 -1
- package/src/report/components/site-components/index.tsx +11 -18
- package/src/report/components/site-components/scope-warnings.ts +16 -39
- package/src/report/components/site-components/site-components.test.tsx +116 -279
- package/src/report/components/site-components/validate.test.ts +8 -11
- package/src/report/components/summaries/validate.test.ts +1 -1
- package/src/report/definition/grid-layout.test.ts +1 -1
- package/src/report/definition/shell-head.test.ts +1 -1
- package/src/report/model/aggregate.ts +14 -3
- package/src/report/model/format.ts +14 -0
- package/src/report/model/locale.ts +9 -20
- package/src/report/model/types.ts +10 -2
- package/src/report/runtime/dual-render.test.tsx +104 -784
- package/src/report/runtime/host.test.ts +1 -1
- package/src/results/annotated-source.test.ts +1 -1
- package/src/results/attempt-evidence.test.ts +1 -1
- package/src/results/host-equivalence.test.ts +78 -243
- package/src/results/index.ts +1 -0
- package/src/results/locator.test.ts +1 -1
- package/src/results/open.ts +6 -3
- package/src/results/results.test.ts +104 -29
- package/src/results/select.ts +96 -90
- package/src/results/types.ts +46 -34
- package/src/runner/attempt.test.ts +1 -1
- package/src/runner/attempt.ts +11 -0
- package/src/runner/cleanup-timeout.test.ts +1 -1
- package/src/runner/discover.test.ts +1 -1
- package/src/runner/eval-selection.test.ts +1 -1
- package/src/runner/eval-source.test.ts +1 -1
- package/src/runner/experiment-cleanup-registry.test.ts +1 -1
- package/src/runner/experiment-labels.test.ts +1 -1
- package/src/runner/feedback/ci.test.ts +8 -528
- package/src/runner/feedback/coordinator.test.ts +1 -1
- package/src/runner/feedback/profile.test.ts +1 -1
- package/src/runner/feedback/reducer.test.ts +1 -1
- package/src/runner/ledger.test.ts +1 -1
- package/src/runner/report.test.ts +6 -8
- package/src/runner/reporters/braintrust.test.ts +1 -1
- package/src/runner/reporters/json.test.ts +1 -1
- package/src/runner/run.test.ts +1 -1
- package/src/runner/run.ts +19 -1
- package/src/runner/sandbox-selection.test.ts +1 -1
- package/src/sandbox/checkpoint.test.ts +1 -1
- package/src/sandbox/e2b-agent-template.test.ts +1 -1
- package/src/sandbox/e2b-reconcile.test.ts +1 -1
- package/src/sandbox/io-retry.test.ts +1 -1
- package/src/sandbox/keep-registry.test.ts +1 -1
- package/src/sandbox/paths.test.ts +1 -1
- package/src/sandbox/retry.test.ts +1 -1
- package/src/scoring/display.test.ts +1 -1
- package/src/scoring/evidence.test.ts +1 -1
- package/src/scoring/judge.test.ts +2 -2
- package/src/scoring/judge.ts +4 -2
- package/src/scoring/scoped.test.ts +239 -0
- package/src/scoring/scoped.ts +84 -16
- package/src/scoring/types.ts +1 -1
- package/src/shared/aggregate.test.ts +1 -1
- package/src/show/command.test.ts +4 -2
- package/src/show/index.ts +4 -2
- package/src/show/render.ts +4 -2
- package/src/show/show.test.ts +34 -879
- package/src/util.test.ts +1 -1
- package/src/view/app/App.tsx +1 -1
- package/src/view/app/lib/attempt-dialog.test.ts +1 -1
- package/src/view/data.test.ts +23 -37
- package/src/view/data.ts +3 -1
- package/src/view/server.ts +1 -1
- package/src/view/site-head.test.ts +30 -62
- package/src/view/site.ts +1 -1
- package/src/view/view-report.test.ts +39 -378
- package/src/agents/ai-sdk-otel.test.ts +0 -23
- package/src/agents/ai-sdk.test.ts +0 -465
- package/src/agents/bub-install-spec.test.ts +0 -34
- package/src/agents/claude-code.test.ts +0 -315
- package/src/agents/codex.test.ts +0 -523
- package/src/agents/coding-cli-versions.test.ts +0 -15
- package/src/agents/langgraph.test.ts +0 -204
- package/src/agents/native-config.test.ts +0 -179
- package/src/agents/openai-compat.test.ts +0 -57
- package/src/agents/openclaw.test.ts +0 -31
- package/src/agents/plugin-config.test.ts +0 -95
- package/src/agents/sdk-streams.test.ts +0 -224
- package/src/agents/skills.test.ts +0 -215
- package/src/agents/streaming.test.ts +0 -146
- package/src/agents/ui-message-stream.test.ts +0 -254
- package/src/o11y/otlp/mappers/claude-code.test.ts +0 -32
- package/src/o11y/otlp/parse.test.ts +0 -128
- package/src/o11y/otlp/turn-otel.test.ts +0 -111
- package/src/o11y/parsers/bub.test.ts +0 -72
- package/src/o11y/parsers/claude-code.test.ts +0 -142
- package/src/o11y/parsers/openclaw.test.ts +0 -154
- package/src/o11y/tool-names.test.ts +0 -62
- package/src/report/components/render.test.tsx +0 -472
- package/src/runner/feedback/agent.test.ts +0 -536
- package/src/runner/feedback/human.test.ts +0 -723
- package/src/view/app/App.test.tsx +0 -131
- package/src/view/artifact-serving.test.ts +0 -141
- package/src/view/site-parity.test.ts +0 -140
package/INDEX.md
CHANGED
|
@@ -26,6 +26,7 @@
|
|
|
26
26
|
- `docs-site/zh/tutorials/dataset-fanout.mdx` — 数据驱动测试(dataset fan-out):用多份数据运行同一套评估用例:从 .eval.ts 文件导出数组或 keyed record,将一套评估逻辑展开为多个 case。用 loadYaml 或 loadJson 读取外部数据集,并获得稳定 ID。
|
|
27
27
|
- `docs-site/zh/tutorials/experiments.mdx` — 实验矩阵:用运行矩阵比较 agents 和 models:使用 NiceEval experiments 让同一批评估用例横跨多个 agents、models 和 flags,比较 pass rate、成本和延迟。
|
|
28
28
|
- `docs-site/zh/tutorials/fixtures.mdx` — Sandbox Fixture:用任务评估 coding agents:用 .eval.ts 给 coding agent 准备隔离 workspace、发送真实任务,并用 Sandbox 文件、命令、diff 和 judge 验证结果。
|
|
29
|
+
- `docs-site/zh/tutorials/local-iteration.mdx` — 本地批跑提速:复用一个热 Sandbox:用 --reuse-sandbox 让一批评估共用一个装好的 Sandbox 串行跑,把重复安装折成一次;并按生命周期把安装写对层,让每道题几乎立即开跑。
|
|
29
30
|
- `docs-site/zh/tutorials/publish-report.mdx` — 通过 CI 发布报告:把经过 copySnapshots 大小预检的结果目录提交进仓库,CI 用一行 view --results 导出报告站;超大文件在 commit 前就会得到可执行错误。
|
|
30
31
|
- `docs-site/zh/tutorials/quickstart.mdx` — 为你的 Agent 项目设置评估:安装 NiceEval,写三个文件,10 分钟内对你自己的应用跑通第一条评估用例。
|
|
31
32
|
- `docs-site/zh/tutorials/reporters.mdx` — 把结果上报到 Braintrust 与其它目的地:用内置 reporters 把评估用例结果送到 Braintrust 实验、JUnit XML 或自定义目的地。
|
package/dist/agents/types.d.ts
CHANGED
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
import type { DiagnosticInput, ProgressUpdate } from "../shared/types.ts";
|
|
2
2
|
import type { StreamEvent, TraceSpan, Usage } from "../o11y/types.ts";
|
|
3
3
|
import type { Sandbox } from "../sandbox/types.ts";
|
|
4
|
+
import type { TurnErrorClassifier } from "../context/turn-errors.ts";
|
|
4
5
|
/**
|
|
5
6
|
* 本地 stdio 形态的 MCP server:沙箱内起子进程,按 stdio 说 MCP 协议。
|
|
6
7
|
* 与 {@link McpHttpServer} 按形状判别(有 `command` 的是 stdio,有 `url` 的是 HTTP)。
|
|
@@ -360,6 +361,13 @@ export interface Agent {
|
|
|
360
361
|
/** 原生 span → canonical 的薄 mapper;省略走通用 heuristic。只影响瀑布图。 */
|
|
361
362
|
spanMapper?: SpanMapper;
|
|
362
363
|
send(input: TurnInput, ctx: AgentContext): Promise<Turn>;
|
|
364
|
+
/**
|
|
365
|
+
* 可选 turn 失败分类器:按重试安全性归类一次 send 失败(抛出或返回 `status: "failed"` 的
|
|
366
|
+
* Turn),返回 `undefined` 回落保守兜底。分类器只声明决策与诊断词,不影响重试策略(次数、
|
|
367
|
+
* 退避对所有 agent 一致);抛错按不可重试处理并被吞掉。形状与分类链、执行体时序见
|
|
368
|
+
* docs/feature/error-classification/architecture.md。
|
|
369
|
+
*/
|
|
370
|
+
classifyTurnError?: TurnErrorClassifier;
|
|
363
371
|
teardown?: AgentTeardown;
|
|
364
372
|
}
|
|
365
373
|
/** `defineSandboxAgent()` 的入参形状(见 src/define.ts)——`kind: "sandbox"` 由 define 固定填入,不由用户声明。 */
|
|
@@ -380,6 +388,8 @@ export interface SandboxAgentDef {
|
|
|
380
388
|
spanMapper?: SpanMapper;
|
|
381
389
|
/** 每轮一次:跑 prompt(fresh / resume)+ 解析成 events。 */
|
|
382
390
|
send(input: TurnInput, ctx: AgentContext): Promise<Turn>;
|
|
391
|
+
/** 可选 turn 失败分类器:见 `Agent.classifyTurnError`。 */
|
|
392
|
+
classifyTurnError?: TurnErrorClassifier;
|
|
383
393
|
/** Sandbox 销毁前的清理,当且仅当本 attempt 走到过 `setup` 时点才执行(`setup` 抛错不豁免),
|
|
384
394
|
* 在 finally 里跑一次。 */
|
|
385
395
|
teardown?: AgentTeardown;
|
|
@@ -402,6 +412,8 @@ export interface RemoteAgentDef {
|
|
|
402
412
|
spanMapper?: SpanMapper;
|
|
403
413
|
/** 每轮一次:把一轮 prompt 发给远程被测对象(HTTP/SDK 等),解析响应成 events。 */
|
|
404
414
|
send(input: TurnInput, ctx: AgentContext): Promise<Turn>;
|
|
415
|
+
/** 可选 turn 失败分类器:见 `Agent.classifyTurnError`。 */
|
|
416
|
+
classifyTurnError?: TurnErrorClassifier;
|
|
405
417
|
/** 运行结束前的清理,当且仅当本 attempt 走到过 `setup` 时点才执行(`setup` 抛错不豁免),
|
|
406
418
|
* 在 finally 里跑一次。 */
|
|
407
419
|
teardown?: AgentTeardown;
|
|
@@ -0,0 +1,49 @@
|
|
|
1
|
+
import type { Turn } from "../types.ts";
|
|
2
|
+
/**
|
|
3
|
+
* 一次 send 失败的分类结果:`retryable` 是执行体唯一消费的决策轴;`reason` 是开放词表的
|
|
4
|
+
* 细分诊断,只进 activity 与耗尽摘要,不参与策略。内建兜底产出 reason `"rate_limit"` /
|
|
5
|
+
* `"network"`;adapter 分类器可自造词。`retryable: true` 时 `reason` 必填——可重试的失败
|
|
6
|
+
* 一定会出现在 activity 行与可能的耗尽摘要里,那里需要一个给人读的词。
|
|
7
|
+
*/
|
|
8
|
+
export type TurnErrorClass = {
|
|
9
|
+
readonly retryable: true;
|
|
10
|
+
readonly reason: string;
|
|
11
|
+
} | {
|
|
12
|
+
readonly retryable: false;
|
|
13
|
+
readonly reason?: string;
|
|
14
|
+
};
|
|
15
|
+
/** 一次 send 失败的两种浮出形态:`send()` 抛出异常,或返回 `status: "failed"` 的 Turn。 */
|
|
16
|
+
export type TurnFailure = {
|
|
17
|
+
readonly type: "thrown";
|
|
18
|
+
readonly error: unknown;
|
|
19
|
+
} | {
|
|
20
|
+
readonly type: "turn-failed";
|
|
21
|
+
readonly turn: Turn;
|
|
22
|
+
};
|
|
23
|
+
/**
|
|
24
|
+
* adapter 可选分类器:返回 `undefined` 表示「不认识,交给保守兜底」。分类器必须快、纯、
|
|
25
|
+
* 不抛错——执行体按「抛错等价于不可重试」处理,自身错误被吞掉,不会掩盖原始失败。
|
|
26
|
+
*/
|
|
27
|
+
export type TurnErrorClassifier = (failure: TurnFailure) => TurnErrorClass | undefined;
|
|
28
|
+
/**
|
|
29
|
+
* 失败 Turn 的错误摘要:取 `events` 里最后一个 `type: "error"` 事件的 message。
|
|
30
|
+
* 与 `context.turnFailed` 报错文案、保守兜底分类器读的同一段文本同源——不出现
|
|
31
|
+
* 「报错说 A、分类看 B」。没有 error 事件(status: "failed" 但 adapter 没吐错误事件)时
|
|
32
|
+
* 返回 `undefined`。
|
|
33
|
+
*/
|
|
34
|
+
export declare function turnErrorText(turn: Turn): string | undefined;
|
|
35
|
+
/** 两种 `TurnFailure` 形态统一取「给人读也给分类器看」的那段文本。 */
|
|
36
|
+
export declare function turnFailureText(failure: TurnFailure): string;
|
|
37
|
+
/**
|
|
38
|
+
* 保守兜底分类器:三道分类链里的第二道。对失败文本做正则匹配,认不出的一律 `{ retryable: false }`
|
|
39
|
+
* ——宁可判死一个 attempt,不产出不可信的 verdict(判据见 README「分类」)。
|
|
40
|
+
*/
|
|
41
|
+
export declare function classifyTurnError(failure: TurnFailure): TurnErrorClass;
|
|
42
|
+
/** 受理证据门:失败 Turn 的 events 里已出现任何 agent 侧产出,即证明 agent 已受理并开始工作。 */
|
|
43
|
+
export declare function hasAgentEvidence(turn: Turn): boolean;
|
|
44
|
+
/**
|
|
45
|
+
* 三道分类链的完整决议:adapter 分类器(可选,抛错按不可重试处理并吞掉)→ 保守兜底 →
|
|
46
|
+
* 受理证据门(否决权,失败 Turn 带 agent 产出事件时强制降级)。执行体只需要调这一个函数,
|
|
47
|
+
* 不必自己拼三道链的顺序。
|
|
48
|
+
*/
|
|
49
|
+
export declare function resolveTurnErrorClass(failure: TurnFailure, adapterClassifier?: TurnErrorClassifier): TurnErrorClass;
|
package/dist/context/types.d.ts
CHANGED
|
@@ -94,26 +94,40 @@ export interface DiffView {
|
|
|
94
94
|
/** 正则是否命中 diff 里任意文件的路径或内容。 */
|
|
95
95
|
matches(re: RegExp): boolean;
|
|
96
96
|
}
|
|
97
|
-
/**
|
|
97
|
+
/**
|
|
98
|
+
* 工具匹配小语言。一条调用的全部可断面——入参、次数、输出、状态——都在这一个对象里表达:
|
|
99
|
+
* `input` / `output` / `status` 之间是 AND,且作用在同一笔调用上;`count` 数的是满足这些
|
|
100
|
+
* 条件的调用笔数,不存在「一笔满足 input、另一笔满足 output」也算命中的读法。
|
|
101
|
+
*/
|
|
98
102
|
export interface ToolMatch {
|
|
99
103
|
/**
|
|
100
|
-
*
|
|
101
|
-
*
|
|
102
|
-
*
|
|
104
|
+
* 入参匹配:对象做**深度部分匹配**(写出的键值要求出现且相等,未写的忽略,嵌套递归比较;
|
|
105
|
+
* 值位置可以放 RegExp 匹配该字段的字符串值,不命中时再对整个 input 的序列化串兜底测一次,
|
|
106
|
+
* 或放谓词函数拿该字段原始值判断);顶层直接给 RegExp 则匹配序列化后的**完整输入**;
|
|
107
|
+
* 顶层给谓词函数 `(input) => boolean` 拿原始输入值自行判断。三种顶层形态互斥,不会退化
|
|
108
|
+
* 成深比对——RegExp / 函数不是"键值对象",不会被当成 plain object 逐键枚举。
|
|
103
109
|
*/
|
|
104
|
-
input?: Record<string, unknown
|
|
105
|
-
/**
|
|
106
|
-
count?: number;
|
|
107
|
-
/**
|
|
108
|
-
|
|
110
|
+
input?: Record<string, unknown> | RegExp | ((input: unknown) => boolean);
|
|
111
|
+
/** 数字精确匹配调用次数;谓词对命中次数自行判定(如 `(n) => n >= 2`);省略则只要求「至少一次」。 */
|
|
112
|
+
count?: number | ((n: number) => boolean);
|
|
113
|
+
/**
|
|
114
|
+
* 输出匹配,值语义同 `input` 的值位置:RegExp 对字符串输出测试(非字符串先序列化再测);
|
|
115
|
+
* 谓词函数拿原始输出自行判断;对象做深度部分匹配;其余值严格相等。
|
|
116
|
+
*/
|
|
117
|
+
output?: unknown;
|
|
118
|
+
/** 只匹配处于该状态的调用。`pending` 是已发起、尚无结果的调用——典型是 HITL 停在审批上的那一笔。 */
|
|
119
|
+
status?: "pending" | "completed" | "failed" | "rejected";
|
|
109
120
|
}
|
|
110
121
|
/** calledSubagent 的匹配小语言,语义同 ToolMatch。 */
|
|
111
122
|
export interface SubagentMatch {
|
|
112
|
-
/**
|
|
113
|
-
count?: number;
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
|
|
123
|
+
/** 数字精确匹配调用次数;谓词对命中次数自行判定;省略则只要求「至少一次」。 */
|
|
124
|
+
count?: number | ((n: number) => boolean);
|
|
125
|
+
/** 子 agent 委派没有 rejected 状态(subagent.completed 只报 completed / failed)。 */
|
|
126
|
+
status?: "pending" | "completed" | "failed";
|
|
127
|
+
/** 只匹配指向该远程地址的子 agent 调用:字符串精确匹配、RegExp 测试、或谓词函数自行判断。 */
|
|
128
|
+
remoteUrl?: string | RegExp | ((url: string) => boolean);
|
|
129
|
+
/** 匹配子 agent 的返回,值语义同 ToolMatch.output。 */
|
|
130
|
+
output?: unknown;
|
|
117
131
|
}
|
|
118
132
|
/** requireInputRequest 的过滤条件;多个字段之间是 AND 关系。 */
|
|
119
133
|
export interface InputRequestFilter {
|
|
@@ -252,8 +266,10 @@ export interface TestContext {
|
|
|
252
266
|
/** 取默认会话里等待中的 HITL 输入请求;不传 filter 要求恰好一条,拿不到就抛。 */
|
|
253
267
|
requireInputRequest(filter?: InputRequestFilter): InputRequest;
|
|
254
268
|
/**
|
|
255
|
-
* 回答默认会话里等待中的输入请求,返回续接的 TurnHandle
|
|
256
|
-
*
|
|
269
|
+
* 回答默认会话里等待中的输入请求,返回续接的 TurnHandle。字符串形式只在恰好一条待处理请求时
|
|
270
|
+
* 才能自动对位——命中该请求 `options` 里的某个 id 就是 `optionId`,否则整句落自由文本;
|
|
271
|
+
* 多个请求并停时字符串形式无法消歧,直接抛 `hitl.stringAmbiguous`,要求改用
|
|
272
|
+
* `{ request, optionId }` / `{ request, text }` 对象形式(RespondAnswer,见其类型注释)显式指名。
|
|
257
273
|
*/
|
|
258
274
|
respond(...responses: (string | RespondAnswer)[]): Promise<TurnHandle>;
|
|
259
275
|
/** 用同一个 optionId 批量回答默认会话里全部等待中的输入请求。 */
|
package/dist/i18n/en.d.ts
CHANGED
|
@@ -238,6 +238,9 @@ export declare const en: {
|
|
|
238
238
|
"session.tools": string;
|
|
239
239
|
"session.turn.primary": string;
|
|
240
240
|
"session.turn.secondary": string;
|
|
241
|
+
"session.turnRetry": string;
|
|
242
|
+
"session.turnRetryBudgetExhausted": string;
|
|
243
|
+
"session.turnRetrySendExhausted": string;
|
|
241
244
|
"util.requiredEnv": string;
|
|
242
245
|
"vercel.fileNotFound": string;
|
|
243
246
|
"vercel.rotateFailed": string;
|
package/dist/i18n/zh-CN.d.ts
CHANGED
|
@@ -238,6 +238,9 @@ export declare const zhCN: {
|
|
|
238
238
|
readonly "session.tools": "{{count}} 工具";
|
|
239
239
|
readonly "session.turn.primary": "第{{turn}}轮";
|
|
240
240
|
readonly "session.turn.secondary": "会话{{session}}·第{{turn}}轮";
|
|
241
|
+
readonly "session.turnRetry": "turn 重试 {{attempt}}/{{maxAttempts}}({{reason}})——等待 {{seconds}}s";
|
|
242
|
+
readonly "session.turnRetryBudgetExhausted": " · attempt 重试预算已耗尽({{maxRetries}} 次重试,{{reason}})";
|
|
243
|
+
readonly "session.turnRetrySendExhausted": " · 重试已耗尽({{maxAttempts}} 次尝试,{{reason}})";
|
|
241
244
|
readonly "util.requiredEnv": "缺少必需的环境变量 {{name}}(请在 .env 里配置)。";
|
|
242
245
|
readonly "vercel.fileNotFound": "File not found: {{path}}";
|
|
243
246
|
readonly "vercel.rotateFailed": "[VercelSandbox] session rotate failed ({{seconds}}s): {{error}}";
|
|
@@ -18,6 +18,16 @@ export interface ExecutionThinkingNode extends ExecutionNodeBase {
|
|
|
18
18
|
kind: "thinking";
|
|
19
19
|
text: string;
|
|
20
20
|
}
|
|
21
|
+
/**
|
|
22
|
+
* 被测系统内部机制注入进上下文的文本,不属于任何一方"说的话",不并进 `message`
|
|
23
|
+
* (见 docs/feature/adapters/architecture/events.md「不变量 9」)。与 thinking / compaction
|
|
24
|
+
* 同一档次的直通节点,不参与 callId 关联。
|
|
25
|
+
*/
|
|
26
|
+
export interface ExecutionContextInjectedNode extends ExecutionNodeBase {
|
|
27
|
+
kind: "context.injected";
|
|
28
|
+
text: string;
|
|
29
|
+
source?: string;
|
|
30
|
+
}
|
|
21
31
|
/** Skill 加载节点——一等,直接来自 StreamEvent 的 "skill.loaded",不靠工具名/文本猜。 */
|
|
22
32
|
export interface ExecutionSkillNode extends ExecutionNodeBase {
|
|
23
33
|
kind: "skill.loaded";
|
|
@@ -79,7 +89,7 @@ export interface ExecutionTelemetryNode {
|
|
|
79
89
|
id: string;
|
|
80
90
|
span: TraceSpan;
|
|
81
91
|
}
|
|
82
|
-
export type ExecutionNode = ExecutionMessageNode | ExecutionThinkingNode | ExecutionSkillNode | ExecutionActionNode | ExecutionSubagentNode | ExecutionInputRequestedNode | ExecutionCompactionNode | ExecutionErrorNode | ExecutionTelemetryNode;
|
|
92
|
+
export type ExecutionNode = ExecutionMessageNode | ExecutionThinkingNode | ExecutionContextInjectedNode | ExecutionSkillNode | ExecutionActionNode | ExecutionSubagentNode | ExecutionInputRequestedNode | ExecutionCompactionNode | ExecutionErrorNode | ExecutionTelemetryNode;
|
|
83
93
|
export interface ExecutionTree {
|
|
84
94
|
/**
|
|
85
95
|
* 骨架节点在前,顺序 = 事件出现顺序;telemetry-only 节点(未能唯一关联的 span)
|
package/dist/o11y/types.d.ts
CHANGED
|
@@ -110,6 +110,16 @@ export type StreamEvent = {
|
|
|
110
110
|
type: "thinking";
|
|
111
111
|
text: string;
|
|
112
112
|
}
|
|
113
|
+
/**
|
|
114
|
+
* 被测系统内部注入的、不披着 `message` 外衣的上下文文本(如 Claude Code 的 SessionStart /
|
|
115
|
+
* UserPromptSubmit hook 在下一轮开始前前置进模型上下文的文本)。只承载带实际文本内容的注入;
|
|
116
|
+
* `source` 是可选的原始来源标记(如 hook 名),adapter 按各自协议原样透传,不强行归一到封闭枚举。
|
|
117
|
+
*/
|
|
118
|
+
| {
|
|
119
|
+
type: "context.injected";
|
|
120
|
+
text: string;
|
|
121
|
+
source?: string;
|
|
122
|
+
}
|
|
113
123
|
/** 上下文被压缩/摘要(如超长会话截断历史);`reason` 是可选的压缩原因说明。 */
|
|
114
124
|
| {
|
|
115
125
|
type: "compaction";
|
|
@@ -120,21 +130,26 @@ export type StreamEvent = {
|
|
|
120
130
|
type: "error";
|
|
121
131
|
message: string;
|
|
122
132
|
});
|
|
123
|
-
/**
|
|
133
|
+
/**
|
|
134
|
+
* core 从事件流折叠出的结构化事实(deriveRunFacts)。折叠按 callId 把 called 与 result 对成一条
|
|
135
|
+
* 调用:配上 result 的取 result 的状态;只有 called、尚未等到 result 的调用状态是 `pending`——
|
|
136
|
+
* HITL 停在审批上的调用就以这个状态被断言,不是容错分支(见 docs/feature/adapters/architecture/events.md)。
|
|
137
|
+
*/
|
|
124
138
|
export interface ToolCall {
|
|
125
139
|
callId: string;
|
|
126
140
|
name: ToolName;
|
|
127
141
|
originalName?: string;
|
|
128
142
|
input: JsonValue;
|
|
129
143
|
output?: JsonValue;
|
|
130
|
-
status: "completed" | "failed" | "rejected";
|
|
144
|
+
status: "pending" | "completed" | "failed" | "rejected";
|
|
131
145
|
}
|
|
132
146
|
export interface SubagentCall {
|
|
133
147
|
callId: string;
|
|
134
148
|
name: string;
|
|
135
149
|
remoteUrl?: string;
|
|
136
150
|
output?: JsonValue;
|
|
137
|
-
|
|
151
|
+
/** 子 agent 委派没有 rejected 状态(subagent.completed 只报 completed / failed)。 */
|
|
152
|
+
status: "pending" | "completed" | "failed";
|
|
138
153
|
}
|
|
139
154
|
export interface DerivedFacts {
|
|
140
155
|
readonly toolCalls: readonly ToolCall[];
|
|
@@ -143,6 +158,8 @@ export interface DerivedFacts {
|
|
|
143
158
|
readonly parked: boolean;
|
|
144
159
|
readonly messageCount: number;
|
|
145
160
|
readonly compactions: number;
|
|
161
|
+
/** 事件流里 `context.injected` 事件的次数;只回答存在性问题,不替代逐条读取原文。 */
|
|
162
|
+
readonly contextInjections: number;
|
|
146
163
|
}
|
|
147
164
|
/**
|
|
148
165
|
* span 的【语义角色】,从 OTel GenAI 语义约定的 gen_ai.operation.name 归一而来
|
|
@@ -2,6 +2,23 @@ import type { ReactElement } from "react";
|
|
|
2
2
|
import type { AttemptListItem } from "../../model/types.ts";
|
|
3
3
|
import type { AttemptLocator } from "../../../results/locator.ts";
|
|
4
4
|
import { type ReportLocale } from "../../model/locale.ts";
|
|
5
|
+
/**
|
|
6
|
+
* 时效标注(`↩` + 紧凑时距):历史执行(携带,或跨快照拼入)才渲染,新执行不标——subdued
|
|
7
|
+
* 行内事实,不占框、不用警示色(docs/feature/reports/library/entity-lists.md「时效标注」)。
|
|
8
|
+
* web 面 hover 显示完整执行时刻;ExperimentList / EvalList / AttemptList 三处共用。
|
|
9
|
+
*/
|
|
10
|
+
export declare function HistoricalMark({ item, locale, }: {
|
|
11
|
+
item: Pick<AttemptListItem, "startedAt" | "historical">;
|
|
12
|
+
locale?: ReportLocale;
|
|
13
|
+
}): ReactElement | null;
|
|
14
|
+
/**
|
|
15
|
+
* Eval 父行的时效标注:全部 attempt 均为历史执行时,标最近一次执行(startedAt 最大)的时距;
|
|
16
|
+
* 新旧混合时父行不标,子行各自可见(docs/feature/reports/library/entity-lists.md「时效标注」)。
|
|
17
|
+
* EvalList 与 ExperimentList 的 Eval 父行共用。
|
|
18
|
+
*/
|
|
19
|
+
export declare function EvalHistoricalMark({ attempts, }: {
|
|
20
|
+
attempts: readonly Pick<AttemptListItem, "startedAt" | "historical">[];
|
|
21
|
+
}): ReactElement | null;
|
|
5
22
|
/**
|
|
6
23
|
* locator + 判定符,AttemptList/EvalList/ExperimentList 共用。没有 target(当前报告没有
|
|
7
24
|
* declare attempt-input page,也没有显式 attemptHref)时是纯文本,不生成空 href 或假链接
|
|
@@ -1,8 +1,29 @@
|
|
|
1
1
|
import { Fragment as _Fragment, jsx as _jsx, jsxs as _jsxs } from "react/jsx-runtime";
|
|
2
2
|
import { DEFAULT_REPORT_LOCALE, countText, localeText } from "../../model/locale.js";
|
|
3
3
|
import { colorClassForKey } from "../../assets/colors.js";
|
|
4
|
-
import { formatDurationMs, formatUSD, verdictMark } from "../../model/format.js";
|
|
4
|
+
import { formatDurationMs, formatHistoricalGap, formatReportDateTime, formatUSD, verdictMark } from "../../model/format.js";
|
|
5
5
|
import { cx } from "../shared.js";
|
|
6
|
+
/**
|
|
7
|
+
* 时效标注(`↩` + 紧凑时距):历史执行(携带,或跨快照拼入)才渲染,新执行不标——subdued
|
|
8
|
+
* 行内事实,不占框、不用警示色(docs/feature/reports/library/entity-lists.md「时效标注」)。
|
|
9
|
+
* web 面 hover 显示完整执行时刻;ExperimentList / EvalList / AttemptList 三处共用。
|
|
10
|
+
*/
|
|
11
|
+
export function HistoricalMark({ item, locale = DEFAULT_REPORT_LOCALE, }) {
|
|
12
|
+
if (!item.historical)
|
|
13
|
+
return null;
|
|
14
|
+
return (_jsxs("span", { className: "nre-historical", title: formatReportDateTime(item.startedAt, locale), children: ["\u21A9 ", formatHistoricalGap(item.startedAt)] }));
|
|
15
|
+
}
|
|
16
|
+
/**
|
|
17
|
+
* Eval 父行的时效标注:全部 attempt 均为历史执行时,标最近一次执行(startedAt 最大)的时距;
|
|
18
|
+
* 新旧混合时父行不标,子行各自可见(docs/feature/reports/library/entity-lists.md「时效标注」)。
|
|
19
|
+
* EvalList 与 ExperimentList 的 Eval 父行共用。
|
|
20
|
+
*/
|
|
21
|
+
export function EvalHistoricalMark({ attempts, }) {
|
|
22
|
+
if (attempts.length === 0 || !attempts.every((a) => a.historical))
|
|
23
|
+
return null;
|
|
24
|
+
const mostRecent = attempts.reduce((a, b) => (b.startedAt > a.startedAt ? b : a));
|
|
25
|
+
return _jsxs("span", { className: "nre-historical", children: ["\u21A9 ", formatHistoricalGap(mostRecent.startedAt)] });
|
|
26
|
+
}
|
|
6
27
|
/**
|
|
7
28
|
* locator + 判定符,AttemptList/EvalList/ExperimentList 共用。没有 target(当前报告没有
|
|
8
29
|
* declare attempt-input page,也没有显式 attemptHref)时是纯文本,不生成空 href 或假链接
|
|
@@ -24,7 +45,7 @@ export function failureSummaryText(item, locale) {
|
|
|
24
45
|
/** 一条 Attempt 的比较卡片;完整 assertions 通过 locator 下钻,不在列表内展开。 */
|
|
25
46
|
export function AttemptRow({ item, attemptHref, locale = DEFAULT_REPORT_LOCALE, }) {
|
|
26
47
|
const reason = failureSummaryText(item, locale);
|
|
27
|
-
return (_jsxs("li", { className: cx("nre-attempt", `nre-attempt-${item.verdict}`), "data-nre-verdict": item.verdict, children: [_jsxs("div", { className: "nre-attempt-head", children: [_jsx(AttemptLocatorBadge, { item: item, attemptHref: attemptHref }), _jsx("span", { className: "nre-attempt-eval", children: item.evalId }), _jsx("span", { className: "nre-attempt-experiment", children: item.experimentId }), _jsx("span", { className: cx("nre-attempt-agent", "nre-key", colorClassForKey(item.agent)), children: item.agent }), _jsx("span", { className: "nre-attempt-duration", children: formatDurationMs(item.durationMs) }), item.costUSD !== null && _jsx("span", { className: "nre-attempt-cost", children: formatUSD(item.costUSD) })] }), reason && _jsx("p", { className: "nre-attempt-result", children: reason })] }));
|
|
48
|
+
return (_jsxs("li", { className: cx("nre-attempt", `nre-attempt-${item.verdict}`), "data-nre-verdict": item.verdict, children: [_jsxs("div", { className: "nre-attempt-head", children: [_jsx(AttemptLocatorBadge, { item: item, attemptHref: attemptHref }), _jsx(HistoricalMark, { item: item, locale: locale }), _jsx("span", { className: "nre-attempt-eval", children: item.evalId }), _jsx("span", { className: "nre-attempt-experiment", children: item.experimentId }), _jsx("span", { className: cx("nre-attempt-agent", "nre-key", colorClassForKey(item.agent)), children: item.agent }), _jsx("span", { className: "nre-attempt-duration", children: formatDurationMs(item.durationMs) }), item.costUSD !== null && _jsx("span", { className: "nre-attempt-cost", children: formatUSD(item.costUSD) })] }), reason && _jsx("p", { className: "nre-attempt-result", children: reason })] }));
|
|
28
49
|
}
|
|
29
50
|
export function AttemptList({ data, total, filter = false, attemptHref, className, locale = DEFAULT_REPORT_LOCALE, }) {
|
|
30
51
|
const remaining = (total ?? data.length) - data.length;
|
|
Binary file
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
import { Fragment as _Fragment, jsx as _jsx, jsxs as _jsxs } from "react/jsx-runtime";
|
|
2
2
|
import { shortestUniqueLabels } from "../../model/format.js";
|
|
3
3
|
import { DEFAULT_REPORT_LOCALE, localeText } from "../../model/locale.js";
|
|
4
|
-
import { AttemptLocatorBadge, failureSummaryText } from "./AttemptList.js";
|
|
4
|
+
import { AttemptLocatorBadge, EvalHistoricalMark, HistoricalMark, failureSummaryText } from "./AttemptList.js";
|
|
5
5
|
import { MetricCellView } from "../cell.js";
|
|
6
6
|
import { colorClassForKey } from "../../assets/colors.js";
|
|
7
7
|
import { formatDurationMs, formatUSD, verdictMark } from "../../model/format.js";
|
|
@@ -35,12 +35,21 @@ function VerdictSummary({ item, locale }) {
|
|
|
35
35
|
}
|
|
36
36
|
function ExperimentAttemptRow({ attempt, last, attemptHref, locale, }) {
|
|
37
37
|
const reason = failureSummaryText(attempt, locale);
|
|
38
|
-
return (_jsxs("li", { className: cx("nre-experiment-attempt-row", `nre-eval-${attempt.verdict}`), children: [_jsx("span", { className: "nre-attempt-branch", "aria-hidden": "true", children: last ? "└─" : "├─" }),
|
|
38
|
+
return (_jsxs("li", { className: cx("nre-experiment-attempt-row", `nre-eval-${attempt.verdict}`), children: [_jsx("span", { className: "nre-attempt-branch", "aria-hidden": "true", children: last ? "└─" : "├─" }), _jsxs("span", { className: "nre-eval-attempt-badges", children: [_jsx(AttemptLocatorBadge, { item: attempt, attemptHref: attemptHref }), _jsx(HistoricalMark, { item: attempt, locale: locale })] }), _jsxs("span", { className: "nre-eval-attempt-metrics", children: [formatDurationMs(attempt.durationMs), attempt.costUSD !== null && _jsxs(_Fragment, { children: [" \u00B7 ", formatUSD(attempt.costUSD)] })] }), _jsx("span", { className: "nre-eval-reason", children: reason ?? "—" })] }));
|
|
39
39
|
}
|
|
40
40
|
function EvalAttempts({ row, attemptHref, locale, }) {
|
|
41
41
|
const duration = row.durationMs.value === null ? localeText(locale, "cell.missing") : formatDurationMs(row.durationMs.value);
|
|
42
42
|
const cost = row.costUSD.value === null ? localeText(locale, "cell.missing") : formatUSD(row.costUSD.value);
|
|
43
|
-
return (_jsxs("li", { className: "nre-experiment-eval", children: [_jsxs("div", { className: cx("nre-experiment-eval-header", `nre-eval-${row.verdict}`), children: [_jsx("span", { className: cx("nre-eval-verdict", `nre-verdict-${row.verdict}`), children: verdictMark(row.verdict) }), _jsx("span", { className: "nre-eval-id", children: row.evalId }), _jsx("span", { className: "nre-eval-attempt-count", children: localeText(locale, "overview.attemptsCount", { n: row.attempts.length }) }), _jsxs("span", { className: "nre-eval-rollup", children: [localeText(locale, "entityList.average", { value: duration }), " · ", localeText(locale, "entityList.average", { value: cost })] })] }), _jsx("ul", { className: "nre-experiment-attempts", children: row.attempts.map((attempt, index) => (_jsx(ExperimentAttemptRow, { attempt: attempt, last: index === row.attempts.length - 1, attemptHref: attemptHref, locale: locale }, attempt.locator))) })] }));
|
|
43
|
+
return (_jsxs("li", { className: "nre-experiment-eval", children: [_jsxs("div", { className: cx("nre-experiment-eval-header", `nre-eval-${row.verdict}`), children: [_jsx("span", { className: cx("nre-eval-verdict", `nre-verdict-${row.verdict}`), children: verdictMark(row.verdict) }), _jsx("span", { className: "nre-eval-id", children: row.evalId }), _jsx(EvalHistoricalMark, { attempts: row.attempts }), _jsx("span", { className: "nre-eval-attempt-count", children: localeText(locale, "overview.attemptsCount", { n: row.attempts.length }) }), _jsxs("span", { className: "nre-eval-rollup", children: [localeText(locale, "entityList.average", { value: duration }), " · ", localeText(locale, "entityList.average", { value: cost })] })] }), _jsx("ul", { className: "nre-experiment-attempts", children: row.attempts.map((attempt, index) => (_jsx(ExperimentAttemptRow, { attempt: attempt, last: index === row.attempts.length - 1, attemptHref: attemptHref, locale: locale }, attempt.locator))) })] }));
|
|
44
|
+
}
|
|
45
|
+
/**
|
|
46
|
+
* 覆盖缺口的占位行:状态列为 —,结果列为「当前配置下无结果」+ 可复制的补跑命令,无 attempt
|
|
47
|
+
* 子行,不参与任何指标聚合——只把分母缺口摆进读者正在看的表里
|
|
48
|
+
* (docs/feature/reports/library/entity-lists.md「ExperimentList」)。
|
|
49
|
+
*/
|
|
50
|
+
function MissingEvalRow({ evalId, experimentId, locale, }) {
|
|
51
|
+
const command = `niceeval exp ${experimentId}`;
|
|
52
|
+
return (_jsx("li", { className: "nre-experiment-eval nre-experiment-eval-missing", children: _jsxs("div", { className: "nre-experiment-eval-header", children: [_jsx("span", { className: "nre-eval-verdict", children: "\u2014" }), _jsx("span", { className: "nre-eval-id", children: evalId }), _jsxs("span", { className: "nre-eval-rollup", children: [localeText(locale, "experimentList.noResultsForConfig"), " \u00B7 ", _jsx("code", { children: command })] })] }) }));
|
|
44
53
|
}
|
|
45
54
|
function Flags({ flags, locale }) {
|
|
46
55
|
if (!flags || Object.keys(flags).length === 0)
|
|
@@ -48,7 +57,14 @@ function Flags({ flags, locale }) {
|
|
|
48
57
|
return (_jsxs("div", { className: "nre-experiment-flags", children: [_jsx("span", { children: localeText(locale, "experimentList.flags") }), Object.entries(flags).map(([key, value]) => (_jsxs("b", { children: [key, "=", typeof value === "string" ? value : JSON.stringify(value)] }, key)))] }));
|
|
49
58
|
}
|
|
50
59
|
function ExperimentRow({ item, label, attemptHref, locale, }) {
|
|
51
|
-
return (_jsxs("details", { className: "nre-experiment-entry", children: [_jsxs("summary", { className: "nre-experiment-summary", children: [_jsxs("span", { className: "nre-experiment-name", "data-sort-value": item.experimentId, children: [_jsx("b", { className: cx("nre-experiment-id", "nre-key"), children: label }), _jsxs("small", { children: [
|
|
60
|
+
return (_jsxs("details", { className: "nre-experiment-entry", children: [_jsxs("summary", { className: "nre-experiment-summary", children: [_jsxs("span", { className: "nre-experiment-name", "data-sort-value": item.experimentId, children: [_jsx("b", { className: cx("nre-experiment-id", "nre-key"), children: label }), _jsxs("small", { children: [item.missingEvalIds.length > 0
|
|
61
|
+
? localeText(locale, "overview.evalsCountPartial", {
|
|
62
|
+
covered: item.evals,
|
|
63
|
+
total: item.evals + item.missingEvalIds.length,
|
|
64
|
+
})
|
|
65
|
+
: localeText(locale, "overview.evalsCount", { n: item.evals }), item.attempts > item.evals ? ` · ${localeText(locale, "overview.attemptsCount", { n: item.attempts })}` : "", item.historicalAttempts > 0
|
|
66
|
+
? ` · ${localeText(locale, "experimentList.historicalAttempts", { n: item.historicalAttempts, m: item.attempts })}`
|
|
67
|
+
: "", ` · ${formatDate(item.lastRunAt, locale)}`] })] }), _jsx("span", { "data-sort-value": item.model ?? "", children: item.model ?? localeText(locale, "experimentList.defaultModel") }), _jsx("span", { "data-sort-value": item.agent, children: _jsx("span", { className: cx("nre-experiment-agent", "nre-key", colorClassForKey(item.agent)), children: item.agent }) }), _jsx("span", { className: "nre-num", "data-sort-value": item.durationMs.value ?? "", children: _jsx(MetricCellView, { cell: item.durationMs, locale: locale }) }), _jsx("span", { className: cx("nre-num", passRateTone(item.endToEndPassRate.value)), "data-sort-value": item.endToEndPassRate.value ?? "", children: _jsx(MetricCellView, { cell: item.endToEndPassRate, locale: locale }) }), _jsx("span", { className: "nre-num", "data-sort-value": item.tokens.value ?? "", children: _jsx(MetricCellView, { cell: item.tokens, locale: locale }) }), _jsx("span", { className: "nre-num", "data-sort-value": item.costUSD.value ?? "", children: _jsx(MetricCellView, { cell: item.costUSD, locale: locale }) }), _jsx("span", { "data-sort-value": item.evalVerdicts.passed, children: _jsx(VerdictSummary, { item: item, locale: locale }) })] }), _jsxs("div", { className: "nre-experiment-detail", children: [_jsx(Flags, { flags: item.flags, locale: locale }), _jsxs("ul", { className: "nre-experiment-evals", children: [item.evalRows.map((row) => (_jsx(EvalAttempts, { row: row, attemptHref: attemptHref, locale: locale }, row.evalId))), item.missingEvalIds.map((evalId) => (_jsx(MissingEvalRow, { evalId: evalId, experimentId: item.experimentId, locale: locale }, evalId)))] })] })] }));
|
|
52
68
|
}
|
|
53
69
|
export function ExperimentList({ data, attemptHref, filter = false, className, locale = DEFAULT_REPORT_LOCALE, }) {
|
|
54
70
|
const experimentLabels = shortestUniqueLabels(data.map((item) => item.experimentId));
|
|
@@ -8,7 +8,7 @@
|
|
|
8
8
|
// - core 中立:只认 Metric / Dimension 接口,不出现具体 agent 名的分支。
|
|
9
9
|
import { comparabilityConfigOf, deepEqualJson } from "../../../results/select.js";
|
|
10
10
|
import { foldEvalVerdict } from "../../../shared/verdict.js";
|
|
11
|
-
import { collectItems, computeCell, evalIdOf, experimentIdOf, fullEvalKey, groupItems, locatorOf, resolveInput, } from "../../model/aggregate.js";
|
|
11
|
+
import { collectItems, computeCell, evalIdOf, experimentIdOf, fullEvalKey, groupItems, historicalOf, locatorOf, resolveInput, } from "../../model/aggregate.js";
|
|
12
12
|
import { attemptCostUSD, costUSD, durationMs, endToEndPassRate, examScore, tokens } from "../../model/metrics.js";
|
|
13
13
|
import { compactAssertionSummary, primaryAssertionSummary, summaryText } from "../../../scoring/display.js";
|
|
14
14
|
import { selectedEvalsOnly, summarizeItems } from "../shared-compute.js";
|
|
@@ -52,6 +52,11 @@ async function attemptListItemOf(item) {
|
|
|
52
52
|
examScore: await computeCell(examScore, [item]),
|
|
53
53
|
durationMs: result.durationMs,
|
|
54
54
|
costUSD: attemptCostUSD(result),
|
|
55
|
+
// 缺 startedAt(legacy / 第三方落盘)时退化到所属快照的 startedAt——时效标注宁可粗一档
|
|
56
|
+
// 时距,不留空字段(与 dedupeAttempts「缺才不去重」同一条「不伪造」纪律,这里伪造的只是
|
|
57
|
+
// 展示粒度,不影响身份判定)。
|
|
58
|
+
startedAt: result.startedAt ?? item.snapshot.startedAt,
|
|
59
|
+
historical: historicalOf(item),
|
|
55
60
|
locator: locatorOf(item),
|
|
56
61
|
};
|
|
57
62
|
}
|
|
@@ -100,8 +105,9 @@ export async function evalListData(input) {
|
|
|
100
105
|
* 看跨配置演化用 snapshot 维度或 MetricLine,不把两套配置拼成一行冒充单一配置。
|
|
101
106
|
*/
|
|
102
107
|
export async function experimentListData(input) {
|
|
103
|
-
const { snapshots: rawSnapshots } = resolveInput(input);
|
|
108
|
+
const { snapshots: rawSnapshots, coverage } = resolveInput(input);
|
|
104
109
|
const snapshots = selectedEvalsOnly(rawSnapshots);
|
|
110
|
+
const coverageByExperiment = new Map(coverage.map((c) => [c.experimentId, c]));
|
|
105
111
|
// 可比性配置单义检查:同一 experiment 的输入快照必须共享一套可比性配置。
|
|
106
112
|
const configByExperiment = new Map();
|
|
107
113
|
for (const snapshot of snapshots) {
|
|
@@ -151,6 +157,8 @@ export async function experimentListData(input) {
|
|
|
151
157
|
tokens: await computeCell(tokens, group),
|
|
152
158
|
evals: stats.evals,
|
|
153
159
|
attempts: stats.attempts,
|
|
160
|
+
historicalAttempts: group.filter(historicalOf).length,
|
|
161
|
+
missingEvalIds: coverageByExperiment.get(experimentId)?.missingEvalIds ?? [],
|
|
154
162
|
lastRunAt: stats.lastRunAt,
|
|
155
163
|
evalRows,
|
|
156
164
|
});
|
|
@@ -3,7 +3,7 @@
|
|
|
3
3
|
// locator,中间不留空格)。ExperimentList / EvalList 逐 attempt 只列这一个标记 + 各自的
|
|
4
4
|
// 摘要,不重复整段 niceeval show 命令;要看某个 attempt 的完整证据,agent 自己拼
|
|
5
5
|
// `niceeval show <locator>`。零 react、零 IO、纯同步。
|
|
6
|
-
import { fitFailureSummary, formatDurationMs, formatUSD, shortestUniqueLabels, verdictMark, } from "../../model/format.js";
|
|
6
|
+
import { fitFailureSummary, formatDurationMs, formatHistoricalGap, formatUSD, shortestUniqueLabels, verdictMark, } from "../../model/format.js";
|
|
7
7
|
import { countText, localeText } from "../../model/locale.js";
|
|
8
8
|
import { stringWidth, wrapDisplay } from "../../model/text-layout.js";
|
|
9
9
|
import { renderTableText } from "../../definition/table-text.js";
|
|
@@ -16,6 +16,21 @@ import { cellText, missingText, verdictTallyText, MISSING_MARK } from "../shared
|
|
|
16
16
|
function locatorBadge(item) {
|
|
17
17
|
return `${item.locator}${verdictMark(item.verdict)}`;
|
|
18
18
|
}
|
|
19
|
+
/**
|
|
20
|
+
* 时效标注(`↩` + 紧凑时距)的 text 面:历史执行(携带,或跨快照拼入)才输出,新执行为
|
|
21
|
+
* 空串;三面(ExperimentList / EvalList / AttemptList)共用
|
|
22
|
+
* (docs/feature/reports/library/entity-lists.md「时效标注」)。
|
|
23
|
+
*/
|
|
24
|
+
function historicalSuffix(item) {
|
|
25
|
+
return item.historical ? ` ↩ ${formatHistoricalGap(item.startedAt)}` : "";
|
|
26
|
+
}
|
|
27
|
+
/** Eval 父行的时效标注:全部 attempt 均为历史执行时,标最近一次执行的时距;新旧混合不标。 */
|
|
28
|
+
function evalHistoricalSuffix(attempts) {
|
|
29
|
+
if (attempts.length === 0 || !attempts.every((a) => a.historical))
|
|
30
|
+
return "";
|
|
31
|
+
const mostRecent = attempts.reduce((a, b) => (b.startedAt > a.startedAt ? b : a));
|
|
32
|
+
return ` ↩ ${formatHistoricalGap(mostRecent.startedAt)}`;
|
|
33
|
+
}
|
|
19
34
|
/**
|
|
20
35
|
* failureSummary + moreFailures 的展示形态:摘要在计算侧已按 Scoring display 契约折好,
|
|
21
36
|
* 这里只加 "+N more failures" 计数与宽度收口,不重算摘要。
|
|
@@ -55,7 +70,23 @@ function experimentSummaryTable(items, ctx, labels) {
|
|
|
55
70
|
cost: cellText(item.costUSD, locale),
|
|
56
71
|
},
|
|
57
72
|
}));
|
|
58
|
-
const metadata = items.flatMap((item) =>
|
|
73
|
+
const metadata = items.flatMap((item) => {
|
|
74
|
+
const evalsText = item.missingEvalIds.length > 0
|
|
75
|
+
? localeText(locale, "overview.evalsCountPartial", {
|
|
76
|
+
covered: item.evals,
|
|
77
|
+
total: item.evals + item.missingEvalIds.length,
|
|
78
|
+
})
|
|
79
|
+
: localeText(locale, "overview.evalsCount", { n: item.evals });
|
|
80
|
+
const parts = [
|
|
81
|
+
evalsText,
|
|
82
|
+
localeText(locale, "overview.attemptsCount", { n: item.attempts }),
|
|
83
|
+
...(item.historicalAttempts > 0
|
|
84
|
+
? [localeText(locale, "experimentList.historicalAttempts", { n: item.historicalAttempts, m: item.attempts })]
|
|
85
|
+
: []),
|
|
86
|
+
item.lastRunAt,
|
|
87
|
+
];
|
|
88
|
+
return wrapDisplay(`${labels.get(item.experimentId) ?? item.experimentId}: ${parts.join(" · ")}`, Math.max(8, ctx.width - 2)).map((line) => ` ${line}`);
|
|
89
|
+
});
|
|
59
90
|
return [renderTableText({ columns: columns, rows, locale }, ctx), metadata.join("\n")].join("\n");
|
|
60
91
|
}
|
|
61
92
|
function experimentDetailTable(item, ctx, label) {
|
|
@@ -80,7 +111,7 @@ function experimentDetailTable(item, ctx, label) {
|
|
|
80
111
|
key: row.evalId,
|
|
81
112
|
cells: {
|
|
82
113
|
status: `${verdictMark(row.verdict)} ${localeText(locale, `verdict.${row.verdict}`)}`,
|
|
83
|
-
entity: row.evalId
|
|
114
|
+
entity: `${row.evalId}${evalHistoricalSuffix(row.attempts)}`,
|
|
84
115
|
result: "",
|
|
85
116
|
duration: localeText(locale, "entityList.average", { value: cellText(row.durationMs, locale) }),
|
|
86
117
|
cost: localeText(locale, "entityList.average", { value: cellText(row.costUSD, locale) }),
|
|
@@ -90,7 +121,7 @@ function experimentDetailTable(item, ctx, label) {
|
|
|
90
121
|
key: attempt.locator,
|
|
91
122
|
cells: {
|
|
92
123
|
status: ` ${verdictMark(attempt.verdict)}`,
|
|
93
|
-
entity: `${index === row.attempts.length - 1 ? "└─" : "├─"} ${attempt.locator}`,
|
|
124
|
+
entity: `${index === row.attempts.length - 1 ? "└─" : "├─"} ${attempt.locator}${historicalSuffix(attempt)}`,
|
|
94
125
|
result: attemptReasonText(attempt, locale, resultBudget) ?? MISSING_MARK,
|
|
95
126
|
duration: attempt.verdict === "skipped" && attempt.durationMs === 0 ? null : formatDurationMs(attempt.durationMs),
|
|
96
127
|
cost: attempt.costUSD === null ? null : formatUSD(attempt.costUSD),
|
|
@@ -98,6 +129,19 @@ function experimentDetailTable(item, ctx, label) {
|
|
|
98
129
|
}));
|
|
99
130
|
return [parent, ...attempts];
|
|
100
131
|
});
|
|
132
|
+
// 覆盖缺口的占位行:状态列为 —,结果列为「当前配置下无结果」+ 可复制的补跑命令,
|
|
133
|
+
// 无 attempt 子行,duration/cost 留空(不参与任何指标聚合)。
|
|
134
|
+
const missingRows = item.missingEvalIds.map((evalId) => ({
|
|
135
|
+
key: evalId,
|
|
136
|
+
cells: {
|
|
137
|
+
status: MISSING_MARK,
|
|
138
|
+
entity: evalId,
|
|
139
|
+
result: `${localeText(locale, "experimentList.noResultsForConfig")} · niceeval exp ${item.experimentId}`,
|
|
140
|
+
duration: null,
|
|
141
|
+
cost: null,
|
|
142
|
+
},
|
|
143
|
+
}));
|
|
144
|
+
rows.push(...missingRows);
|
|
101
145
|
const flags = item.flags && Object.keys(item.flags).length > 0
|
|
102
146
|
? `${localeText(locale, "experimentList.flags")} ${Object.entries(item.flags)
|
|
103
147
|
.map(([key, value]) => `${key}=${typeof value === "string" ? value : JSON.stringify(value)}`)
|
|
@@ -124,14 +168,14 @@ export function experimentListText(items, ctx) {
|
|
|
124
168
|
function evalListAttemptLine(item, ctx) {
|
|
125
169
|
// 行式列表同守「Result 最多两行」:预算 = 两行终端宽,超出按尾截收口。
|
|
126
170
|
const reason = attemptReasonText(item, ctx.locale, ctx.width * 2 - stringWidth(locatorBadge(item)) - 6);
|
|
127
|
-
return ` ${locatorBadge(item)}${reason ? ` · ${reason}` : ""}`;
|
|
171
|
+
return ` ${locatorBadge(item)}${historicalSuffix(item)}${reason ? ` · ${reason}` : ""}`;
|
|
128
172
|
}
|
|
129
173
|
export function evalListText(items, ctx) {
|
|
130
174
|
const locale = ctx.locale;
|
|
131
175
|
if (items.length === 0)
|
|
132
176
|
return localeText(locale, "attemptList.empty");
|
|
133
177
|
const blocks = items.map((item) => {
|
|
134
|
-
const identity = `${item.evalId} · ${item.experimentId} · ${localeText(locale, `verdict.${item.verdict}`)}`;
|
|
178
|
+
const identity = `${item.evalId}${evalHistoricalSuffix(item.attempts)} · ${item.experimentId} · ${localeText(locale, `verdict.${item.verdict}`)}`;
|
|
135
179
|
const summary = [
|
|
136
180
|
localeText(locale, "attemptList.score", { score: cellText(item.examScore, locale) }),
|
|
137
181
|
localeText(locale, "overview.attemptsCount", { n: item.attempts.length }),
|
|
@@ -151,7 +195,7 @@ export function evalListText(items, ctx) {
|
|
|
151
195
|
/** Attempt 比较卡片:只显示一条主失败摘要(至多两行终端宽);完整 assertions 走 locator 下钻。 */
|
|
152
196
|
function attemptListItemText(item, ctx) {
|
|
153
197
|
const head = [
|
|
154
|
-
`${verdictMark(item.verdict)} ${item.locator}`,
|
|
198
|
+
`${verdictMark(item.verdict)} ${item.locator}${historicalSuffix(item)}`,
|
|
155
199
|
item.evalId,
|
|
156
200
|
item.experimentId,
|
|
157
201
|
formatDurationMs(item.durationMs),
|
|
@@ -37,6 +37,10 @@ function attemptListItemProblem(value, path) {
|
|
|
37
37
|
return `"${path}.durationMs" must be a number`;
|
|
38
38
|
if (!(value.costUSD === null || typeof value.costUSD === "number"))
|
|
39
39
|
return `"${path}.costUSD" must be a number or null`;
|
|
40
|
+
if (typeof value.startedAt !== "string")
|
|
41
|
+
return `"${path}.startedAt" must be a string`;
|
|
42
|
+
if (typeof value.historical !== "boolean")
|
|
43
|
+
return `"${path}.historical" must be a boolean`;
|
|
40
44
|
if (typeof value.locator !== "string")
|
|
41
45
|
return `"${path}.locator" must be a string`;
|
|
42
46
|
return null;
|
|
@@ -67,6 +71,11 @@ export const validateExperimentListData = (data) => arrayProblem(data, "data", (
|
|
|
67
71
|
return `"${path}.evals" must be a number`;
|
|
68
72
|
if (typeof item.attempts !== "number")
|
|
69
73
|
return `"${path}.attempts" must be a number`;
|
|
74
|
+
if (typeof item.historicalAttempts !== "number")
|
|
75
|
+
return `"${path}.historicalAttempts" must be a number`;
|
|
76
|
+
const missingProblem = arrayProblem(item.missingEvalIds, `${path}.missingEvalIds`, (id, idPath) => typeof id === "string" ? null : `"${idPath}" must be a string`);
|
|
77
|
+
if (missingProblem !== null)
|
|
78
|
+
return missingProblem;
|
|
70
79
|
if (typeof item.lastRunAt !== "string")
|
|
71
80
|
return `"${path}.lastRunAt" must be a string`;
|
|
72
81
|
return arrayProblem(item.evalRows, `${path}.evalRows`, (row, rowPath) => {
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import type { AttemptListItem, DeltaData,
|
|
1
|
+
import type { AttemptListItem, DeltaData, ExperimentListItem, LineData, MatrixData, MetricColumn, ScatterData, ScopeSummaryData, ScoreboardData, TableData } from "../model/types.ts";
|
|
2
2
|
export declare const passRateColumn: MetricColumn;
|
|
3
3
|
export declare const codeLinesColumn: MetricColumn;
|
|
4
4
|
export declare const costColumn: MetricColumn;
|
|
@@ -11,5 +11,4 @@ export declare const scatterData: ScatterData;
|
|
|
11
11
|
export declare const lineData: LineData;
|
|
12
12
|
export declare const deltaData: DeltaData;
|
|
13
13
|
export declare const attemptListItems: AttemptListItem[];
|
|
14
|
-
export declare const evalListItems: EvalListItem[];
|
|
15
14
|
export declare const experimentListItems: ExperimentListItem[];
|