niceeval 0.10.3-canary.6 → 0.11.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/bin/niceeval.js +9 -4
- package/dist/agents/types.d.ts +5 -4
- package/dist/context/turn-errors.d.ts +27 -23
- package/dist/i18n/en.d.ts +15 -0
- package/dist/i18n/en.js +16 -1
- package/dist/i18n/zh-CN.d.ts +16 -1
- package/dist/i18n/zh-CN.js +16 -1
- package/dist/o11y/derive.js +28 -24
- package/dist/o11y/types.d.ts +8 -5
- package/dist/report/components/attempt-detail/UsageTable.js +4 -7
- package/dist/report/components/attempt-detail/compute.d.ts +2 -2
- package/dist/report/components/attempt-detail/compute.js +2 -6
- package/dist/report/components/attempt-detail/faces.js +7 -7
- package/dist/report/components/attempt-detail/index.js +0 -3
- package/dist/report/components/entity-lists/EvalList.js +0 -0
- package/dist/report/components/metric-views/compute.js +1 -1
- package/dist/report/model/types.d.ts +3 -4
- package/dist/results/locator.js +0 -0
- package/dist/results/select.d.ts +6 -0
- package/dist/results/select.js +8 -0
- package/dist/runner/feedback/sink.d.ts +26 -1
- package/dist/runner/fingerprint.d.ts +23 -0
- package/dist/runner/types.d.ts +105 -7
- package/dist/sandbox/errors.d.ts +29 -0
- package/dist/sandbox/resolve.d.ts +9 -0
- package/dist/shared/failure-class.d.ts +91 -0
- package/dist/types.d.ts +1 -0
- package/dist/util.d.ts +3 -2
- package/dist/util.js +31 -5
- package/docs-site/zh/explanation/runner.mdx +35 -0
- package/docs-site/zh/reference/cli.mdx +10 -1
- package/docs-site/zh/reference/events.mdx +4 -4
- package/docs-site/zh/troubleshooting/debugging.mdx +11 -0
- package/docs-site/zh/tutorials/viewing-results.mdx +35 -3
- package/package.json +28 -24
- package/src/agents/ai-sdk.test.ts +26 -0
- package/src/agents/ai-sdk.ts +7 -4
- package/src/agents/index.ts +5 -3
- package/src/agents/langgraph.test.ts +30 -0
- package/src/agents/langgraph.ts +5 -2
- package/src/agents/openai-compat.test.ts +35 -0
- package/src/agents/openai-compat.ts +16 -4
- package/src/agents/sdk-streams.test.ts +66 -0
- package/src/agents/sdk-streams.ts +3 -1
- package/src/agents/types.ts +5 -4
- package/src/cli.ts +81 -16
- package/src/context/context.test.ts +34 -0
- package/src/context/context.ts +14 -5
- package/src/context/send-retry.test.ts +86 -0
- package/src/context/send-retry.ts +37 -12
- package/src/context/session.ts +24 -0
- package/src/context/turn-errors.test.ts +124 -17
- package/src/context/turn-errors.ts +60 -50
- package/src/define.ts +5 -0
- package/src/i18n/en.ts +22 -1
- package/src/i18n/zh-CN.ts +22 -1
- package/src/index.ts +8 -0
- package/src/o11y/cost.ts +4 -2
- package/src/o11y/derive.test.ts +40 -0
- package/src/o11y/derive.ts +28 -22
- package/src/o11y/otlp/sandbox-receiver.test.ts +201 -0
- package/src/o11y/otlp/sandbox-receiver.ts +73 -27
- package/src/o11y/parsers/bub.test.ts +30 -0
- package/src/o11y/parsers/bub.ts +5 -2
- package/src/o11y/parsers/codex.test.ts +19 -0
- package/src/o11y/parsers/codex.ts +5 -2
- package/src/o11y/types.ts +8 -5
- package/src/report/components/attempt-detail/UsageTable.tsx +4 -6
- package/src/report/components/attempt-detail/attempt-components.test.tsx +5 -8
- package/src/report/components/attempt-detail/compute.ts +2 -7
- package/src/report/components/attempt-detail/faces.ts +7 -7
- package/src/report/components/attempt-detail/index.tsx +0 -3
- package/src/report/components/entity-lists/EvalList.tsx +0 -0
- package/src/report/components/metric-views/compute.ts +1 -1
- package/src/report/model/types.ts +3 -4
- package/src/results/format.ts +9 -2
- package/src/results/index.ts +2 -0
- package/src/results/locator.ts +0 -0
- package/src/results/open.ts +132 -21
- package/src/results/select.ts +12 -0
- package/src/results/skipped-notice.ts +0 -0
- package/src/runner/attempt.test.ts +116 -0
- package/src/runner/attempt.ts +89 -8
- package/src/runner/discover.ts +21 -3
- package/src/runner/feedback/coordinator.ts +24 -2
- package/src/runner/feedback/eval-conclusions.ts +6 -3
- package/src/runner/feedback/human.test.ts +300 -6
- package/src/runner/feedback/human.ts +132 -30
- package/src/runner/feedback/json.test.ts +127 -2
- package/src/runner/feedback/json.ts +51 -3
- package/src/runner/feedback/reducer.test.ts +328 -29
- package/src/runner/feedback/reducer.ts +84 -9
- package/src/runner/feedback/sink.ts +39 -1
- package/src/runner/fingerprint.ts +49 -19
- package/src/runner/gate-lease.test.ts +510 -0
- package/src/runner/gate-lease.ts +350 -0
- package/src/runner/lock.test.ts +454 -0
- package/src/runner/lock.ts +288 -0
- package/src/runner/report.test.ts +1 -0
- package/src/runner/run.test.ts +2044 -9
- package/src/runner/run.ts +825 -61
- package/src/runner/teardown-registry.ts +20 -78
- package/src/runner/types.ts +103 -7
- package/src/sandbox/errors.test.ts +72 -0
- package/src/sandbox/errors.ts +83 -0
- package/src/sandbox/keep-registry.ts +22 -46
- package/src/sandbox/resolve.test.ts +100 -0
- package/src/sandbox/resolve.ts +84 -27
- package/src/shared/entry-file-store.test.ts +149 -0
- package/src/shared/entry-file-store.ts +117 -0
- package/src/shared/failure-class.test.ts +137 -0
- package/src/shared/failure-class.ts +175 -0
- package/src/show/index.ts +5 -6
- package/src/show/render.test.ts +116 -15
- package/src/show/render.ts +124 -35
- package/src/types.ts +9 -0
- package/src/util.ts +31 -4
- package/src/view/app/App.tsx +5 -1
- package/src/view/client-dist/app.js +1 -1
- package/src/view/data.ts +5 -9
- package/src/view/shared/types.ts +7 -1
- package/src/view/view-report.test.ts +54 -0
|
@@ -7,8 +7,13 @@ import type { AttemptLocator } from "../../results/locator.ts";
|
|
|
7
7
|
export interface DiagnosticInput {
|
|
8
8
|
/** 稳定去重 key —— 同一种 warning/error 用同一个 key(见 cli.md「同一 dedupeKey 并发出现时
|
|
9
9
|
* 只留一条并显示次数」),不要把可变的实例细节(如具体 sandbox id)编进 key 本身,
|
|
10
|
-
* 那些细节放 `data
|
|
10
|
+
* 那些细节放 `data`。折叠身份(实验 / eval)可以编进 key,那是「折叠到多细」的表达;
|
|
11
|
+
* 对外展示的稳定词由 `code` 单独给,不从 key 反推。 */
|
|
11
12
|
key: string;
|
|
13
|
+
/** 对外的稳定词法:`--json` 的 `warning.code`、human 诊断行的标题都读它(见 cli.md
|
|
14
|
+
* `WarningEvent`,如 `lock-taken-over` / `dispatch-halted`)。省略 = 与 `key` 相同——
|
|
15
|
+
* 折叠身份不进 key 的那些诊断天生就是干净字面量,不必重复写一遍。 */
|
|
16
|
+
code?: string;
|
|
12
17
|
severity: "warning" | "error";
|
|
13
18
|
/** 一句话人类可读摘要;renderer 的 appendDurable 直接展示,不需要再解析。 */
|
|
14
19
|
message: string;
|
|
@@ -52,6 +57,21 @@ export interface PrecheckInput {
|
|
|
52
57
|
status: "started" | "done";
|
|
53
58
|
durationMs?: number;
|
|
54
59
|
}
|
|
60
|
+
/** `sink.lockWait()` 的输入 —— 与 `DurableFeedbackEvent` 的 "lock-wait" 变体字段一致,省略
|
|
61
|
+
* `type`/`at`(由 coordinator 补上)。调用方(run.ts)只在真正探测到撞上新鲜锁、需要等待时才
|
|
62
|
+
* 调一次 "started"(取锁立即成功或接管成功都不调用这个函数);等待解决(锁释放/接管后重查
|
|
63
|
+
* 携带完毕)时调一次 "resolved"。 */
|
|
64
|
+
export interface LockWaitInput {
|
|
65
|
+
experimentId: string;
|
|
66
|
+
evalId: string;
|
|
67
|
+
status: "started" | "resolved";
|
|
68
|
+
holderPid?: number;
|
|
69
|
+
holderHost?: string;
|
|
70
|
+
attempts?: number;
|
|
71
|
+
carried?: number;
|
|
72
|
+
dispatched?: number;
|
|
73
|
+
waitedMs?: number;
|
|
74
|
+
}
|
|
55
75
|
/** `sink.kept()` 的输入 —— 与 `DurableFeedbackEvent` 的 "kept" 变体字段一致,省略 type/at。 */
|
|
56
76
|
export interface KeptInput {
|
|
57
77
|
locator: AttemptLocator;
|
|
@@ -84,6 +104,8 @@ export interface FeedbackSink {
|
|
|
84
104
|
experimentHook(input: ExperimentHookInput): void;
|
|
85
105
|
/** judge 预检的起止(见 `PrecheckInput`)。整次运行至多一次,发生在任何 attempt 派发之前。 */
|
|
86
106
|
precheck(input: PrecheckInput): void;
|
|
107
|
+
/** 用例锁等待的起止(见 `LockWaitInput`)。 */
|
|
108
|
+
lockWait(input: LockWaitInput): void;
|
|
87
109
|
/** 实验级 `ctx.progress` 的短命投影:只更新运行级行的 detail。与 `lifecycle` 同级别的
|
|
88
110
|
* 「只服务正在画着的 dashboard」信号,没有活跃 coordinator 时静默丢弃是安全的。 */
|
|
89
111
|
experimentProgress(input: ExperimentProgressInput): void;
|
|
@@ -123,6 +145,9 @@ export declare function reportExperimentProgress(input: ExperimentProgressInput)
|
|
|
123
145
|
/** judge 预检的起止(见 `PrecheckInput`)。没有活跃 coordinator 时退回一行 stderr ——
|
|
124
146
|
* 慢预检的可见性正是这条通道存在的理由,不能像 lifecycle 那样静默丢弃。 */
|
|
125
147
|
export declare function reportPrecheck(input: PrecheckInput): void;
|
|
148
|
+
/** 用例锁等待的起止(见 `LockWaitInput`)。没有活跃 coordinator 时退回一行 stderr ——
|
|
149
|
+
* 长等待期间的可见性正是这条通道存在的理由,不能像 lifecycle 那样静默丢弃。 */
|
|
150
|
+
export declare function reportLockWait(input: LockWaitInput): void;
|
|
126
151
|
export declare function reportBudgetExhausted(input: BudgetExhaustedInput): void;
|
|
127
152
|
/** 用户中断(Ctrl+C)。没有活跃 coordinator 时的兜底文案与迁移前的 `runner.interrupted` 完全
|
|
128
153
|
* 相同 —— 调用方(run.ts)不再需要自己持有这段 i18n 文案。 */
|
|
@@ -20,6 +20,29 @@ export interface CarryPlan {
|
|
|
20
20
|
/** carriedAttemptsByKey 对应的完整结果对象,供 run.ts 直接并入 summary、cli.ts 直接取 verdict 展示。 */
|
|
21
21
|
carriedResults: EvalResult[];
|
|
22
22
|
}
|
|
23
|
+
/**
|
|
24
|
+
* 与 `attempt.ts` 里实际生效超时的解析顺序保持一致(`run.timeoutMs ?? evalDef.timeoutMs ??
|
|
25
|
+
* configTimeoutMs`),但刻意不叠加 attempt.ts 的硬编码兜底(10 分钟):携带判据要问的是「用户
|
|
26
|
+
* 有没有显式配置一条线」,三层都未设时线本身不存在,`Infinity` 让 durationMs 判据恒成立
|
|
27
|
+
* (「当前未设上限 = 恒可携带」,见 docs/runner.md「缓存:指纹去重」)。10 分钟兜底是
|
|
28
|
+
* attempt.ts 的执行期默认值,不是携带判据要遵守的线。
|
|
29
|
+
*/
|
|
30
|
+
export declare function resolvedTimeoutMsForCarry(run: AgentRun, evalDef: DiscoveredEval, configTimeoutMs?: number): number;
|
|
31
|
+
/**
|
|
32
|
+
* 携带资格判据的**唯一**实现:从 `priorResults` 里挑出 `key` 这条 `(experimentId, evalId)`
|
|
33
|
+
* 可以携入(跳过重跑)的 attempt。三条判据逐条 attempt 独立成立才算命中——
|
|
34
|
+
*
|
|
35
|
+
* 1. 该 attempt 自己是终态(`passed` / `failed`)。`errored` 是框架/环境层面的不确定失败,
|
|
36
|
+
* 判定本身不可信;`skipped` 根本没跑。同一 eval 的别的序号命中不能连带把它捎上
|
|
37
|
+
* (反例与修法见 memory 的 carry-must-be-per-attempt-not-whole-eval-key)。
|
|
38
|
+
* 2. 该 attempt 落盘的 `fingerprint` 与本次规划的 `fingerprint` 相等。
|
|
39
|
+
* 3. 该 attempt 的 `durationMs` 不超过本次 resolved 的 `timeoutMs`——`timeoutMs` 是携带资格
|
|
40
|
+
* 判据、不进指纹哈希(docs/runner.md「缓存:指纹去重」)。
|
|
41
|
+
*
|
|
42
|
+
* `planCarry`(整场静态规划)与 run.ts 派发时刻的携带重查共用这一个函数:两条路径一旦把判据
|
|
43
|
+
* 各写一份就会分叉,重查会携入静态规划判过不可携带的条目(或反过来)。
|
|
44
|
+
*/
|
|
45
|
+
export declare function carriableAttempts(priorResults: EvalResult[] | undefined, key: string, fingerprint: string | undefined, timeoutMs: number): EvalResult[];
|
|
23
46
|
/**
|
|
24
47
|
* 算出这一批 (agentRun × eval) 的指纹,并据此从 priorResults 里筛出可以携入(跳过重跑)的结果。
|
|
25
48
|
* run.ts 与 cli.ts(live 表格构建)必须共用这同一份计算 —— 否则两边一旦对"哪些携入"的判断
|
package/dist/runner/types.d.ts
CHANGED
|
@@ -1,4 +1,5 @@
|
|
|
1
1
|
import type { JsonValue, LocalizedText, ScopedFeedback, SourceArtifact } from "../shared/types.ts";
|
|
2
|
+
import type { AttemptFailureClassifier } from "../shared/failure-class.ts";
|
|
2
3
|
import type { O11ySummary, StreamEvent, TraceSpan, Truncation, Usage } from "../o11y/types.ts";
|
|
3
4
|
import type { Agent, AgentSetupManifest } from "../agents/types.ts";
|
|
4
5
|
import type { Sandbox, SandboxHookContext, SandboxOption } from "../sandbox/types.ts";
|
|
@@ -542,6 +543,15 @@ export interface ExperimentDef {
|
|
|
542
543
|
* `maxConcurrency: 1` 因此是严格的临界区,不会被同实验的下一个 attempt 提前闯入。
|
|
543
544
|
*/
|
|
544
545
|
maxConcurrency?: number;
|
|
546
|
+
/**
|
|
547
|
+
* 本实验的失败分类器:识别以第三方错误形态浮出的自家共享基建死因(对自家隧道 host 的拒连
|
|
548
|
+
* 一类),返回 `undefined` 表示「不认识,交给后续链路」。本实验任意 per-attempt 阶段的失败
|
|
549
|
+
* 都会问到它;turn 失败链上它排在 adapter 的 `classifyTurnError` 之前——按自家坐标过滤的
|
|
550
|
+
* 特异性高于协议通用形状,两者同时认领时空间轴才赢得下来。分类器要快、纯、不抛错(抛错按
|
|
551
|
+
* `undefined` 回落并被吞掉);只声明决策轴与 `reason` 词,重试与落闸策略归执行体。
|
|
552
|
+
* 见 docs/feature/error-classification/library.md「实验 / eval 作者:声明死因的波及范围」。
|
|
553
|
+
*/
|
|
554
|
+
classifyFailure?: AttemptFailureClassifier;
|
|
545
555
|
/**
|
|
546
556
|
* 实验级生命周期钩子对的 setup 侧:整场至多一次、宿主机侧,管「每实验一份、所有 attempt
|
|
547
557
|
* 共享」的宿主机资源(隧道、mock server、license 租约)。本实验第一个通过派发许可的
|
|
@@ -689,6 +699,9 @@ export interface AgentRun {
|
|
|
689
699
|
* (语义见 ExperimentDef 对应字段)。 */
|
|
690
700
|
setup?: (ctx: ExperimentHookContext) => void | Promise<void>;
|
|
691
701
|
teardown?: (ctx: ExperimentHookContext) => void | Promise<void>;
|
|
702
|
+
/** 实验声明的失败分类器(来自 ExperimentDef.classifyFailure):turn 链上排在 adapter 之前,
|
|
703
|
+
* 生命周期链上排在抛出点声明之后;产出的空间轴由止损闸在 attempt 封口消费。 */
|
|
704
|
+
classifyFailure?: AttemptFailureClassifier;
|
|
692
705
|
}
|
|
693
706
|
export interface RunOptions {
|
|
694
707
|
config: Config;
|
|
@@ -791,7 +804,7 @@ export type ExperimentHookName = "setup" | "teardown";
|
|
|
791
804
|
/**
|
|
792
805
|
* dashboard 当前可见的一个实验级钩子运行级行(见 docs/feature/experiments/cli.md
|
|
793
806
|
* 「实验级钩子的显示」)。与 `ActiveAttempt` 分开建模:钩子不属于任何单个 attempt、不占并发位,
|
|
794
|
-
* 也不参与 `
|
|
807
|
+
* 也不参与 `RunFeedbackState` 的计数不变量——等待 setup 的
|
|
795
808
|
* attempt 保持 `queued`,这行就是「为什么它们还在排队」的解释。`detail` 来自实验级
|
|
796
809
|
* `ctx.progress`,后一条覆盖前一条。
|
|
797
810
|
*/
|
|
@@ -813,6 +826,27 @@ export interface ActiveExperimentHook {
|
|
|
813
826
|
/** 强杀后启动自愈补执行的 teardown(见 `DurableFeedbackEvent` 的 "experiment-hook" 变体)。 */
|
|
814
827
|
recovery?: boolean;
|
|
815
828
|
}
|
|
829
|
+
/**
|
|
830
|
+
* dashboard 当前可见的一个「等待并行 run」运行级行(见 docs/feature/experiments/cli.md
|
|
831
|
+
* 「等待并发 run 的显示」)。用例锁的等待粒度是单个 `(experimentId, evalId)`,但运行级行按
|
|
832
|
+
* experimentId 聚合展示——一个实验可能同时有多个用例撞锁,只占一行,给出条数与代表持有方。
|
|
833
|
+
*/
|
|
834
|
+
export interface ActiveLockWait {
|
|
835
|
+
experimentId: string;
|
|
836
|
+
/** 当前仍在等待的 evalId → 该用例开始等待的时间与持有方身份。`size` 就是运行级行要展示的
|
|
837
|
+
* 等待条数;为空表示这个实验当前没有在等的用例(条目仍保留在 map 里,供非 TTY 聚合文案
|
|
838
|
+
* 读取下面两个累计字段,直到下一次 "started" 事件开启新窗口时清零)。 */
|
|
839
|
+
waiting: ReadonlyMap<string, {
|
|
840
|
+
startedAt: number;
|
|
841
|
+
holderPid?: number;
|
|
842
|
+
holderHost?: string;
|
|
843
|
+
}>;
|
|
844
|
+
/** 本次「有等待用例」窗口内,累计已经 resolved 且携入 reused 的 attempt 数——供非 TTY 聚合
|
|
845
|
+
* 收尾行(如 `lock wait resolved · compare/codex (2 carried · 1 to run, 1m 34s)`)读取。 */
|
|
846
|
+
resolvedCarried: number;
|
|
847
|
+
/** 同上,累计已经 resolved 且转为自跑(进入 queued)的 attempt 数。 */
|
|
848
|
+
resolvedDispatched: number;
|
|
849
|
+
}
|
|
816
850
|
/**
|
|
817
851
|
* 一次失败/错误的永久通知:human 撤下 dashboard 后追加一行、agent/ci 立即追加一行,都读它。
|
|
818
852
|
* 字段全部结构化(locator / identity / verdict / phase 都是具名字段),profile renderer 不需要
|
|
@@ -840,9 +874,18 @@ export interface FailureNotice extends FailureDetail {
|
|
|
840
874
|
* `data` 携带结构化字段(如 budget 的 experimentId/spent/unstarted),agent/ci 直接读取,
|
|
841
875
|
* 不解析 `message`(`message` 只是 human 展示用的一句话)。
|
|
842
876
|
*/
|
|
877
|
+
/**
|
|
878
|
+
* 止损闸落闸诊断的稳定词法(`--json` 的 `warning.code`、`snapshot.json` 的诊断 `code`,契约见
|
|
879
|
+
* docs/feature/error-classification/architecture.md「止损执行体」)。emitter(run.ts)与两种
|
|
880
|
+
* profile 的 renderer 共用这一个常量,谁都不在自己这边再写一遍字面量。
|
|
881
|
+
*/
|
|
882
|
+
export declare const HALT_DIAGNOSTIC_CODE = "dispatch-halted";
|
|
843
883
|
export interface DiagnosticNotice {
|
|
844
884
|
at: number;
|
|
845
885
|
key: string;
|
|
886
|
+
/** 对外的稳定词法(`--json` 的 `warning.code`、human 诊断行标题);省略 = 与 `key` 相同。
|
|
887
|
+
* `key` 可以把折叠身份(experimentId / evalId)编进去,`code` 恒是干净字面量。 */
|
|
888
|
+
code?: string;
|
|
846
889
|
severity: "warning" | "error";
|
|
847
890
|
message: string;
|
|
848
891
|
/** 相同 key 累计出现的次数,由 reducer 去重时递增。 */
|
|
@@ -872,15 +915,32 @@ export interface InvocationCompletion {
|
|
|
872
915
|
* cost 累计、failure/diagnostic 去重都只在 reducer 里算一次;三种 profile 的 renderer 只读取
|
|
873
916
|
* 这份状态,不各自维护第二份推导。
|
|
874
917
|
*
|
|
875
|
-
* `total = reused + running + queued +
|
|
876
|
-
*
|
|
918
|
+
* `total = reused + running + elsewhere + queued + passed + failed + errored + skipped`
|
|
919
|
+
* (八项恒等式,见 docs/feature/experiments/cli.md「等待并发 run 的显示」)在处理完每一个事件
|
|
920
|
+
* 之后都成立,是 reducer 的不变量:任何一次迁移都是「从一项减 x、往另一项加 x」,不存在两项
|
|
921
|
+
* 同时计数或都不计数的中间态(见 reducer.test.ts 的表驱动用例,每一步都断言,不只在流程末尾
|
|
922
|
+
* 断言一次)。
|
|
877
923
|
*/
|
|
878
924
|
export interface RunFeedbackState {
|
|
879
925
|
total: number;
|
|
880
926
|
reused: number;
|
|
881
927
|
running: number;
|
|
928
|
+
/** 正被并行 Invocation 持锁运行、本次在等待中的用例的 attempt 数(用例锁,见 `lock-wait`
|
|
929
|
+
* 变体与 docs/feature/experiments/cli.md「等待并发 run 的显示」);与 `queued` 互斥——
|
|
930
|
+
* `queued` 是「等本进程并发位/setup」,`elsewhere` 是「等别的进程」。
|
|
931
|
+
* 恒等式(见接口注释)在处理完每一个事件之后都成立。 */
|
|
932
|
+
elsewhere: number;
|
|
882
933
|
queued: number;
|
|
883
|
-
|
|
934
|
+
/** 以下四项是本次派发并已了结的 attempt 按 verdict 的划分——reducer 不保留一个笼统的
|
|
935
|
+
* 「完成数」:盯着运行的人问的是「到现在为止挂了几个」,一个合计数回答不了。携入结果的
|
|
936
|
+
* verdict 留在 `reused`,不摊进这四项(计数口径与成本口径一致地区分「本次派发」与
|
|
937
|
+
* 「缓存携入」,见 docs/feature/experiments/cli.md「运行中的 live 面板」)。 */
|
|
938
|
+
passed: number;
|
|
939
|
+
failed: number;
|
|
940
|
+
errored: number;
|
|
941
|
+
/** 本次不产生 verdict 的了结:eval 自身 skip、首过即停省略的轮次、budget 未派发。
|
|
942
|
+
* 它们不冒充 `passed`/`failed`;三者彼此的区别由结束结论与题目级 `eval` 事件给出。 */
|
|
943
|
+
skipped: number;
|
|
884
944
|
/** attempt:early-exit 事件的累计次数(首过即停省略 + fail-fast 未派发;后者由 fail-fast
|
|
885
945
|
* diagnostic 的 count 单独区分,见 cli.ts 的 assembleRunCompletion)。 */
|
|
886
946
|
earlyExitSkipped: number;
|
|
@@ -899,12 +959,15 @@ export interface RunFeedbackState {
|
|
|
899
959
|
active: ReadonlyMap<AttemptKey, ActiveAttempt>;
|
|
900
960
|
/** 在飞的 judge 预检运行级行(见 `DurableFeedbackEvent` 的 "precheck" 变体):`started` 置位、
|
|
901
961
|
* `done` 清空。预检发生在任何 attempt 派发之前、作用于整次 invocation,不属于任何 attempt,
|
|
902
|
-
*
|
|
962
|
+
* 也不参与五项恒等式计数——预检期间 attempt 保持 `queued`,
|
|
903
963
|
* 这行就是「为什么它们还在排队」的解释。undefined = 当前没有在飞的预检。 */
|
|
904
964
|
activePrecheck?: ActivePrecheck;
|
|
905
965
|
/** 在飞的实验级钩子(experimentId → 运行级行状态),由 "experiment-hook" 事件增删、
|
|
906
966
|
* "experiment:progress" 更新 detail(见 docs/feature/experiments/cli.md「实验级钩子的显示」)。 */
|
|
907
967
|
experimentHooks: ReadonlyMap<string, ActiveExperimentHook>;
|
|
968
|
+
/** 在飞的用例锁等待,按 experimentId 聚合(见 `ActiveLockWait`、docs/feature/experiments/cli.md
|
|
969
|
+
* 「等待并发 run 的显示」)。由 "lock-wait" 事件增删/累计;没有等待用例的实验不出现在这个 map 里。 */
|
|
970
|
+
lockWaits: ReadonlyMap<string, ActiveLockWait>;
|
|
908
971
|
failures: readonly FailureNotice[];
|
|
909
972
|
/** 本次实际派发后产生的去重失败数;复用失败不消耗 profile 的流式输出上限。 */
|
|
910
973
|
freshFailureCount: number;
|
|
@@ -935,7 +998,7 @@ export interface RunFeedbackPlan {
|
|
|
935
998
|
* 只影响 dashboard 当前帧、reducer 不为它保留历史的事件:新值使旧值失去意义,所以覆盖而不是
|
|
936
999
|
* 追加(见 docs/feature/experiments/cli.md「什么动态更新,什么逐条追加」的判断标准)。
|
|
937
1000
|
* `attempt:early-exit` 同样折进这一组 —— 它不打印永久行,只把已知 verdict 的省略次数收进
|
|
938
|
-
* `
|
|
1001
|
+
* `skipped`(见 reducer 实现)。
|
|
939
1002
|
*/
|
|
940
1003
|
export type AttemptLifecycleEvent = {
|
|
941
1004
|
type: "attempt:queued";
|
|
@@ -1016,8 +1079,13 @@ export type DurableFeedbackEvent = {
|
|
|
1016
1079
|
type: "diagnostic";
|
|
1017
1080
|
at: number;
|
|
1018
1081
|
key: string;
|
|
1082
|
+
/** 对外稳定词法(见 `DiagnosticNotice.code`);省略 = 与 `key` 相同。 */
|
|
1083
|
+
code?: string;
|
|
1019
1084
|
severity: "warning" | "error";
|
|
1020
1085
|
message: string;
|
|
1086
|
+
/** attempt 级诊断的归属身份。运行级诊断(实验闸 / eval 闸这类不属于任何单条 attempt 的
|
|
1087
|
+
* 事实)不许伪造 identity——它们的 experimentId / evalId 走 `data` 的同名字段,
|
|
1088
|
+
* `--json` 的 `warning` 事件两处都读、identity 优先。 */
|
|
1021
1089
|
identity?: AttemptRef;
|
|
1022
1090
|
data?: Readonly<Record<string, JsonValue>>;
|
|
1023
1091
|
}
|
|
@@ -1070,7 +1138,7 @@ export type DurableFeedbackEvent = {
|
|
|
1070
1138
|
/**
|
|
1071
1139
|
* judge 配置预检的起止,由 runner 在探测真正开始/结束时各发一次(见 docs/feature/experiments/
|
|
1072
1140
|
* cli.md「judge 预检的显示」)。预检作用于整次 invocation、发生在任何 attempt 派发之前,不属于
|
|
1073
|
-
* 任何单个 attempt
|
|
1141
|
+
* 任何单个 attempt,也不触碰五项恒等式计数不变量。human
|
|
1074
1142
|
* TTY 用它维护一条运行级 active 行(不写 scrollback),append-only profile 起止各追加一行。
|
|
1075
1143
|
* 预检失败不走这个事件——它以既有错误路径中止本次运行。
|
|
1076
1144
|
*/
|
|
@@ -1079,6 +1147,36 @@ export type DurableFeedbackEvent = {
|
|
|
1079
1147
|
at: number;
|
|
1080
1148
|
status: "started" | "done";
|
|
1081
1149
|
durationMs?: number;
|
|
1150
|
+
}
|
|
1151
|
+
/**
|
|
1152
|
+
* 用例锁等待的起止(见 docs/feature/experiments/cli.md「等待并发 run 的显示」)。粒度是单个
|
|
1153
|
+
* `(experimentId, evalId)`——同一 eval 的全部 attempt 作为一个整体一起等、一起解决,不按
|
|
1154
|
+
* attempt 拆分。emitter(run.ts)只在这批 attempt 需要重查携带时才发这对事件:全携带用例
|
|
1155
|
+
* 不取锁、无竞争的全新取锁(锁目录里从没出现过这个 key)都不发——静态携带规划的结论不可能
|
|
1156
|
+
* 过时,没有理由重新读盘。撞上新鲜锁(真正等待)与接管一把无人竞争的过期锁(从未真正等待,
|
|
1157
|
+
* `waitedMs` 因此可能接近 0)都算「需要重查」,统一走这对事件——即便是瞬时接管,这批
|
|
1158
|
+
* attempt 也必须先经 "started" 迁入 `elsewhere`,"resolved" 才能把它们正确迁回
|
|
1159
|
+
* `reused`/`queued`,否则它们会永远卡在 `queued`、打破五项恒等式。
|
|
1160
|
+
*/
|
|
1161
|
+
| {
|
|
1162
|
+
type: "lock-wait";
|
|
1163
|
+
at: number;
|
|
1164
|
+
experimentId: string;
|
|
1165
|
+
evalId: string;
|
|
1166
|
+
status: "started" | "resolved";
|
|
1167
|
+
/** status 为 "started" 时给出:锁持有方身份,以及这次撞锁进入 elsewhere 等待的 attempt 数
|
|
1168
|
+
* (该 eval 本轮需要真实派发、被这把锁挡住的 attempt 数;省略按 1 处理)。 */
|
|
1169
|
+
holderPid?: number;
|
|
1170
|
+
holderHost?: string;
|
|
1171
|
+
attempts?: number;
|
|
1172
|
+
/** status 为 "resolved" 时给出:锁释放后重查携带,分别有多少 attempt 从 elsewhere 迁入
|
|
1173
|
+
* reused(carried)、多少迁入 queued 转为自跑(dispatched)——`runs` 下可能两者都非零
|
|
1174
|
+
* (部分携入部分补跑)。`--json` 的 `lock_wait` 事件把两者折成单一 `resolution` 字段:
|
|
1175
|
+
* `dispatched > 0` 记 "dispatched"(这个用例仍需要真实派发,等待没有让它完全免于执行),
|
|
1176
|
+
* 否则记 "carried"(全部由携带满足,零新成本)。 */
|
|
1177
|
+
carried?: number;
|
|
1178
|
+
dispatched?: number;
|
|
1179
|
+
waitedMs?: number;
|
|
1082
1180
|
} | {
|
|
1083
1181
|
type: "interrupted";
|
|
1084
1182
|
at: number;
|
package/dist/sandbox/errors.d.ts
CHANGED
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
import { type FailureScope } from "../shared/failure-class.ts";
|
|
1
2
|
/**
|
|
2
3
|
* Provisioning 失败的两维分类(见 docs/feature/sandbox/architecture.md「Provisioning 失败与重试」):
|
|
3
4
|
* **性质**(瞬时 / 确定性)决定要不要重试,**后果**(远端是否可能已创建实例)决定能不能直接重试。
|
|
@@ -29,3 +30,31 @@ export declare function isRetryableSandboxIoError(kind: SandboxIoErrorKind): boo
|
|
|
29
30
|
* `cause` 中,因此最多沿 cause 链向下检查几层;Abort/沙箱终止明确不重试。
|
|
30
31
|
*/
|
|
31
32
|
export declare function classifySandboxIoError(error: unknown): SandboxIoErrorKind;
|
|
33
|
+
/**
|
|
34
|
+
* 确定性 provisioning 死因的配置解析域细分(见
|
|
35
|
+
* docs/feature/sandbox/architecture.md「Provisioning 失败与重试」「对外的空间轴映射」)。
|
|
36
|
+
* 只在 provider 自身的 `classifyProvisionError` 判定为 `"unknown"`(性质轴:确定性,重试
|
|
37
|
+
* 没有意义)之后调用一次,进一步细分成三档可证明死因:凭据缺失、权限不足、模板不存在。
|
|
38
|
+
* 认不出的确定性错误(参数非法、通用 SDK 错误等)返回 `undefined`——不附带 scope 比误判
|
|
39
|
+
* 安全,悬空错误照常落成本 attempt `errored`,不触发实验 / eval 级止损。
|
|
40
|
+
*/
|
|
41
|
+
export type ProvisionConfigCause = "credentials" | "permission" | "template_not_found";
|
|
42
|
+
export declare function classifyProvisionConfigCause(error: unknown): ProvisionConfigCause | undefined;
|
|
43
|
+
/**
|
|
44
|
+
* 三档死因 → 空间轴映射,按配置解析域定档(判据单源见
|
|
45
|
+
* docs/feature/sandbox/architecture.md#provisioning-失败与重试):凭据缺失 / 权限不足
|
|
46
|
+
* 来自实验级配置,恒 `"experiment"`;模板不存在按 spec 是否带 `environments` 表二分——
|
|
47
|
+
* 带表时模板逐 eval 解析,可证明的档收紧到 `"eval"`(错杀健康模板的 eval 比多撞几次死
|
|
48
|
+
* 模板更贵);不带表时模板全实验共享,`"experiment"`。
|
|
49
|
+
*/
|
|
50
|
+
export declare function provisionConfigCauseScope(cause: ProvisionConfigCause, hasEnvironmentsTable: boolean): FailureScope;
|
|
51
|
+
/**
|
|
52
|
+
* 把可证明的配置死因原地附着到 provisioning 错误对象上,供 `failureClassOf` 沿 cause 链
|
|
53
|
+
* 识别(第一道「抛出点携带的分类」即命中)。只用于确定性失败(`retryable: false`);瞬时
|
|
54
|
+
* 失败重试耗尽后不调用这个函数——死因不可证明为兄弟共享,原样抛出不带 scope。
|
|
55
|
+
*
|
|
56
|
+
* 附着走 shared 层同一个 `attachFailureClass`,不在这里重写一份:两个标记字段必须不可枚举
|
|
57
|
+
* (不进 JSON / console,只是路由标记),已携带分类的对象不覆盖,冻结对象静默跳过——这三条
|
|
58
|
+
* 纪律单源在那边,复制一份迟早漂移。shared 是中性层,不是 context,沙箱依赖它不构成反向依赖。
|
|
59
|
+
*/
|
|
60
|
+
export declare function attachProvisionFailureScope(error: unknown, scope: FailureScope): void;
|
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
import { Effect } from "effect";
|
|
2
2
|
import type { CustomSandboxSpec, JsonValue, Sandbox, SandboxOption, SandboxRuntime, ScopedFeedback } from "../types.ts";
|
|
3
3
|
import { type ProvisionSlot } from "./retry.ts";
|
|
4
|
+
import { type SandboxProvisionErrorKind } from "./errors.ts";
|
|
4
5
|
/** 归一化后的沙箱描述:确定的 provider + 各 provider 参数(只有对应 provider 用得上的会有值)。 */
|
|
5
6
|
export interface ResolvedSandbox {
|
|
6
7
|
provider: string;
|
|
@@ -61,3 +62,11 @@ export declare function createSandbox(opts: {
|
|
|
61
62
|
*/
|
|
62
63
|
release?: (sb: Sandbox) => Promise<void>;
|
|
63
64
|
}): Effect.Effect<Sandbox, never, import("effect/Scope").Scope>;
|
|
65
|
+
/**
|
|
66
|
+
* provisioning 失败向外浮出确定性配置死因的 scope(契约见
|
|
67
|
+
* docs/feature/sandbox/architecture.md#provisioning-失败与重试「对外的空间轴映射」)。
|
|
68
|
+
* `work` 失败后,只有 provider 自身分类判定为 `"unknown"`(确定性)时才进一步细分死因;
|
|
69
|
+
* 瞬时失败(拒绝类/歧义类)不论是否重试耗尽都原样抛出,不附带 scope——死因不可证明为
|
|
70
|
+
* 兄弟共享。导出供单测直接注入 `work`/`classify`,不需要经过真实 provider SDK。
|
|
71
|
+
*/
|
|
72
|
+
export declare function withDeterministicProvisionScope<T>(work: () => Promise<T>, classify: (e: unknown) => SandboxProvisionErrorKind, r: ResolvedSandbox): Promise<T>;
|
|
@@ -0,0 +1,91 @@
|
|
|
1
|
+
import type { LifecyclePhase } from "../runner/types.ts";
|
|
2
|
+
/** 空间轴取值:失败死因的波及范围。 */
|
|
3
|
+
export type FailureScope = "attempt" | "eval" | "experiment";
|
|
4
|
+
/**
|
|
5
|
+
* 一次执行失败的分类:`retryable`(时间轴)与 `scope`(空间轴)是仅有的两条决策轴;
|
|
6
|
+
* `reason` 是开放词表的细分诊断,只进 activity 与诊断文案,不参与策略。内建兜底产出
|
|
7
|
+
* reason `"rate_limit"` / `"network"`;声明方可自造词。`scope` 缺省 `"attempt"`。
|
|
8
|
+
*
|
|
9
|
+
* `retryable: true` 时 `reason` 必填:可重试的失败一定出现在 activity 行与可能的耗尽摘要里,
|
|
10
|
+
* 那里需要一个给人读的词;不可重试的失败常常说不清是什么(这正是它不可重试的原因)。
|
|
11
|
+
*/
|
|
12
|
+
export type FailureClass = {
|
|
13
|
+
readonly retryable: true;
|
|
14
|
+
readonly reason: string;
|
|
15
|
+
readonly scope?: FailureScope;
|
|
16
|
+
} | {
|
|
17
|
+
readonly retryable: false;
|
|
18
|
+
readonly reason?: string;
|
|
19
|
+
readonly scope?: FailureScope;
|
|
20
|
+
};
|
|
21
|
+
/**
|
|
22
|
+
* 实验级分类器的输入:本实验任意 per-attempt 阶段的一次终局失败。
|
|
23
|
+
*/
|
|
24
|
+
export interface AttemptFailureInfo {
|
|
25
|
+
/** 失败发生在哪个生命周期阶段;turn 失败恒为 `"agent.run"`。 */
|
|
26
|
+
readonly phase: LifecyclePhase;
|
|
27
|
+
/** 与报错文案同源的失败文本:thrown 取错误链(含 cause 链)message 串接,turn 失败取 `turnErrorText`。 */
|
|
28
|
+
readonly text: string;
|
|
29
|
+
/** 原始失败对象:thrown 形态是抛出的错误,turn 失败形态是那个失败 Turn。 */
|
|
30
|
+
readonly cause: unknown;
|
|
31
|
+
}
|
|
32
|
+
/**
|
|
33
|
+
* 实验可选分类器,挂载在 `ExperimentDef.classifyFailure`:识别自家共享基建的死因
|
|
34
|
+
* (对自家隧道 host 的拒连一类)。返回 `undefined` 表示「不认识,交给后续链路」。
|
|
35
|
+
* 分类器必须快、纯、不抛错——抛错按 `undefined` 回落并被吞掉,不掩盖原始失败。
|
|
36
|
+
*/
|
|
37
|
+
export type AttemptFailureClassifier = (failure: AttemptFailureInfo) => FailureClass | undefined;
|
|
38
|
+
/**
|
|
39
|
+
* 从任意 per-attempt 阶段抛出:全实验剩余 attempt 同因必死,停止派发。
|
|
40
|
+
* 携带 `{ retryable: false, scope: "experiment" }`,message 原样走完反馈流与
|
|
41
|
+
* `dispatch-halted` 诊断——把它写成「现象 + 下一步」的修复提示。
|
|
42
|
+
*/
|
|
43
|
+
export declare class ExperimentFatalError extends Error {
|
|
44
|
+
readonly _tag = "NiceevalClassifiedError";
|
|
45
|
+
readonly class: FailureClass;
|
|
46
|
+
constructor(message: string, options?: {
|
|
47
|
+
cause?: unknown;
|
|
48
|
+
});
|
|
49
|
+
}
|
|
50
|
+
/**
|
|
51
|
+
* 从任意 per-attempt 阶段抛出:本 eval 剩余 attempt 同因必死,停止派发。
|
|
52
|
+
* 携带 `{ retryable: false, scope: "eval" }`;message 同样是走完全程的修复提示。
|
|
53
|
+
*/
|
|
54
|
+
export declare class EvalFatalError extends Error {
|
|
55
|
+
readonly _tag = "NiceevalClassifiedError";
|
|
56
|
+
readonly class: FailureClass;
|
|
57
|
+
constructor(message: string, options?: {
|
|
58
|
+
cause?: unknown;
|
|
59
|
+
});
|
|
60
|
+
}
|
|
61
|
+
/**
|
|
62
|
+
* 结构守卫:识别任何携带分类的错误对象(`_tag` + `class` 两个数据字段),沿 `cause` 链逐层
|
|
63
|
+
* 查找、取**最外层**命中——糖衣类被上层库包装再抛时声明不丢失。识别不依赖 `instanceof`:
|
|
64
|
+
* 依赖树里出现第二份 niceeval 实例(link、版本重复)时类身份静默失效,数据不会。
|
|
65
|
+
*/
|
|
66
|
+
export declare function failureClassOf(error: unknown): FailureClass | undefined;
|
|
67
|
+
/**
|
|
68
|
+
* @internal 把框架决议出的分类挂到即将浮出的失败对象上,供 attempt 封口经 `failureClassOf`
|
|
69
|
+
* 读取(止损闸的消费点)。两个字段都不可枚举——不进 JSON、不进 console 输出,只是路由标记;
|
|
70
|
+
* 已携带分类的对象不覆盖(抛出点声明优先),不可写对象静默跳过(分类是旁路,不制造新失败)。
|
|
71
|
+
*/
|
|
72
|
+
export declare function attachFailureClass<T>(target: T, cls: FailureClass): T;
|
|
73
|
+
/**
|
|
74
|
+
* @internal 错误链(含 `cause` 链)的 message 串接:给人读的报错文案与给分类器看的失败文本
|
|
75
|
+
* 用这同一段,不出现「报错说 A、分类看 B」。
|
|
76
|
+
*/
|
|
77
|
+
export declare function errorChainText(error: unknown): string;
|
|
78
|
+
/**
|
|
79
|
+
* @internal 调分类器的统一纪律:抛错按 `undefined` 回落(自身错误被吞掉,不得用新错误掩盖
|
|
80
|
+
* 原始失败),返回值原样透出。turn 链与生命周期链共用这一处,两条链的纪律不会跑偏。
|
|
81
|
+
*/
|
|
82
|
+
export declare function callClassifier<T>(classifier: ((failure: T) => FailureClass | undefined) | undefined, failure: T): FailureClass | undefined;
|
|
83
|
+
/**
|
|
84
|
+
* @internal 生命周期阶段失败的分类链(三道):抛出点携带的分类 → 实验分类器 → 缺省
|
|
85
|
+
* `{ retryable: false }`。这些位置(sandbox 钩子、`EvalDef.setup`、`test(t)` 体内、
|
|
86
|
+
* per-attempt teardown)没有重试执行体,链上不挂产时间轴的兜底正则——时间轴即使给出也无人
|
|
87
|
+
* 消费(见 architecture.md「分类链」)。
|
|
88
|
+
*/
|
|
89
|
+
export declare function resolveAttemptFailureClass(info: AttemptFailureInfo, experimentClassifier?: AttemptFailureClassifier): FailureClass;
|
|
90
|
+
/** @internal 从一个抛出的错误构造实验分类器的输入(文本与报错文案同源)。 */
|
|
91
|
+
export declare function attemptFailureInfo(phase: LifecyclePhase, error: unknown): AttemptFailureInfo;
|
package/dist/types.d.ts
CHANGED
package/dist/util.d.ts
CHANGED
|
@@ -6,8 +6,9 @@ export declare function getEnv(name: string): string | undefined;
|
|
|
6
6
|
export declare function stripComments(code: string): string;
|
|
7
7
|
/**
|
|
8
8
|
* 把 catch 到的 e 转成报告用字符串。优先带 stack(定位到 eval 脚本抛错的具体 file:line),
|
|
9
|
-
* 只在没有 stack 时才退化到 `name: message
|
|
10
|
-
* `e instanceof Error ? e.message : String(e)
|
|
9
|
+
* 只在没有 stack 时才退化到 `name: message`;两种形态都补上 `cause` 链。EvalResult.error 走
|
|
10
|
+
* 这个,别再手写 `e instanceof Error ? e.message : String(e)`——那样用户永远看不出错误发生在
|
|
11
|
+
* 哪一行、也看不到真实死因。`firstLine()` 的消费方不受影响:cause 恒在第一行之后。
|
|
11
12
|
*/
|
|
12
13
|
export declare function formatThrown(e: unknown): string;
|
|
13
14
|
/**
|
package/dist/util.js
CHANGED
|
@@ -17,15 +17,41 @@ export function getEnv(name) {
|
|
|
17
17
|
export function stripComments(code) {
|
|
18
18
|
return code.replace(/\/\*[\s\S]*?\*\//g, "").replace(/\/\/.*$/gm, "");
|
|
19
19
|
}
|
|
20
|
+
/** `cause` 链最多展开这么多层——足够穿透「包装了三四层」的常见形态,又不至于把整棵树倒出来。 */
|
|
21
|
+
const CAUSE_CHAIN_DEPTH = 5;
|
|
22
|
+
/**
|
|
23
|
+
* `cause` 链的多行后缀。`Error.stack` **不含** `cause`,而真实死因常常只在那里:`fetch failed`
|
|
24
|
+
* 的 stack 一个字都不说为什么失败,`error.cause.code` 才是 `ECONNRESET` / `ENOTFOUND` / 证书错误。
|
|
25
|
+
* 不展开这条链,用户拿到的就是一句无法行动的 `TypeError: fetch failed`。
|
|
26
|
+
* 逐层带上 `code`(有的话)——它比 message 更适合搜索和按值分支。
|
|
27
|
+
*/
|
|
28
|
+
function causeChainSuffix(error) {
|
|
29
|
+
const lines = [];
|
|
30
|
+
let current = error.cause;
|
|
31
|
+
for (let depth = 0; depth < CAUSE_CHAIN_DEPTH && current != null; depth++) {
|
|
32
|
+
if (current instanceof Error) {
|
|
33
|
+
const code = current.code;
|
|
34
|
+
const label = typeof code === "string" ? `${current.name} (${code})` : current.name;
|
|
35
|
+
lines.push(` caused by: ${label}: ${current.message}`);
|
|
36
|
+
current = current.cause;
|
|
37
|
+
}
|
|
38
|
+
else {
|
|
39
|
+
lines.push(` caused by: ${String(current)}`);
|
|
40
|
+
break;
|
|
41
|
+
}
|
|
42
|
+
}
|
|
43
|
+
return lines.length > 0 ? `\n${lines.join("\n")}` : "";
|
|
44
|
+
}
|
|
20
45
|
/**
|
|
21
46
|
* 把 catch 到的 e 转成报告用字符串。优先带 stack(定位到 eval 脚本抛错的具体 file:line),
|
|
22
|
-
* 只在没有 stack 时才退化到 `name: message
|
|
23
|
-
* `e instanceof Error ? e.message : String(e)
|
|
47
|
+
* 只在没有 stack 时才退化到 `name: message`;两种形态都补上 `cause` 链。EvalResult.error 走
|
|
48
|
+
* 这个,别再手写 `e instanceof Error ? e.message : String(e)`——那样用户永远看不出错误发生在
|
|
49
|
+
* 哪一行、也看不到真实死因。`firstLine()` 的消费方不受影响:cause 恒在第一行之后。
|
|
24
50
|
*/
|
|
25
51
|
export function formatThrown(e) {
|
|
26
|
-
if (e instanceof Error)
|
|
27
|
-
return e
|
|
28
|
-
return
|
|
52
|
+
if (!(e instanceof Error))
|
|
53
|
+
return String(e);
|
|
54
|
+
return (e.stack ?? `${e.name}: ${e.message}`) + causeChainSuffix(e);
|
|
29
55
|
}
|
|
30
56
|
/**
|
|
31
57
|
* 截到第一个换行为止。`formatThrown()` 优先带完整 `.stack`(含本地绝对文件路径的多行调用栈)
|
|
@@ -56,10 +56,45 @@ npx niceeval exp local fixtures/button --runs 5 --early-exit
|
|
|
56
56
|
|
|
57
57
|
[NiceEval](https://niceeval.com/) 可以根据输入、配置和相关文件 fingerprint 跳过已判定为 `passed` 或 `failed` 的结果——两者都是判定确定的终态。`errored`(超时、Sandbox 异常等框架/环境层面的不确定失败)永远重试。缓存适合加速迭代,但如果你在调试非确定性行为,应该明确关闭(`--force`)或清理相关缓存。
|
|
58
58
|
|
|
59
|
+
## 并行开多个终端
|
|
60
|
+
|
|
61
|
+
想同时跑几个 Experiment,直接开几个终端各跑各的;一场大实验嫌慢,就再开一个终端跑同一条命令,两边会自动分工:
|
|
62
|
+
|
|
63
|
+
```bash
|
|
64
|
+
# 各跑各的:两个实验并行
|
|
65
|
+
npx niceeval exp compare/codex # 终端 1
|
|
66
|
+
npx niceeval exp compare/claude # 终端 2
|
|
67
|
+
|
|
68
|
+
# 给同一场实验加速:同一条命令再跑一遍
|
|
69
|
+
npx niceeval exp compare/bub # 终端 1
|
|
70
|
+
npx niceeval exp compare/bub # 终端 2
|
|
71
|
+
```
|
|
72
|
+
|
|
73
|
+
每次运行写自己的结果快照目录,互不覆盖。就算两条命令选中了同一批评估用例,NiceEval 也不会把任何一条跑两遍:每条评估用例开跑前会在 `.niceeval/locks/` 下拿一把带心跳的锁,锁是跑到哪条拿哪条,不会一口气把整批占住。另一次运行撞到锁,就先去跑还没人认领的评估用例——所以同一条命令开两个终端就是加速:两边自动认领不同的评估用例,各按自己的 `--max-concurrency` 推进,总吞吐接近两边之和。撞锁的评估用例在 live 面板上单独计成 `elsewhere`(别人在运行),并显示一行 `waiting on another run` 说明在等谁;对方跑完、锁一放开,等待方接着往下走。每条评估用例拿到锁的那一刻,都会按缓存规则再看一眼它当前的结果:另一次运行已经跑完落盘的直接复用,还缺的 Attempt 才自己跑。所以不管两个终端的快慢怎么错开,两边最后都拿到完整结果,每条评估用例只花一份钱。
|
|
74
|
+
|
|
75
|
+
锁不会合并两边的全局并发名额:两个终端各有自己的 `--max-concurrency`,Sandbox provider 和模型接口承受的是两边之和,配额紧张时把两边都调低一点。例外是实验自己的 `maxConcurrency`:这个名额是同一个实验的所有运行共用的——声明了 `maxConcurrency: 1` 的串行实验,开几个终端也还是同一时刻只跑一个 Attempt,累积状态不会被并行打乱。
|
|
76
|
+
|
|
77
|
+
某次运行被 `kill -9` 或断电杀掉时,它留下的锁不会卡住后来者:心跳停止 30 秒后,下一次运行会自动接管这把锁照常跑。`.niceeval/locks/` 不需要手工清理。
|
|
78
|
+
|
|
59
79
|
## Turn 瞬时错误重试
|
|
60
80
|
|
|
61
81
|
限流、连接建立失败这类瞬时错误,NiceEval 会在同一个 Turn 里自动做有限次数的指数退避重试,不需要你重跑整个实验;等待重试的 Attempt 会把全局并发名额让给别的 Attempt,但实验自己的 `maxConcurrency` 名额不让——声明了 `maxConcurrency: 1` 的串行实验在退避期间也不会有第二个 Attempt 提前进场。只有能确认 Agent 还没开始处理这次输入的错误才会重试——请求已经开始、中途断流的情况不重试,直接记为 `errored`,避免 Agent 把已经做过的操作再做一遍。重试用尽仍失败时,该 Attempt 记为 `errored`,下次运行照常重试(见上面的缓存规则)。
|
|
62
82
|
|
|
83
|
+
## 声明失败的波及范围
|
|
84
|
+
|
|
85
|
+
有些失败你第一次看到就知道不止这一次:整个 Experiment 共享的服务死了,每条评估用例都会同样死;某条评估用例的 fixture 文件缺失,`runs: 5` 五次都会同样死。不声明的话,NiceEval 只能一条条撞——每个 Attempt 各自创建 Sandbox、各自失败,同一个死因烧几十遍。把你知道的告诉它,第一次撞上就停:
|
|
86
|
+
|
|
87
|
+
```ts
|
|
88
|
+
import { ExperimentFatalError } from "niceeval";
|
|
89
|
+
|
|
90
|
+
// 在 setup 或校验代码里抛出:这个 Experiment 还没跑的 Attempt 全部停止派发
|
|
91
|
+
throw new ExperimentFatalError("记忆服务探活失败——修好隧道、更新 .env 后重跑");
|
|
92
|
+
```
|
|
93
|
+
|
|
94
|
+
只影响一条评估用例的死因(fixture 缺失)抛 `EvalFatalError`,只停这一条的剩余 Attempt,别的评估用例照常跑。错误 message 会原样出现在终端通知和结果快照里,写成「现象 + 下一步」,就是留给修的人的字条。服务在 run 中途死掉、失败以连接错误的形态冒出来时,给 Experiment 配 `classifyFailure`,只认自己共享服务的 host 来识别它。
|
|
95
|
+
|
|
96
|
+
停下来的 Attempt 计入 `unstarted`,这次运行的结论是 `incomplete`,不会伪装成全绿;同批其它 Experiment 不受影响。修好环境后重跑同一条命令就是续跑:已通过的照常复用,只补跑死掉与没跑的部分,没有任何标记要解除。只声明你能确定的死因——「看起来像基建问题」不算确定;拿不准就让它落成单条 Attempt 的失败,多烧的是一个 Sandbox 的钱,错停的是整批数据。
|
|
97
|
+
|
|
63
98
|
## 超时和预算
|
|
64
99
|
|
|
65
100
|
```bash
|
|
@@ -93,7 +93,7 @@ npx niceeval exp models weather
|
|
|
93
93
|
| `--out` | string | `view` 命令专用:把结果查看器静态导出到指定目录。 |
|
|
94
94
|
| `--port` | number | `view` 命令专用:指定本地服务器监听端口。 |
|
|
95
95
|
| `--source` | boolean | `show` 命令专用:该 attempt 运行时保存的 Eval 源码,gate/soft 断言标回源码行(证据切面)。 |
|
|
96
|
-
| `--execution` | boolean | `show` 命令专用:该 attempt 的标准执行事件流(消息、thinking、Skill load、工具调用/结果);有 OTel 时同一节点补时间(证据切面)
|
|
96
|
+
| `--execution` | boolean | `show` 命令专用:该 attempt 的标准执行事件流(消息、thinking、Skill load、工具调用/结果);有 OTel 时同一节点补时间(证据切面)。每个内容段最多预览前 3 行,截断尾巴自带 `--expand` 展开句柄。 |
|
|
97
97
|
| `--timing` | boolean | `show` 命令专用:整个 Attempt 的统一时间树;裸 `--timing` 给有界诊断投影,`--timing=full` 逐节点展开全部 runner/已关联 OTel 节点。 |
|
|
98
98
|
| `--grep` | string | `show` 命令专用:只与 `--execution` 组合;JS 正则,只输出命中的执行卡片(角色文本、工具名、input、result,失败命令再加 display/stdout/stderr),末尾报跨 attempt 汇总 `N matches in M attempts`。与 `--expand` 互斥。 |
|
|
99
99
|
| `--expand` | string | `show` 命令专用:只与 `--execution` 组合,要求范围恰好命中一个 attempt;展开一张卡片的完整落盘内容(不截断)。句柄语法 `t<轮次>.c<卡片>`(agent 事件)或 `cmd<n>`(失败 Sandbox 命令),来自截断卡片自带的提示。与 `--grep` 互斥。 |
|
|
@@ -170,6 +170,7 @@ npx niceeval show weather/brooklyn
|
|
|
170
170
|
npx niceeval show @1k2m9qrs
|
|
171
171
|
npx niceeval show @1k2m9qrs --source
|
|
172
172
|
npx niceeval show @1k2m9qrs --execution
|
|
173
|
+
npx niceeval show @1k2m9qrs --execution --expand t2.c3
|
|
173
174
|
npx niceeval show fixtures/button --diff
|
|
174
175
|
npx niceeval show weather/brooklyn --history
|
|
175
176
|
```
|
|
@@ -178,6 +179,14 @@ npx niceeval show weather/brooklyn --history
|
|
|
178
179
|
|
|
179
180
|
`@<locator>` 不带证据 flag 时给出该 attempt 的紧凑全景(断言摘要、执行摘要、可选 OTel 时间、diff 摘要);`--source`、`--execution`、`--timing`、`--diff` 是证据切面,各自接受任意范围,不限于单个 `@<locator>`——范围含多个 attempt 时按 experiment id、eval id、attempt 序逐 attempt 分节。`--results <目录>` 指定结果根,`--history` 查看跨 run 趋势。`--exp` 可重复:出现一次收窄范围,两次以上进入对照语义(每个 `--exp` 必须恰好解析到一个 experiment,顺序即对照条件顺序)。完整的阅读顺序、输出示例和 artifact 说明见[查看结果](/zh/tutorials/viewing-results)。
|
|
180
181
|
|
|
182
|
+
`--execution` 的卡片正文是有界预览:每个内容段(消息正文;工具卡的 input 与 result 各算一段)最多显示前 3 行,每段另有 1 KiB(UTF-8 字节)兜底防单行超长。有折叠的卡片在卡尾报被折的行数与字符数,并自带该卡片的展开句柄,整行就是可复制的完整命令:
|
|
183
|
+
|
|
184
|
+
```text
|
|
185
|
+
(+3055 lines · 263161 chars · niceeval show @1k2m9qrs --execution --expand t2.c3)
|
|
186
|
+
```
|
|
187
|
+
|
|
188
|
+
`--expand <handle>` 只与 `--execution` 组合、要求范围恰好命中一个 attempt,单独输出该卡片的完整落盘内容。句柄 `t<轮次>.c<轮内卡序>`(agent 事件卡)或 `cmd<n>`(失败 Sandbox 命令卡)从截断提示里抄,不自己数。落盘时已过 256 KiB 单值上限的内容展开后如实标注 `truncated` 与原始字节数。`--grep` 命中的卡片同样受预览预算,尾巴照带句柄;`--json` 恒输出完整值、从不截断,与 `--expand` 互斥。
|
|
189
|
+
|
|
181
190
|
## `--early-exit` 与 `--strict`
|
|
182
191
|
|
|
183
192
|
`--early-exit` 默认关闭:`--runs` > 1 时默认把每次 attempt 都跑完,给出真实通过率——这是 NiceEval 衡量 agent 稳不稳的核心指标,默认不该被无声截断。只想知道"这题能不能过"、不在乎完整分布时,显式加 `--early-exit`:某个评估用例的一次 attempt 通过后,自动停止该评估用例剩余的 attempts(省钱)。实验文件里写了 `earlyExit: true` 时,用 `--no-early-exit` 强制关掉它。
|
|
@@ -31,7 +31,7 @@ interface Turn {
|
|
|
31
31
|
inputTokens?: number;
|
|
32
32
|
```
|
|
33
33
|
|
|
34
|
-
|
|
34
|
+
未命中缓存、按全价计费的输入 token;与两个 cache 桶互斥。
|
|
35
35
|
|
|
36
36
|
#### `outputTokens`
|
|
37
37
|
|
|
@@ -47,7 +47,7 @@ outputTokens?: number;
|
|
|
47
47
|
cacheReadTokens?: number;
|
|
48
48
|
```
|
|
49
49
|
|
|
50
|
-
|
|
50
|
+
从提示缓存命中的输入 token;独立计价桶,不包含在 inputTokens 里(省略表示该 agent 不上报此项)。
|
|
51
51
|
|
|
52
52
|
#### `cacheCreationTokens`
|
|
53
53
|
|
|
@@ -55,7 +55,7 @@ cacheReadTokens?: number;
|
|
|
55
55
|
cacheCreationTokens?: number;
|
|
56
56
|
```
|
|
57
57
|
|
|
58
|
-
|
|
58
|
+
写入提示缓存的输入 token;独立计价桶,不包含在 inputTokens 里(省略表示该 agent 不上报此项)。
|
|
59
59
|
|
|
60
60
|
#### `reasoningTokens`
|
|
61
61
|
|
|
@@ -63,7 +63,7 @@ cacheCreationTokens?: number;
|
|
|
63
63
|
reasoningTokens?: number;
|
|
64
64
|
```
|
|
65
65
|
|
|
66
|
-
推理(thinking)token
|
|
66
|
+
推理(thinking)token 数,outputTokens 的已含明细,单列展示用;只在协议真实提供时存在。
|
|
67
67
|
|
|
68
68
|
#### `requests`
|
|
69
69
|
|