niceeval 0.10.3-canary.7 → 0.11.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (121) hide show
  1. package/dist/agents/types.d.ts +5 -4
  2. package/dist/context/turn-errors.d.ts +27 -23
  3. package/dist/i18n/en.d.ts +14 -0
  4. package/dist/i18n/en.js +15 -1
  5. package/dist/i18n/zh-CN.d.ts +15 -1
  6. package/dist/i18n/zh-CN.js +15 -1
  7. package/dist/o11y/derive.js +28 -24
  8. package/dist/o11y/types.d.ts +8 -5
  9. package/dist/report/components/attempt-detail/UsageTable.js +4 -7
  10. package/dist/report/components/attempt-detail/compute.d.ts +2 -2
  11. package/dist/report/components/attempt-detail/compute.js +2 -6
  12. package/dist/report/components/attempt-detail/faces.js +7 -7
  13. package/dist/report/components/attempt-detail/index.js +0 -3
  14. package/dist/report/components/entity-lists/EvalList.js +0 -0
  15. package/dist/report/components/metric-views/compute.js +1 -1
  16. package/dist/report/model/types.d.ts +3 -4
  17. package/dist/results/locator.js +0 -0
  18. package/dist/results/select.d.ts +6 -0
  19. package/dist/results/select.js +8 -0
  20. package/dist/runner/feedback/sink.d.ts +26 -1
  21. package/dist/runner/fingerprint.d.ts +23 -0
  22. package/dist/runner/types.d.ts +105 -7
  23. package/dist/sandbox/errors.d.ts +29 -0
  24. package/dist/sandbox/resolve.d.ts +9 -0
  25. package/dist/shared/failure-class.d.ts +91 -0
  26. package/dist/types.d.ts +1 -0
  27. package/dist/util.d.ts +3 -2
  28. package/dist/util.js +31 -5
  29. package/docs-site/zh/explanation/runner.mdx +35 -0
  30. package/docs-site/zh/reference/cli.mdx +3 -3
  31. package/docs-site/zh/reference/events.mdx +4 -4
  32. package/docs-site/zh/troubleshooting/debugging.mdx +2 -2
  33. package/docs-site/zh/tutorials/viewing-results.mdx +4 -5
  34. package/package.json +4 -12
  35. package/src/agents/ai-sdk.test.ts +26 -0
  36. package/src/agents/ai-sdk.ts +7 -4
  37. package/src/agents/index.ts +5 -3
  38. package/src/agents/langgraph.test.ts +30 -0
  39. package/src/agents/langgraph.ts +5 -2
  40. package/src/agents/openai-compat.test.ts +35 -0
  41. package/src/agents/openai-compat.ts +16 -4
  42. package/src/agents/sdk-streams.test.ts +66 -0
  43. package/src/agents/sdk-streams.ts +3 -1
  44. package/src/agents/types.ts +5 -4
  45. package/src/cli.ts +55 -14
  46. package/src/context/context.test.ts +34 -0
  47. package/src/context/context.ts +14 -5
  48. package/src/context/send-retry.test.ts +86 -0
  49. package/src/context/send-retry.ts +37 -12
  50. package/src/context/session.ts +24 -0
  51. package/src/context/turn-errors.test.ts +124 -17
  52. package/src/context/turn-errors.ts +60 -50
  53. package/src/define.ts +5 -0
  54. package/src/i18n/en.ts +20 -1
  55. package/src/i18n/zh-CN.ts +20 -1
  56. package/src/index.ts +8 -0
  57. package/src/o11y/cost.ts +4 -2
  58. package/src/o11y/derive.test.ts +40 -0
  59. package/src/o11y/derive.ts +28 -22
  60. package/src/o11y/otlp/sandbox-receiver.test.ts +201 -0
  61. package/src/o11y/otlp/sandbox-receiver.ts +73 -27
  62. package/src/o11y/parsers/bub.test.ts +30 -0
  63. package/src/o11y/parsers/bub.ts +5 -2
  64. package/src/o11y/parsers/codex.test.ts +19 -0
  65. package/src/o11y/parsers/codex.ts +5 -2
  66. package/src/o11y/types.ts +8 -5
  67. package/src/report/components/attempt-detail/UsageTable.tsx +4 -6
  68. package/src/report/components/attempt-detail/attempt-components.test.tsx +5 -8
  69. package/src/report/components/attempt-detail/compute.ts +2 -7
  70. package/src/report/components/attempt-detail/faces.ts +7 -7
  71. package/src/report/components/attempt-detail/index.tsx +0 -3
  72. package/src/report/components/entity-lists/EvalList.tsx +0 -0
  73. package/src/report/components/metric-views/compute.ts +1 -1
  74. package/src/report/model/types.ts +3 -4
  75. package/src/results/format.ts +9 -2
  76. package/src/results/index.ts +2 -0
  77. package/src/results/locator.ts +0 -0
  78. package/src/results/open.ts +132 -21
  79. package/src/results/select.ts +12 -0
  80. package/src/results/skipped-notice.ts +0 -0
  81. package/src/runner/attempt.test.ts +116 -0
  82. package/src/runner/attempt.ts +89 -8
  83. package/src/runner/discover.ts +21 -3
  84. package/src/runner/feedback/coordinator.ts +24 -2
  85. package/src/runner/feedback/eval-conclusions.ts +6 -3
  86. package/src/runner/feedback/human.test.ts +300 -6
  87. package/src/runner/feedback/human.ts +132 -30
  88. package/src/runner/feedback/json.test.ts +127 -2
  89. package/src/runner/feedback/json.ts +51 -3
  90. package/src/runner/feedback/reducer.test.ts +328 -29
  91. package/src/runner/feedback/reducer.ts +84 -9
  92. package/src/runner/feedback/sink.ts +39 -1
  93. package/src/runner/fingerprint.ts +49 -19
  94. package/src/runner/gate-lease.test.ts +510 -0
  95. package/src/runner/gate-lease.ts +350 -0
  96. package/src/runner/lock.test.ts +454 -0
  97. package/src/runner/lock.ts +288 -0
  98. package/src/runner/report.test.ts +1 -0
  99. package/src/runner/run.test.ts +2044 -9
  100. package/src/runner/run.ts +825 -61
  101. package/src/runner/teardown-registry.ts +20 -78
  102. package/src/runner/types.ts +103 -7
  103. package/src/sandbox/errors.test.ts +72 -0
  104. package/src/sandbox/errors.ts +83 -0
  105. package/src/sandbox/keep-registry.ts +22 -46
  106. package/src/sandbox/resolve.test.ts +100 -0
  107. package/src/sandbox/resolve.ts +84 -27
  108. package/src/shared/entry-file-store.test.ts +149 -0
  109. package/src/shared/entry-file-store.ts +117 -0
  110. package/src/shared/failure-class.test.ts +137 -0
  111. package/src/shared/failure-class.ts +175 -0
  112. package/src/show/index.ts +5 -6
  113. package/src/show/render.test.ts +116 -15
  114. package/src/show/render.ts +124 -35
  115. package/src/types.ts +9 -0
  116. package/src/util.ts +31 -4
  117. package/src/view/app/App.tsx +5 -1
  118. package/src/view/client-dist/app.js +1 -1
  119. package/src/view/data.ts +5 -9
  120. package/src/view/shared/types.ts +7 -1
  121. package/src/view/view-report.test.ts +54 -0
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "niceeval",
3
- "version": "0.10.3-canary.7",
3
+ "version": "0.11.0",
4
4
  "description": "Agent-native eval tool — eval agents, services, functions, and coding-agent fixtures",
5
5
  "type": "module",
6
6
  "license": "MIT",
@@ -93,6 +93,8 @@
93
93
  "autoevals": "0.0.132",
94
94
  "effect": "^3.21.4",
95
95
  "mermaid": "^11.16.0",
96
+ "react": "^19.0.0",
97
+ "react-dom": "^19.0.0",
96
98
  "tar-stream": "^3.1.7",
97
99
  "tsx": "^4.19.2"
98
100
  },
@@ -121,8 +123,6 @@
121
123
  "mixpanel-browser": "^2.80.0",
122
124
  "next": "16.2.10",
123
125
  "prism-react-renderer": "^2.4.1",
124
- "react": "^19.2.7",
125
- "react-dom": "^19.2.7",
126
126
  "react-grab": "^0.1.48",
127
127
  "shiki": "^4.3.0",
128
128
  "tailwind-merge": "^3.6.0",
@@ -139,9 +139,7 @@
139
139
  "ai": ">=5.0.0",
140
140
  "braintrust": ">=0.0.150",
141
141
  "dockerode": ">=4.0.0",
142
- "e2b": ">=2.0.0",
143
- "react": ">=18",
144
- "react-dom": ">=18"
142
+ "e2b": ">=2.0.0"
145
143
  },
146
144
  "peerDependenciesMeta": {
147
145
  "@ai-sdk/otel": {
@@ -167,12 +165,6 @@
167
165
  },
168
166
  "e2b": {
169
167
  "optional": true
170
- },
171
- "react": {
172
- "optional": true
173
- },
174
- "react-dom": {
175
- "optional": true
176
168
  }
177
169
  },
178
170
  "scripts": {
@@ -0,0 +1,26 @@
1
+ // cases: docs/engineering/testing/unit/results.md
2
+ // 「Usage、facts 与失败命令证据落盘」桶恒互斥归一:AI SDK 的 inputTokens 是含缓存明细的
3
+ // 输入总量,落桶前扣掉在场的 cacheRead / cacheWrite 明细。
4
+ // bug: memory/estimatecost-openai-inclusive-cache-double-billed.md
5
+
6
+ import { describe, expect, it } from "vitest";
7
+
8
+ import { fromAiSdk } from "./ai-sdk.ts";
9
+
10
+ describe("fromAiSdk usage 归一(含明细口径)", () => {
11
+ it("v5 形状:cachedInputTokens 从 inputTokens 里扣出", () => {
12
+ const turn = fromAiSdk({
13
+ text: "ok",
14
+ usage: { inputTokens: 1000, outputTokens: 20, cachedInputTokens: 900 },
15
+ });
16
+ expect(turn.usage).toMatchObject({ inputTokens: 100, cacheReadTokens: 900, outputTokens: 20 });
17
+ });
18
+
19
+ it("v7 形状:inputTokenDetails 的 cacheRead 与 cacheWrite 都从总量里扣出", () => {
20
+ const turn = fromAiSdk({
21
+ text: "ok",
22
+ usage: { inputTokens: 1000, outputTokens: 20, inputTokenDetails: { cacheReadTokens: 800, cacheWriteTokens: 100 } },
23
+ });
24
+ expect(turn.usage).toMatchObject({ inputTokens: 100, cacheReadTokens: 800, cacheCreationTokens: 100 });
25
+ });
26
+ });
@@ -327,13 +327,16 @@ function unwrapToolOutput(output: unknown): { output?: JsonValue; status: "compl
327
327
  function readUsage(result: AiSdkResultLike, stepCount: number): Usage | undefined {
328
328
  const u = result.totalUsage ?? result.usage;
329
329
  if (!u) return undefined;
330
- const inputTokens = num(u.inputTokens) ?? num(u.promptTokens) ?? 0;
330
+ const rawInput = num(u.inputTokens) ?? num(u.promptTokens) ?? 0;
331
331
  const outputTokens = num(u.outputTokens) ?? num(u.completionTokens) ?? 0;
332
- if (inputTokens === 0 && outputTokens === 0) return undefined;
332
+ if (rawInput === 0 && outputTokens === 0) return undefined;
333
+ const cacheRead = num(u.cachedInputTokens) ?? num(u.inputTokenDetails?.cacheReadTokens) ?? 0;
334
+ const cacheCreation = num(u.inputTokenDetails?.cacheWriteTokens) ?? 0;
335
+ // AI SDK 的 inputTokens 是含缓存明细的输入总量,落互斥桶前扣掉在场的明细
336
+ // (docs/feature/adapters/sdk/ai-sdk/cost.md)
337
+ const inputTokens = Math.max(0, rawInput - cacheRead - cacheCreation);
333
338
  const usage: Usage = { inputTokens, outputTokens, requests: Math.max(stepCount, 1) };
334
- const cacheRead = num(u.cachedInputTokens) ?? num(u.inputTokenDetails?.cacheReadTokens);
335
339
  if (cacheRead) usage.cacheReadTokens = cacheRead;
336
- const cacheCreation = num(u.inputTokenDetails?.cacheWriteTokens);
337
340
  if (cacheCreation) usage.cacheCreationTokens = cacheCreation;
338
341
  const reasoning = num(u.reasoningTokens);
339
342
  if (reasoning) usage.reasoningTokens = reasoning;
@@ -9,11 +9,13 @@ export type { Shared } from "./shared.ts";
9
9
  export { completeCoverage } from "../scoring/coverage.ts";
10
10
  export type { CoverageStatus, CoverageDeclaration, EvidenceCoverage } from "../types.ts";
11
11
 
12
- // turn 级瞬时错误分类:`Agent.classifyTurnError` 认的输入/输出形状 + 摘要取值器
13
- // (与 turn-failed 报错文案同源)。分类判据、分类链与重试执行体见
12
+ // 执行失败分类:`Agent.classifyTurnError` 认的输入/输出形状 + 摘要取值器(与 turn-failed
13
+ // 报错文案同源)。两轴词表(FailureClass / FailureScope)与包根导出的是同一个形状——adapter
14
+ // 作者与 eval 作者各自的入口拿到同一份类型。判据、分类链与重试执行体见
14
15
  // docs/feature/error-classification/architecture.md。
15
16
  export { turnErrorText } from "../context/turn-errors.ts";
16
- export type { TurnErrorClass, TurnErrorClassifier, TurnFailure } from "../context/turn-errors.ts";
17
+ export type { TurnErrorClassifier, TurnFailure } from "../context/turn-errors.ts";
18
+ export type { FailureClass, FailureScope } from "../shared/failure-class.ts";
17
19
 
18
20
  // span → canonical GenAI 归一(只服务瀑布图,不喂断言)。私有埋点写自己的 spanMapper 时用:
19
21
  // tagSpan 把判定写回 span(原属性只增不改),heuristicTag 是通用兜底判定;mapCodexSpans 是
@@ -0,0 +1,30 @@
1
+ // cases: docs/engineering/testing/unit/results.md
2
+ // 「Usage、facts 与失败命令证据落盘」桶恒互斥归一:LangChain usage_metadata 的 input_tokens
3
+ // 是含缓存读写的输入总量,落桶前扣掉 input_token_details 的 cache_read / cache_creation。
4
+ // bug: memory/estimatecost-openai-inclusive-cache-double-billed.md
5
+
6
+ import { describe, expect, it } from "vitest";
7
+
8
+ import { fromLangGraphEvents } from "./langgraph.ts";
9
+
10
+ describe("fromLangGraphEvents usage 归一(含明细口径)", () => {
11
+ it("cache_read 与 cache_creation 都从 input_tokens 里扣出", () => {
12
+ const stream = fromLangGraphEvents();
13
+ stream.add({
14
+ channel: "messages",
15
+ event: "finish",
16
+ data: {
17
+ message: {
18
+ role: "assistant",
19
+ content: "ok",
20
+ usage_metadata: {
21
+ input_tokens: 1000,
22
+ output_tokens: 20,
23
+ input_token_details: { cache_read: 800, cache_creation: 100 },
24
+ },
25
+ },
26
+ },
27
+ });
28
+ expect(stream.usage).toMatchObject({ inputTokens: 100, cacheReadTokens: 800, cacheCreationTokens: 100, outputTokens: 20 });
29
+ });
30
+ });
@@ -152,12 +152,15 @@ export function fromLangGraphEvents(): LangGraphStream {
152
152
  }
153
153
  return 0;
154
154
  };
155
- const input = num("input_tokens", "inputTokens");
155
+ const rawInput = num("input_tokens", "inputTokens");
156
156
  const output = num("output_tokens", "outputTokens");
157
- if (input === 0 && output === 0) return;
157
+ if (rawInput === 0 && output === 0) return;
158
158
  const details = isRecord(raw.input_token_details) ? raw.input_token_details : undefined;
159
159
  const cacheRead = typeof details?.cache_read === "number" ? details.cache_read : 0;
160
160
  const cacheCreation = typeof details?.cache_creation === "number" ? details.cache_creation : 0;
161
+ // LangChain usage_metadata 的 input_tokens 是含缓存读写的输入总量,落互斥桶前扣掉明细
162
+ // (docs/feature/adapters/sdk/langgraph/cost.md)
163
+ const input = Math.max(0, rawInput - cacheRead - cacheCreation);
161
164
  // LangChain UsageMetadata.output_token_details.reasoning:推理模型经 LangGraph 透传时带回。
162
165
  const outputDetails = isRecord(raw.output_token_details) ? raw.output_token_details : undefined;
163
166
  const reasoning = typeof outputDetails?.reasoning === "number" ? outputDetails.reasoning : 0;
@@ -0,0 +1,35 @@
1
+ // cases: docs/engineering/testing/unit/results.md
2
+ // 「Usage、facts 与失败命令证据落盘」桶恒互斥归一:Chat Completions / Responses 两种形状的
3
+ // cached_tokens 都是输入总量的子集,落 inputTokens 前扣掉;缺 cached 明细时总量原样保留。
4
+ // bug: memory/estimatecost-openai-inclusive-cache-double-billed.md
5
+
6
+ import { describe, expect, it } from "vitest";
7
+
8
+ import { fromChatCompletion, fromResponses } from "./openai-compat.ts";
9
+
10
+ describe("openai-compat usage 归一(OpenAI 口径)", () => {
11
+ it("Chat Completions:prompt_tokens 扣掉 prompt_tokens_details.cached_tokens", () => {
12
+ const turn = fromChatCompletion({
13
+ choices: [{ message: { role: "assistant", content: "ok" } }],
14
+ usage: { prompt_tokens: 1000, completion_tokens: 20, prompt_tokens_details: { cached_tokens: 900 } },
15
+ });
16
+ expect(turn.usage).toMatchObject({ inputTokens: 100, cacheReadTokens: 900, outputTokens: 20 });
17
+ });
18
+
19
+ it("Responses:input_tokens 扣掉 input_tokens_details.cached_tokens", () => {
20
+ const turn = fromResponses({
21
+ output: [{ type: "message", role: "assistant", content: [{ type: "output_text", text: "ok" }] }],
22
+ usage: { input_tokens: 500, output_tokens: 10, input_tokens_details: { cached_tokens: 200 } },
23
+ });
24
+ expect(turn.usage).toMatchObject({ inputTokens: 300, cacheReadTokens: 200, outputTokens: 10 });
25
+ });
26
+
27
+ it("缺 cached 明细时输入总量原样保留,不虚构扣减,cache 桶省略", () => {
28
+ const turn = fromChatCompletion({
29
+ choices: [{ message: { role: "assistant", content: "ok" } }],
30
+ usage: { prompt_tokens: 1000, completion_tokens: 20 },
31
+ });
32
+ expect(turn.usage?.inputTokens).toBe(1000);
33
+ expect(turn.usage?.cacheReadTokens).toBeUndefined();
34
+ });
35
+ });
@@ -50,8 +50,14 @@ export interface ChatCompletionLike {
50
50
 
51
51
  function chatCompletionUsage(usage: ChatCompletionUsageLike | undefined): Usage | undefined {
52
52
  if (!usage) return undefined;
53
- const u: Usage = { inputTokens: usage.prompt_tokens ?? 0, outputTokens: usage.completion_tokens ?? 0 };
54
- if (usage.prompt_tokens_details?.cached_tokens) u.cacheReadTokens = usage.prompt_tokens_details.cached_tokens;
53
+ // prompt_tokens 含缓存命中,cached_tokens 是其子集;落互斥桶前扣掉
54
+ // (docs/feature/adapters/sdk/openai-compat/cost.md)
55
+ const cached = usage.prompt_tokens_details?.cached_tokens ?? 0;
56
+ const u: Usage = {
57
+ inputTokens: Math.max(0, (usage.prompt_tokens ?? 0) - cached),
58
+ outputTokens: usage.completion_tokens ?? 0,
59
+ };
60
+ if (cached) u.cacheReadTokens = cached;
55
61
  if (usage.completion_tokens_details?.reasoning_tokens) u.reasoningTokens = usage.completion_tokens_details.reasoning_tokens;
56
62
  return u;
57
63
  }
@@ -125,8 +131,14 @@ export interface ResponseLike {
125
131
 
126
132
  function responsesUsage(usage: ResponseUsageLike | undefined): Usage | undefined {
127
133
  if (!usage) return undefined;
128
- const u: Usage = { inputTokens: usage.input_tokens ?? 0, outputTokens: usage.output_tokens ?? 0 };
129
- if (usage.input_tokens_details?.cached_tokens) u.cacheReadTokens = usage.input_tokens_details.cached_tokens;
134
+ // input_tokens 含缓存命中,cached_tokens 是其子集;落互斥桶前扣掉
135
+ // (docs/feature/adapters/sdk/openai-compat/cost.md)
136
+ const cached = usage.input_tokens_details?.cached_tokens ?? 0;
137
+ const u: Usage = {
138
+ inputTokens: Math.max(0, (usage.input_tokens ?? 0) - cached),
139
+ outputTokens: usage.output_tokens ?? 0,
140
+ };
141
+ if (cached) u.cacheReadTokens = cached;
130
142
  if (usage.output_tokens_details?.reasoning_tokens) u.reasoningTokens = usage.output_tokens_details.reasoning_tokens;
131
143
  return u;
132
144
  }
@@ -0,0 +1,66 @@
1
+ // cases: docs/engineering/testing/unit/results.md
2
+ // 「Usage、facts 与失败命令证据落盘」桶恒互斥归一:codex(OpenAI 口径,cached ⊂ input)扣减、
3
+ // Anthropic / pi(互斥口径)如实转发。fixture 数值刻意让「扣与不扣」结果可区分。
4
+ // bug: memory/estimatecost-openai-inclusive-cache-double-billed.md
5
+
6
+ import { describe, expect, it } from "vitest";
7
+
8
+ import { fromClaudeSdkMessages, fromCodexThreadEvents, fromPiAgentEvents } from "./sdk-streams.ts";
9
+
10
+ describe("fromCodexThreadEvents usage 归一(OpenAI 口径)", () => {
11
+ it("cached_input_tokens 是 input_tokens 子集:落 inputTokens 前扣掉,cache 单独成桶", () => {
12
+ const stream = fromCodexThreadEvents();
13
+ stream.add({ type: "turn.completed", usage: { input_tokens: 1000, cached_input_tokens: 900, output_tokens: 50 } });
14
+ expect(stream.usage).toMatchObject({ inputTokens: 100, cacheReadTokens: 900, outputTokens: 50, requests: 1 });
15
+ });
16
+
17
+ it("逐轮累加在扣减之后进行,总量仍互斥", () => {
18
+ const stream = fromCodexThreadEvents();
19
+ stream.add({ type: "turn.completed", usage: { input_tokens: 1000, cached_input_tokens: 900, output_tokens: 50 } });
20
+ stream.add({ type: "turn.completed", usage: { input_tokens: 2000, cached_input_tokens: 1700, output_tokens: 30 } });
21
+ expect(stream.usage).toMatchObject({ inputTokens: 400, cacheReadTokens: 2600, outputTokens: 80, requests: 2 });
22
+ });
23
+
24
+ it("协议报出 cached > input 的病态数据时扣减夹底到 0,不产生负 token", () => {
25
+ const stream = fromCodexThreadEvents();
26
+ stream.add({ type: "turn.completed", usage: { input_tokens: 100, cached_input_tokens: 200, output_tokens: 1 } });
27
+ expect(stream.usage?.inputTokens).toBe(0);
28
+ expect(stream.usage?.cacheReadTokens).toBe(200);
29
+ });
30
+ });
31
+
32
+ describe("fromClaudeSdkMessages usage 转发(Anthropic 互斥口径)", () => {
33
+ it("input_tokens 原生不含 cache read:如实转发,不做扣减", () => {
34
+ const stream = fromClaudeSdkMessages();
35
+ stream.add({
36
+ type: "result",
37
+ usage: { input_tokens: 100, output_tokens: 5, cache_read_input_tokens: 900, cache_creation_input_tokens: 50 },
38
+ });
39
+ expect(stream.usage).toMatchObject({
40
+ inputTokens: 100,
41
+ outputTokens: 5,
42
+ cacheReadTokens: 900,
43
+ cacheCreationTokens: 50,
44
+ });
45
+ });
46
+ });
47
+
48
+ describe("fromPiAgentEvents usage 转发(pi 互斥口径)", () => {
49
+ it("input 原生不含 cacheRead/cacheWrite:如实转发,cost.total 累进实测 costUSD", () => {
50
+ const stream = fromPiAgentEvents();
51
+ stream.add({
52
+ type: "message_end",
53
+ message: {
54
+ role: "assistant",
55
+ content: [],
56
+ usage: { input: 100, output: 5, cacheRead: 900, cacheWrite: 50, cost: { total: 0.42 } },
57
+ },
58
+ });
59
+ expect(stream.usage).toMatchObject({
60
+ inputTokens: 100,
61
+ cacheReadTokens: 900,
62
+ cacheCreationTokens: 50,
63
+ costUSD: 0.42,
64
+ });
65
+ });
66
+ });
@@ -484,7 +484,9 @@ export function fromCodexThreadEvents(): CodexThreadStream {
484
484
  if (isRecord(u)) {
485
485
  const num = (v: unknown): number => (typeof v === "number" ? v : 0);
486
486
  usage = {
487
- inputTokens: (usage?.inputTokens ?? 0) + num(u.input_tokens),
487
+ // codex-rs TokenUsage 的 cached_input_tokens 是 input_tokens 的子集,
488
+ // 落互斥桶前扣掉(docs/feature/adapters/sdk/codex-sdk/cost.md)
489
+ inputTokens: (usage?.inputTokens ?? 0) + Math.max(0, num(u.input_tokens) - num(u.cached_input_tokens)),
488
490
  outputTokens: (usage?.outputTokens ?? 0) + num(u.output_tokens),
489
491
  cacheReadTokens: (usage?.cacheReadTokens ?? 0) + num(u.cached_input_tokens),
490
492
  // codex-rs TokenUsage 结构体同一份字段(与 o11y/parsers/codex.ts 的
@@ -369,10 +369,11 @@ export interface Agent {
369
369
  spanMapper?: SpanMapper;
370
370
  send(input: TurnInput, ctx: AgentContext): Promise<Turn>;
371
371
  /**
372
- * 可选 turn 失败分类器:按重试安全性归类一次 send 失败(抛出或返回 `status: "failed"` 的
373
- * Turn),返回 `undefined` 回落保守兜底。分类器只声明决策与诊断词,不影响重试策略(次数、
374
- * 退避对所有 agent 一致);抛错按不可重试处理并被吞掉。形状与分类链、执行体时序见
375
- * docs/feature/error-classification/architecture.md。
372
+ * 可选 turn 失败分类器:归类一次 send 失败(抛出或返回 `status: "failed"` 的 Turn),
373
+ * 返回 `undefined` 表示不认识、回落保守兜底。链上排在实验的 `classifyFailure` 之后,
374
+ * 实验作者认领过的失败问不到这里。分类器只声明决策轴与诊断词,不影响重试策略(次数、
375
+ * 退避对所有 agent 一致);抛错按 `undefined` 回落并被吞掉,不掩盖原始失败。形状与分类链、
376
+ * 执行体时序见 docs/feature/error-classification/architecture.md。
376
377
  */
377
378
  classifyTurnError?: TurnErrorClassifier;
378
379
  teardown?: AgentTeardown;
package/src/cli.ts CHANGED
@@ -19,6 +19,8 @@ import { fingerprintEvalsFilter, resolveExperimentEvals, selectedEvalsForRun, sp
19
19
  import { failureDetailFromResult } from "./runner/feedback/failure.ts";
20
20
  import { stopAllSandboxes, liveSandboxCount } from "./sandbox/registry.ts";
21
21
  import { drainExperimentTeardowns } from "./runner/experiment-cleanup-registry.ts";
22
+ import { drainHeldCaseLocks, isCaseLockStale, readCaseLock } from "./runner/lock.ts";
23
+ import { drainHeldGateLeases } from "./runner/gate-lease.ts";
22
24
  import { CLEANUP_TIMEOUT_MS, withCleanupTimeout } from "./runner/cleanup-timeout.ts";
23
25
  import type { ExperimentHookContext } from "./runner/types.ts";
24
26
  import { evalLevelStats } from "./shared/verdict.ts";
@@ -175,7 +177,7 @@ const FLAG_OPTIONS = {
175
177
  // 数字 `--attempt`,选哪个 attempt 由 locator 精确指名,不是「先选 eval 再挑第几次」。
176
178
  /** `show` 命令专用:该 attempt 运行时保存的 Eval 源码,gate/soft 断言标回源码行(证据切面)。 */
177
179
  source: { type: "boolean" },
178
- /** `show` 命令专用:该 attempt 的标准执行事件流(消息、thinking、Skill load、工具调用/结果);有 OTel 时同一节点补时间(证据切面)。单张卡片正文超过 8 KiB 预览预算会被截断,截断尾巴自带 `--expand` 展开句柄。 */
180
+ /** `show` 命令专用:该 attempt 的标准执行事件流(消息、thinking、Skill load、工具调用/结果);有 OTel 时同一节点补时间(证据切面)。每个内容段最多预览前 3 行,截断尾巴自带 `--expand` 展开句柄。 */
179
181
  execution: { type: "boolean" },
180
182
  /** `show` 命令专用:整个 Attempt 的统一时间树;裸 `--timing` 给有界诊断投影,`--timing=full` 逐节点展开全部 runner/已关联 OTel 节点。 */
181
183
  timing: { type: "boolean" },
@@ -545,19 +547,32 @@ async function openBrowser(url: string): Promise<boolean> {
545
547
  function assembleInvocationCompletion(state: RunFeedbackState): InvocationCompletion {
546
548
  let unstarted = 0;
547
549
  let failFastSkipped = 0;
550
+ let haltedSkipped = 0;
548
551
  let interrupted = false;
549
552
  const reporterErrors: ReporterError[] = [];
550
553
  for (const d of state.diagnostics) {
551
- if (d.key === "interrupted") {
554
+ // 归类按**稳定词法** `code`,不按 `key`:`key` 里编着折叠身份(experimentId / evalId /
555
+ // reporter 名),拿它做前缀匹配会在「把身份从 key 里摘出去」时静默失配——记账悄悄归零,
556
+ // 没有任何测试或类型会报警。`code` 省略时回落到 key 的首段:缺省 key 恒是
557
+ // `${code}:${identity}`(见 sink.ts 的 DiagnosticInput),首段即 code。
558
+ const code = d.code ?? d.key.split(":", 1)[0];
559
+ if (code === "interrupted") {
552
560
  interrupted = true;
553
- } else if (d.key.startsWith("budget-exhausted:")) {
561
+ } else if (code === "budget-exhausted") {
554
562
  unstarted += d.count;
555
- } else if (d.key.startsWith("fail-fast:")) {
563
+ } else if (code === "fail-fast") {
556
564
  // run 级 fail-fast 造成的未派发同样计入 unstarted(结论落 incomplete,见
557
565
  // docs/feature/experiments/architecture.md「Completion 与退出」)。
558
566
  unstarted += d.count;
559
567
  failFastSkipped += d.count;
560
- } else if (d.key.startsWith("reporter-error:")) {
568
+ } else if (code === "dispatch-halted") {
569
+ // 止损闸停派发造成的未派发(见 docs/feature/error-classification/architecture.md
570
+ // 「记账」)。这条诊断的 count 是「同一死因被声明了几次」(重复声明折叠),不是未派发数——
571
+ // 未派发数由 emitter 累计后写在 data.unstarted 里(与 budget-exhausted 同一口径)。
572
+ const halted = typeof d.data?.unstarted === "number" ? d.data.unstarted : 0;
573
+ unstarted += halted;
574
+ haltedSkipped += halted;
575
+ } else if (code === "reporter-error") {
561
576
  // required 决定这条错误是否写进 InvocationCompletion.reporterErrors 并让 completion 非 complete
562
577
  // (见 docs/cli.md「required reporter」);best-effort reporter 的失败只保留为 diagnostic。
563
578
  if (d.data?.required !== true) continue;
@@ -571,9 +586,9 @@ function assembleInvocationCompletion(state: RunFeedbackState): InvocationComple
571
586
  // 中断造成的未派发(仍在 queued 的 attempt)同样计入 unstarted(见 docs/feature/experiments/
572
587
  // architecture.md「Completion 与退出」:budget 耗尽、fail-fast 或中断造成的未派发都不伪装成全绿)。
573
588
  if (interrupted) unstarted += state.queued;
574
- // attempt:early-exit 计数含 fail-fast 的未派发(反馈层同一事件驱动计数守恒);
575
- // 「省下的重复验证」= 总数减去 fail-fast 那部分。
576
- const earlyExitUnstarted = Math.max(0, state.earlyExitSkipped - failFastSkipped);
589
+ // attempt:early-exit 计数含 fail-fast 与止损闸的未派发(反馈层同一事件驱动计数守恒);
590
+ // 「省下的重复验证」= 总数减去那两部分。
591
+ const earlyExitUnstarted = Math.max(0, state.earlyExitSkipped - failFastSkipped - haltedSkipped);
577
592
  const status: CompletionStatus = interrupted
578
593
  ? "interrupted"
579
594
  : unstarted > 0 || reporterErrors.length > 0
@@ -875,6 +890,9 @@ async function main(): Promise<void> {
875
890
  maxConcurrency: exp.maxConcurrency,
876
891
  setup: exp.setup,
877
892
  teardown: exp.teardown,
893
+ // 实验级失败分类器:随 AgentRun 进 attempt(turn 链与生命周期链共用同一份),
894
+ // 产出的 scope 由止损闸消费(见 docs/feature/error-classification/architecture.md)。
895
+ classifyFailure: exp.classifyFailure,
878
896
  });
879
897
  }
880
898
  } else {
@@ -926,21 +944,36 @@ async function main(): Promise<void> {
926
944
  // 人读 `--dry` 首行的携入摘要(见 docs/feature/experiments/cli.md 开头示例与「事件与计划
927
945
  // 文档的 TypeScript 形状」),口径必须与真正开跑时一致。
928
946
  const priorResults = flags.force ? undefined : await loadLatestResultsPerEval(join(cwd, ".niceeval"));
929
- const carryPlan = priorResults?.length ? await planCarry(evals, agentRuns, priorResults, config.sandbox) : undefined;
947
+ const carryPlan = priorResults?.length
948
+ ? await planCarry(evals, agentRuns, priorResults, config.sandbox, config.timeoutMs)
949
+ : undefined;
930
950
 
931
951
  if (flags.dry) {
932
952
  // --dry 只按所选形态打印计划,不运行、不落盘——一次完成的读取,不是事件流
933
953
  // (见 docs/feature/experiments/cli.md「机器怎么读:--json」)。两种形态共用同一份摊平
934
954
  // 矩阵——(experimentId, evalId) 逐行,携带同一口径的 reused 预测——不是各自重算一遍。
935
955
  const dryRuns = Math.max(1, ...agentRuns.map((r) => r.runs));
936
- const matrix: JsonPlanRow[] = [];
956
+ const rowInputs: { experimentId: string; evalId: string; reused: boolean }[] = [];
937
957
  for (let i = 0; i < agentRuns.length; i++) {
938
958
  const run = agentRuns[i]!;
939
959
  for (const e of matchedByRun[i]!) {
940
960
  const carriedCount = carryPlan?.carriedAttemptsByKey.get(cacheKey(run, e.id))?.size ?? 0;
941
- matrix.push({ experimentId: run.experimentId ?? "", evalId: e.id, reused: carriedCount >= run.runs });
961
+ rowInputs.push({ experimentId: run.experimentId ?? "", evalId: e.id, reused: carriedCount >= run.runs });
942
962
  }
943
963
  }
964
+ // 只读锁目录,不取锁、不等待(见 docs/feature/experiments/architecture.md「并发
965
+ // Invocation:用例锁」);过期(无人续心跳)的锁不算"正被持锁运行",不标注。裸 run(没有
966
+ // experimentId)不参与锁,恒不标注。并行读——矩阵行数可能不小,不逐行串行等磁盘。
967
+ const niceevalRootForDry = resolvePath(cwd, ".niceeval");
968
+ const now = Date.now();
969
+ const lockedFlags = await Promise.all(
970
+ rowInputs.map(async (row) => {
971
+ if (!row.experimentId) return false;
972
+ const lock = await readCaseLock(niceevalRootForDry, row.experimentId, row.evalId).catch(() => undefined);
973
+ return lock !== undefined && !isCaseLockStale(lock, now);
974
+ }),
975
+ );
976
+ const matrix: JsonPlanRow[] = rowInputs.map((row, i) => ({ ...row, ...(lockedFlags[i] ? { locked: true } : {}) }));
944
977
  if (outputForm === "json") {
945
978
  process.stdout.write(
946
979
  renderJsonPlanDocument({
@@ -959,7 +992,7 @@ async function main(): Promise<void> {
959
992
  configs: agentRuns.length,
960
993
  runs: dryRuns,
961
994
  reused: carryPlan?.carriedResults.length ?? 0,
962
- rows: matrix.map((row) => ({ experimentId: row.experimentId, evalId: row.evalId })),
995
+ rows: matrix.map((row) => ({ experimentId: row.experimentId, evalId: row.evalId, locked: row.locked })),
963
996
  }),
964
997
  );
965
998
  }
@@ -1025,6 +1058,8 @@ async function main(): Promise<void> {
1025
1058
  const settled = Promise.allSettled([
1026
1059
  ...(runInFlight ? [runInFlight] : []),
1027
1060
  drainExperimentTeardowns(),
1061
+ drainHeldCaseLocks(),
1062
+ drainHeldGateLeases(),
1028
1063
  ]);
1029
1064
  await Promise.race([
1030
1065
  settled.then(() => {}),
@@ -1101,9 +1136,12 @@ async function main(): Promise<void> {
1101
1136
  }
1102
1137
 
1103
1138
  // 正常返回(含被中断后走部分汇总)后再兜一刀:Scope finalizer 没停掉的残留沙箱、没被运行
1104
- // 路径消费的实验级 cleanup 在这里强清。跑顺利时两份登记表都已空,是 no-op。
1139
+ // 路径消费的实验级 cleanup、没被 per-attempt Effect.ensuring 释放的用例锁与实验闸租约在这里
1140
+ // 强清。跑顺利时四份登记表都已空,是 no-op。
1105
1141
  await stopAllSandboxes();
1106
1142
  await drainExperimentTeardowns();
1143
+ await drainHeldCaseLocks();
1144
+ await drainHeldGateLeases();
1107
1145
 
1108
1146
  // completion 要先算好,--junit 是否"这次真的写出"才有依据(见下)。
1109
1147
  const completion = assembleInvocationCompletion(coordinator.state);
@@ -1140,8 +1178,11 @@ async function main(): Promise<void> {
1140
1178
 
1141
1179
  main().catch(async (e) => {
1142
1180
  process.stderr.write(t("cli.error", { error: formatThrown(e) }));
1143
- // 真·崩溃路径也别留孤儿:强清还活着的沙箱(带超时)、排空实验级 cleanup 注册表,再退。
1181
+ // 真·崩溃路径也别留孤儿:强清还活着的沙箱(带超时)、排空实验级 cleanup 注册表、用例锁与
1182
+ // 实验闸租约,再退。
1144
1183
  await stopAllSandboxes();
1145
1184
  await drainExperimentTeardowns();
1185
+ await drainHeldCaseLocks();
1186
+ await drainHeldGateLeases();
1146
1187
  process.exit(2);
1147
1188
  });
@@ -355,6 +355,40 @@ function baseScoringContext(state: ContextState) {
355
355
  };
356
356
  }
357
357
 
358
+ describe("t.* 作用域断言聚合全部轮次(callId 跨轮复用)", () => {
359
+ // 回归:续轮场景下 adapter 常按轮各自编号(复用 c1)。第一轮读了 INDEX,第二轮才给答复;
360
+ // t.calledTool 聚合全部轮次,应命中第一轮的 read——旧折叠按 callId 覆盖会让它「只扫最后一轮」而 miss。
361
+ it("t.calledTool 命中发生在第一轮、callId 被第二轮复用的工具调用", async () => {
362
+ const agent = scriptedAgent([
363
+ {
364
+ status: "completed",
365
+ events: [
366
+ { type: "action.called", callId: "c1", name: "read", input: { path: "INDEX.md" } },
367
+ { type: "action.result", callId: "c1", output: "index contents", status: "completed" },
368
+ { type: "message", role: "assistant", text: "读完了 INDEX,继续" },
369
+ ],
370
+ },
371
+ {
372
+ status: "completed",
373
+ events: [
374
+ { type: "action.called", callId: "c1", name: "write", input: { path: "note.md" } },
375
+ { type: "action.result", callId: "c1", output: "ok", status: "completed" },
376
+ { type: "message", role: "assistant", text: "答复" },
377
+ ],
378
+ },
379
+ ]);
380
+ const { context, state } = makeContext(agent);
381
+ await context.send("第一轮"); // 读 INDEX
382
+ await context.send("第二轮"); // 续轮,复用 callId c1
383
+
384
+ context.calledTool("read", { input: { path: "INDEX.md" } });
385
+
386
+ const [result] = await state.collector.finalize(baseScoringContext(state));
387
+ expect(result.name).toBe("calledTool(read)");
388
+ expect(result.outcome).toBe("passed");
389
+ });
390
+ });
391
+
358
392
  describe("TurnHandle scoped assertions (parked/loadedSkill/noFailedActions/maxTokens/maxCost)", () => {
359
393
  it("mirror t/session scope: turn.parked() reflects this turn's own waiting status", async () => {
360
394
  const agent = scriptedAgent([
@@ -12,6 +12,7 @@ import * as Scoped from "../scoring/scoped.ts";
12
12
  import { buildJudge } from "../scoring/judge.ts";
13
13
  import { EvalSkipped, EvalRequirementFailed, TurnFailed } from "./control-flow.ts";
14
14
  import { turnErrorText } from "./turn-errors.ts";
15
+ import { attachFailureClass, type FailureClass } from "../shared/failure-class.ts";
15
16
  import type { ConcurrencySlot } from "./send-retry.ts";
16
17
  import { deriveRunFacts } from "../o11y/derive.ts";
17
18
  import { diffIsEmpty, diffMatches, emptyDiffData } from "../scoring/diff.ts";
@@ -91,6 +92,8 @@ export interface ContextDeps {
91
92
  onTurn?: import("./session.ts").SessionDeps["onTurn"];
92
93
  /** turn 级重试退避期间释放/收回的全局并发槽位;透传给 SessionManager。 */
93
94
  concurrencySlot?: ConcurrencySlot;
95
+ /** 实验声明的失败分类器(`ExperimentDef.classifyFailure`);透传给 SessionManager。 */
96
+ experimentClassifier?: import("./session.ts").SessionDeps["experimentClassifier"];
94
97
  /** 仅供确定性单测注入:透传给 SessionManager 的 turn 重试随机数/睡眠(生产路径省略)。 */
95
98
  retryRandom?: import("./session.ts").SessionDeps["retryRandom"];
96
99
  retrySleep?: import("./session.ts").SessionDeps["retrySleep"];
@@ -131,6 +134,7 @@ export function createEvalContext(deps: ContextDeps): { context: TestContext; st
131
134
  onTurn: deps.onTurn,
132
135
  ledgerHooks: deps.ledgerHooks,
133
136
  concurrencySlot: deps.concurrencySlot,
137
+ experimentClassifier: deps.experimentClassifier,
134
138
  retryRandom: deps.retryRandom,
135
139
  retrySleep: deps.retrySleep,
136
140
  });
@@ -332,11 +336,11 @@ export function createEvalContext(deps: ContextDeps): { context: TestContext; st
332
336
  const text = typeof input === "string" ? input : input.text;
333
337
  const files = typeof input === "string" ? undefined : input.files;
334
338
  const turn = await send(session, text, files);
335
- return makeTurnHandle(turn, collector, deps, text, manager.resolveTurnCoverage(turn));
339
+ return makeTurnHandle(turn, collector, deps, text, manager.resolveTurnCoverage(turn), manager.resolveTurnFailureClass(turn));
336
340
  },
337
341
  sendFile: async (path, text) => {
338
342
  const turn = await send(session, text ?? "", [await readInputFile(path)]);
339
- return makeTurnHandle(turn, collector, deps, text ?? "", manager.resolveTurnCoverage(turn));
343
+ return makeTurnHandle(turn, collector, deps, text ?? "", manager.resolveTurnCoverage(turn), manager.resolveTurnFailureClass(turn));
340
344
  },
341
345
  requireInputRequest: (filter) => requireInputRequest(session, filter),
342
346
  respond: async (...answers) => {
@@ -344,7 +348,7 @@ export function createEvalContext(deps: ContextDeps): { context: TestContext; st
344
348
  const built = buildRespondInput(session, answers);
345
349
  session.pendingInputRequests.length = 0;
346
350
  const turn = await send(session, built.text, undefined, built.responses);
347
- return makeTurnHandle(turn, collector, deps, built.text, manager.resolveTurnCoverage(turn));
351
+ return makeTurnHandle(turn, collector, deps, built.text, manager.resolveTurnCoverage(turn), manager.resolveTurnFailureClass(turn));
348
352
  },
349
353
  respondAll: async (optionId) => {
350
354
  if (session.pendingInputRequests.length === 0) {
@@ -359,7 +363,7 @@ export function createEvalContext(deps: ContextDeps): { context: TestContext; st
359
363
  session.pendingInputRequests.length = 0;
360
364
  const input = requests.map(() => optionId).join("\n");
361
365
  const turn = await send(session, input, undefined, responses);
362
- return makeTurnHandle(turn, collector, deps, input, manager.resolveTurnCoverage(turn));
366
+ return makeTurnHandle(turn, collector, deps, input, manager.resolveTurnCoverage(turn), manager.resolveTurnFailureClass(turn));
363
367
  },
364
368
  get reply() {
365
369
  return session.lastMessage;
@@ -565,6 +569,7 @@ function makeTurnHandle(
565
569
  deps: ContextDeps,
566
570
  input: string,
567
571
  coverage: ResolvedCoverage,
572
+ failureClass?: FailureClass,
568
573
  ): TurnHandle {
569
574
  const message = lastAssistantText(turn.events) ?? "";
570
575
  const facts = deriveRunFacts(turn.events);
@@ -598,7 +603,11 @@ function makeTurnHandle(
598
603
  // 与保守兜底分类器、turn 级重试摘要读的同一段文本(见 turn-errors.ts 的 turnErrorText)——
599
604
  // 不出现「报错说 A、分类看 B」。
600
605
  const message = turnErrorText(turn);
601
- throw new TurnFailed(message !== undefined ? t("context.turnFailed", { message }) : undefined);
606
+ const error = new TurnFailed(message !== undefined ? t("context.turnFailed", { message }) : undefined);
607
+ // 终局失败的分类随错误浮出:attempt 封口据此落止损闸(见
608
+ // docs/feature/error-classification/architecture.md「止损执行体」)。没有分类
609
+ // (未经重试执行体的手工构造 Turn)时不标记,落缺省 attempt 档。
610
+ throw failureClass ? attachFailureClass(error, failureClass) : error;
602
611
  }
603
612
  return handle;
604
613
  },