niceeval 0.10.3-canary.6 → 0.11.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (122) hide show
  1. package/bin/niceeval.js +9 -4
  2. package/dist/agents/types.d.ts +5 -4
  3. package/dist/context/turn-errors.d.ts +27 -23
  4. package/dist/i18n/en.d.ts +15 -0
  5. package/dist/i18n/en.js +16 -1
  6. package/dist/i18n/zh-CN.d.ts +16 -1
  7. package/dist/i18n/zh-CN.js +16 -1
  8. package/dist/o11y/derive.js +28 -24
  9. package/dist/o11y/types.d.ts +8 -5
  10. package/dist/report/components/attempt-detail/UsageTable.js +4 -7
  11. package/dist/report/components/attempt-detail/compute.d.ts +2 -2
  12. package/dist/report/components/attempt-detail/compute.js +2 -6
  13. package/dist/report/components/attempt-detail/faces.js +7 -7
  14. package/dist/report/components/attempt-detail/index.js +0 -3
  15. package/dist/report/components/entity-lists/EvalList.js +0 -0
  16. package/dist/report/components/metric-views/compute.js +1 -1
  17. package/dist/report/model/types.d.ts +3 -4
  18. package/dist/results/locator.js +0 -0
  19. package/dist/results/select.d.ts +6 -0
  20. package/dist/results/select.js +8 -0
  21. package/dist/runner/feedback/sink.d.ts +26 -1
  22. package/dist/runner/fingerprint.d.ts +23 -0
  23. package/dist/runner/types.d.ts +105 -7
  24. package/dist/sandbox/errors.d.ts +29 -0
  25. package/dist/sandbox/resolve.d.ts +9 -0
  26. package/dist/shared/failure-class.d.ts +91 -0
  27. package/dist/types.d.ts +1 -0
  28. package/dist/util.d.ts +3 -2
  29. package/dist/util.js +31 -5
  30. package/docs-site/zh/explanation/runner.mdx +35 -0
  31. package/docs-site/zh/reference/cli.mdx +10 -1
  32. package/docs-site/zh/reference/events.mdx +4 -4
  33. package/docs-site/zh/troubleshooting/debugging.mdx +11 -0
  34. package/docs-site/zh/tutorials/viewing-results.mdx +35 -3
  35. package/package.json +28 -24
  36. package/src/agents/ai-sdk.test.ts +26 -0
  37. package/src/agents/ai-sdk.ts +7 -4
  38. package/src/agents/index.ts +5 -3
  39. package/src/agents/langgraph.test.ts +30 -0
  40. package/src/agents/langgraph.ts +5 -2
  41. package/src/agents/openai-compat.test.ts +35 -0
  42. package/src/agents/openai-compat.ts +16 -4
  43. package/src/agents/sdk-streams.test.ts +66 -0
  44. package/src/agents/sdk-streams.ts +3 -1
  45. package/src/agents/types.ts +5 -4
  46. package/src/cli.ts +81 -16
  47. package/src/context/context.test.ts +34 -0
  48. package/src/context/context.ts +14 -5
  49. package/src/context/send-retry.test.ts +86 -0
  50. package/src/context/send-retry.ts +37 -12
  51. package/src/context/session.ts +24 -0
  52. package/src/context/turn-errors.test.ts +124 -17
  53. package/src/context/turn-errors.ts +60 -50
  54. package/src/define.ts +5 -0
  55. package/src/i18n/en.ts +22 -1
  56. package/src/i18n/zh-CN.ts +22 -1
  57. package/src/index.ts +8 -0
  58. package/src/o11y/cost.ts +4 -2
  59. package/src/o11y/derive.test.ts +40 -0
  60. package/src/o11y/derive.ts +28 -22
  61. package/src/o11y/otlp/sandbox-receiver.test.ts +201 -0
  62. package/src/o11y/otlp/sandbox-receiver.ts +73 -27
  63. package/src/o11y/parsers/bub.test.ts +30 -0
  64. package/src/o11y/parsers/bub.ts +5 -2
  65. package/src/o11y/parsers/codex.test.ts +19 -0
  66. package/src/o11y/parsers/codex.ts +5 -2
  67. package/src/o11y/types.ts +8 -5
  68. package/src/report/components/attempt-detail/UsageTable.tsx +4 -6
  69. package/src/report/components/attempt-detail/attempt-components.test.tsx +5 -8
  70. package/src/report/components/attempt-detail/compute.ts +2 -7
  71. package/src/report/components/attempt-detail/faces.ts +7 -7
  72. package/src/report/components/attempt-detail/index.tsx +0 -3
  73. package/src/report/components/entity-lists/EvalList.tsx +0 -0
  74. package/src/report/components/metric-views/compute.ts +1 -1
  75. package/src/report/model/types.ts +3 -4
  76. package/src/results/format.ts +9 -2
  77. package/src/results/index.ts +2 -0
  78. package/src/results/locator.ts +0 -0
  79. package/src/results/open.ts +132 -21
  80. package/src/results/select.ts +12 -0
  81. package/src/results/skipped-notice.ts +0 -0
  82. package/src/runner/attempt.test.ts +116 -0
  83. package/src/runner/attempt.ts +89 -8
  84. package/src/runner/discover.ts +21 -3
  85. package/src/runner/feedback/coordinator.ts +24 -2
  86. package/src/runner/feedback/eval-conclusions.ts +6 -3
  87. package/src/runner/feedback/human.test.ts +300 -6
  88. package/src/runner/feedback/human.ts +132 -30
  89. package/src/runner/feedback/json.test.ts +127 -2
  90. package/src/runner/feedback/json.ts +51 -3
  91. package/src/runner/feedback/reducer.test.ts +328 -29
  92. package/src/runner/feedback/reducer.ts +84 -9
  93. package/src/runner/feedback/sink.ts +39 -1
  94. package/src/runner/fingerprint.ts +49 -19
  95. package/src/runner/gate-lease.test.ts +510 -0
  96. package/src/runner/gate-lease.ts +350 -0
  97. package/src/runner/lock.test.ts +454 -0
  98. package/src/runner/lock.ts +288 -0
  99. package/src/runner/report.test.ts +1 -0
  100. package/src/runner/run.test.ts +2044 -9
  101. package/src/runner/run.ts +825 -61
  102. package/src/runner/teardown-registry.ts +20 -78
  103. package/src/runner/types.ts +103 -7
  104. package/src/sandbox/errors.test.ts +72 -0
  105. package/src/sandbox/errors.ts +83 -0
  106. package/src/sandbox/keep-registry.ts +22 -46
  107. package/src/sandbox/resolve.test.ts +100 -0
  108. package/src/sandbox/resolve.ts +84 -27
  109. package/src/shared/entry-file-store.test.ts +149 -0
  110. package/src/shared/entry-file-store.ts +117 -0
  111. package/src/shared/failure-class.test.ts +137 -0
  112. package/src/shared/failure-class.ts +175 -0
  113. package/src/show/index.ts +5 -6
  114. package/src/show/render.test.ts +116 -15
  115. package/src/show/render.ts +124 -35
  116. package/src/types.ts +9 -0
  117. package/src/util.ts +31 -4
  118. package/src/view/app/App.tsx +5 -1
  119. package/src/view/client-dist/app.js +1 -1
  120. package/src/view/data.ts +5 -9
  121. package/src/view/shared/types.ts +7 -1
  122. package/src/view/view-report.test.ts +54 -0
@@ -93,6 +93,16 @@ $ niceeval show @1qrdcfq8 --execution | grep "TOOL ·" | sort | uniq -c
93
93
  niceeval show @1qrdcfq8 --execution | grep proposals
94
94
  ```
95
95
 
96
+ 超长的工具结果只显示前 3 行预览,但截断尾巴自带这张卡片的展开句柄——整行复制就是看全量的下一条命令,不用去翻 `events.json`:
97
+
98
+ ```text
99
+ result · completed · exit 0
100
+ Proposal 1: …
101
+ (+412 lines · 18553 chars · niceeval show @1qrdcfq8 --execution --expand t1.c17)
102
+
103
+ $ niceeval show @1qrdcfq8 --execution --expand t1.c17 # 只输出这张卡片,完整不截断
104
+ ```
105
+
96
106
  **3. 看它到底改了什么。** `--diff` 只显示 **agent 自己改动的文件**——你上传的起始文件、跑完后写入的验证材料不会混在里面,所以列表里的每一行都真的是 agent 干的:
97
107
 
98
108
  ```text
@@ -225,6 +235,7 @@ niceeval view --results site-data/run
225
235
  | 断言挂了,不知道为什么 | `show @loc` → `show @loc --source` |
226
236
  | 想知道 agent 当时做了什么 | `show @loc --execution` |
227
237
  | 想确认有没有调用过某个工具 / 搜对话关键词 | `show @loc --execution \| grep "TOOL ·" \| sort \| uniq -c` / `\| grep <关键词>` |
238
+ | 工具结果被截断,想看完整内容 | 复制截断尾巴里的命令:`show @loc --execution --expand t2.c3` |
228
239
  | 想确认 agent 改了哪些文件 | `show @loc --diff` → `--diff=<path>` |
229
240
  | 哪一步慢 / 超时死在哪 | `show @loc`(看 `timing:` 行)→ `show @loc --timing` |
230
241
  | Sandbox 创建就失败(配额 / 凭据 / 镜像) | `show @loc` 看 error 的 code 与 cause → 查账号配额、核对凭据、降 `--max-concurrency`(没有现场可留) |
@@ -57,6 +57,7 @@ niceeval show weather/brooklyn # 收窄到一个 eval:同一份榜单,
57
57
  niceeval show @1k2m9qrs # 精确到一次 Attempt:断言、执行、diff 与可用证据摘要
58
58
  niceeval show @1k2m9qrs --source # 该 Attempt 运行时保存的 Eval 源码,断言标回源码行
59
59
  niceeval show @1k2m9qrs --execution # 该 Attempt 的消息、thinking、Skill 加载、工具调用,有 OTel 时补时间
60
+ niceeval show @1k2m9qrs --execution --expand t2.c3 # 展开一张被截断卡片的完整落盘内容,句柄从截断尾巴里抄
60
61
  niceeval show @1c3h6twx --timing # 有界诊断时间树:phase、hook、operation、shell、turn、OTel 与收尾
61
62
  niceeval show @1c3h6twx --timing=full # 同一棵时间树逐节点完整展开
62
63
  niceeval show @1c3h6twx --diff # sandbox 里的文件改动
@@ -251,6 +252,38 @@ full events: fixtures/button/a1/events.json
251
252
  full OTel trace: fixtures/button/a1/trace.json
252
253
  ```
253
254
 
255
+ 超长的工具结果既不整段打印、也不被一刀切丢掉。卡片正文是**有界预览**:每个内容段(消息正文;工具卡的 input 与 result 各算一段)最多显示前 3 行,保留原始换行,另有每段 1 KiB 兜底防单行超长。有折叠的卡片在卡尾报被折的行数与字符数,并自带这张卡片的**展开句柄**——整行就是下一条可直接复制的命令。预览负责回答「这一步做了什么、结果开头长什么样」,让整个 attempt 的树落在一两屏内;要看全量再花第二条命令:
256
+
257
+ ```text
258
+ $ niceeval show @1c3h6twx --execution
259
+ …
260
+ TOOL · command_execution 5.2s · 1.9s
261
+ input
262
+ /bin/bash -lc 'rg --files node_modules/niceeval | sort'
263
+ result · completed · exit 0
264
+ node_modules/niceeval/AGENTS.md
265
+ node_modules/niceeval/INDEX.md
266
+ node_modules/niceeval/INIT.md
267
+ (+3055 lines · 263161 chars · niceeval show @1c3h6twx --execution --expand t2.c3)
268
+ ```
269
+
270
+ `--expand` 接截断尾巴里的句柄,单独输出这一张卡片的完整落盘内容(原始换行,不再截断),其余卡片不打印;范围必须恰好命中一个 attempt。句柄 `t<轮次>.c<轮内卡序>`(失败 Sandbox 命令卡是 `cmd<n>`)由 `events.json` 的事件序确定性派生,同一 attempt 反复解析恒定——昨天终端记录里抄来的句柄今天照样有效:
271
+
272
+ ```text
273
+ $ niceeval show @1c3h6twx --execution --expand t2.c3
274
+ @1c3h6twx · fixtures/button · compare/bub-gpt-5.4 · errored
275
+
276
+ TOOL · command_execution 5.2s · 1.9s
277
+ input
278
+ /bin/bash -lc 'rg --files node_modules/niceeval | sort'
279
+ result · completed · exit 0
280
+ node_modules/niceeval/AGENTS.md
281
+ node_modules/niceeval/INDEX.md
282
+ …完整落盘内容逐行继续,不再截断…
283
+ ```
284
+
285
+ 两点边界:展开还原的是**落盘证据**——单个值落盘时有 256 KiB 上限,超限的值展开后如实标注 `truncated` 与原始字节数,不冒充运行时全量;`--json` 面恒输出完整值、从不截断,因此它与 `--expand` 互斥——没有可展开的东西。
286
+
254
287
  没有 OTel 接入时,同一棵树去掉时间列,节点内容原样保留:
255
288
 
256
289
  ```text
@@ -491,7 +524,7 @@ locator eval result turns tools un
491
524
  total 6/8 passed 16 16 880.6k 3.5M* 54.8k 131 $4.05
492
525
  ```
493
526
 
494
- `uncached in` 与 `cache read` 是同一份 `inputTokens` 的两半拆分:协议逐轮重发上下文时,膨胀多在 `cache read`,这层拆分才能回答「贵在哪」,只看总 token 数看不出来。缺字段的格显示 `—`,不按 0 计入合计;合计列有缺失参与时标 `*`。
527
+ `uncached in`(`inputTokens`,未命中缓存的输入)与 `cache read`(`cacheReadTokens`,缓存命中)是两个互斥的计价桶,相加才是每轮真正送进模型的上下文量:协议逐轮重发上下文时,膨胀多在 `cache read`,这层拆分才能回答「贵在哪」,只看单个数字看不出来。缺字段的格显示 `—`,不按 0 计入合计;合计列有缺失参与时标 `*`。
495
528
 
496
529
  ### agent 到底 search 了没:`--execution --grep`
497
530
 
@@ -537,8 +570,7 @@ niceeval show --exp compare/claude-mempal --usage --json
537
570
  "locator": "@1xc8bnj5",
538
571
  "evalId": "memory/agent-037-updatetag-cache",
539
572
  "verdict": "passed",
540
- "usage": { "inputTokens": 249700, "outputTokens": 5100, "cacheReadTokens": 201500, "requests": 13 },
541
- "uncachedInputTokens": 48200,
573
+ "usage": { "inputTokens": 48200, "outputTokens": 5100, "cacheReadTokens": 201500, "requests": 13 },
542
574
  "estimatedCostUSD": 0.44
543
575
  }
544
576
  ]
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "niceeval",
3
- "version": "0.10.3-canary.6",
3
+ "version": "0.11.0",
4
4
  "description": "Agent-native eval tool — eval agents, services, functions, and coding-agent fixtures",
5
5
  "type": "module",
6
6
  "license": "MIT",
@@ -18,51 +18,63 @@
18
18
  "exports": {
19
19
  ".": {
20
20
  "types": "./src/index.ts",
21
- "import": "./src/index.ts"
21
+ "import": "./src/index.ts",
22
+ "require": "./src/index.ts"
22
23
  },
23
24
  "./sandbox": {
24
25
  "types": "./src/sandbox/index.ts",
25
- "import": "./src/sandbox/index.ts"
26
+ "import": "./src/sandbox/index.ts",
27
+ "require": "./src/sandbox/index.ts"
26
28
  },
27
29
  "./sandbox/e2b-template": {
28
30
  "types": "./src/sandbox/e2b-agent-template.ts",
29
- "import": "./src/sandbox/e2b-agent-template.ts"
31
+ "import": "./src/sandbox/e2b-agent-template.ts",
32
+ "require": "./src/sandbox/e2b-agent-template.ts"
30
33
  },
31
34
  "./adapter": {
32
35
  "types": "./src/agents/index.ts",
33
- "import": "./src/agents/index.ts"
36
+ "import": "./src/agents/index.ts",
37
+ "require": "./src/agents/index.ts"
34
38
  },
35
39
  "./adapter/otel": {
36
40
  "types": "./src/agents/ai-sdk-otel.ts",
37
- "import": "./src/agents/ai-sdk-otel.ts"
41
+ "import": "./src/agents/ai-sdk-otel.ts",
42
+ "require": "./src/agents/ai-sdk-otel.ts"
38
43
  },
39
44
  "./expect": {
40
45
  "types": "./src/expect/index.ts",
41
- "import": "./src/expect/index.ts"
46
+ "import": "./src/expect/index.ts",
47
+ "require": "./src/expect/index.ts"
42
48
  },
43
49
  "./reporters": {
44
50
  "types": "./src/runner/reporters/index.ts",
45
- "import": "./src/runner/reporters/index.ts"
51
+ "import": "./src/runner/reporters/index.ts",
52
+ "require": "./src/runner/reporters/index.ts"
46
53
  },
47
54
  "./loaders": {
48
55
  "types": "./src/loaders/index.ts",
49
- "import": "./src/loaders/index.ts"
56
+ "import": "./src/loaders/index.ts",
57
+ "require": "./src/loaders/index.ts"
50
58
  },
51
59
  "./results": {
52
60
  "types": "./src/results/index.ts",
53
- "import": "./src/results/index.ts"
61
+ "import": "./src/results/index.ts",
62
+ "require": "./src/results/index.ts"
54
63
  },
55
64
  "./report": {
56
65
  "types": "./dist/report/index.d.ts",
57
- "import": "./dist/report/index.js"
66
+ "import": "./dist/report/index.js",
67
+ "require": "./dist/report/index.js"
58
68
  },
59
69
  "./report/react": {
60
70
  "types": "./dist/report/react/index.d.ts",
61
- "import": "./dist/report/react/index.js"
71
+ "import": "./dist/report/react/index.js",
72
+ "require": "./dist/report/react/index.js"
62
73
  },
63
74
  "./report/built-in": {
64
75
  "types": "./dist/report/built-in/index.d.ts",
65
- "import": "./dist/report/built-in/index.js"
76
+ "import": "./dist/report/built-in/index.js",
77
+ "require": "./dist/report/built-in/index.js"
66
78
  },
67
79
  "./report/react/styles.css": "./src/report/assets/styles.css",
68
80
  "./report/react/enhance.js": "./src/report/assets/enhance.js"
@@ -81,6 +93,8 @@
81
93
  "autoevals": "0.0.132",
82
94
  "effect": "^3.21.4",
83
95
  "mermaid": "^11.16.0",
96
+ "react": "^19.0.0",
97
+ "react-dom": "^19.0.0",
84
98
  "tar-stream": "^3.1.7",
85
99
  "tsx": "^4.19.2"
86
100
  },
@@ -109,8 +123,6 @@
109
123
  "mixpanel-browser": "^2.80.0",
110
124
  "next": "16.2.10",
111
125
  "prism-react-renderer": "^2.4.1",
112
- "react": "^19.2.7",
113
- "react-dom": "^19.2.7",
114
126
  "react-grab": "^0.1.48",
115
127
  "shiki": "^4.3.0",
116
128
  "tailwind-merge": "^3.6.0",
@@ -127,9 +139,7 @@
127
139
  "ai": ">=5.0.0",
128
140
  "braintrust": ">=0.0.150",
129
141
  "dockerode": ">=4.0.0",
130
- "e2b": ">=2.0.0",
131
- "react": ">=18",
132
- "react-dom": ">=18"
142
+ "e2b": ">=2.0.0"
133
143
  },
134
144
  "peerDependenciesMeta": {
135
145
  "@ai-sdk/otel": {
@@ -155,12 +165,6 @@
155
165
  },
156
166
  "e2b": {
157
167
  "optional": true
158
- },
159
- "react": {
160
- "optional": true
161
- },
162
- "react-dom": {
163
- "optional": true
164
168
  }
165
169
  },
166
170
  "scripts": {
@@ -0,0 +1,26 @@
1
+ // cases: docs/engineering/testing/unit/results.md
2
+ // 「Usage、facts 与失败命令证据落盘」桶恒互斥归一:AI SDK 的 inputTokens 是含缓存明细的
3
+ // 输入总量,落桶前扣掉在场的 cacheRead / cacheWrite 明细。
4
+ // bug: memory/estimatecost-openai-inclusive-cache-double-billed.md
5
+
6
+ import { describe, expect, it } from "vitest";
7
+
8
+ import { fromAiSdk } from "./ai-sdk.ts";
9
+
10
+ describe("fromAiSdk usage 归一(含明细口径)", () => {
11
+ it("v5 形状:cachedInputTokens 从 inputTokens 里扣出", () => {
12
+ const turn = fromAiSdk({
13
+ text: "ok",
14
+ usage: { inputTokens: 1000, outputTokens: 20, cachedInputTokens: 900 },
15
+ });
16
+ expect(turn.usage).toMatchObject({ inputTokens: 100, cacheReadTokens: 900, outputTokens: 20 });
17
+ });
18
+
19
+ it("v7 形状:inputTokenDetails 的 cacheRead 与 cacheWrite 都从总量里扣出", () => {
20
+ const turn = fromAiSdk({
21
+ text: "ok",
22
+ usage: { inputTokens: 1000, outputTokens: 20, inputTokenDetails: { cacheReadTokens: 800, cacheWriteTokens: 100 } },
23
+ });
24
+ expect(turn.usage).toMatchObject({ inputTokens: 100, cacheReadTokens: 800, cacheCreationTokens: 100 });
25
+ });
26
+ });
@@ -327,13 +327,16 @@ function unwrapToolOutput(output: unknown): { output?: JsonValue; status: "compl
327
327
  function readUsage(result: AiSdkResultLike, stepCount: number): Usage | undefined {
328
328
  const u = result.totalUsage ?? result.usage;
329
329
  if (!u) return undefined;
330
- const inputTokens = num(u.inputTokens) ?? num(u.promptTokens) ?? 0;
330
+ const rawInput = num(u.inputTokens) ?? num(u.promptTokens) ?? 0;
331
331
  const outputTokens = num(u.outputTokens) ?? num(u.completionTokens) ?? 0;
332
- if (inputTokens === 0 && outputTokens === 0) return undefined;
332
+ if (rawInput === 0 && outputTokens === 0) return undefined;
333
+ const cacheRead = num(u.cachedInputTokens) ?? num(u.inputTokenDetails?.cacheReadTokens) ?? 0;
334
+ const cacheCreation = num(u.inputTokenDetails?.cacheWriteTokens) ?? 0;
335
+ // AI SDK 的 inputTokens 是含缓存明细的输入总量,落互斥桶前扣掉在场的明细
336
+ // (docs/feature/adapters/sdk/ai-sdk/cost.md)
337
+ const inputTokens = Math.max(0, rawInput - cacheRead - cacheCreation);
333
338
  const usage: Usage = { inputTokens, outputTokens, requests: Math.max(stepCount, 1) };
334
- const cacheRead = num(u.cachedInputTokens) ?? num(u.inputTokenDetails?.cacheReadTokens);
335
339
  if (cacheRead) usage.cacheReadTokens = cacheRead;
336
- const cacheCreation = num(u.inputTokenDetails?.cacheWriteTokens);
337
340
  if (cacheCreation) usage.cacheCreationTokens = cacheCreation;
338
341
  const reasoning = num(u.reasoningTokens);
339
342
  if (reasoning) usage.reasoningTokens = reasoning;
@@ -9,11 +9,13 @@ export type { Shared } from "./shared.ts";
9
9
  export { completeCoverage } from "../scoring/coverage.ts";
10
10
  export type { CoverageStatus, CoverageDeclaration, EvidenceCoverage } from "../types.ts";
11
11
 
12
- // turn 级瞬时错误分类:`Agent.classifyTurnError` 认的输入/输出形状 + 摘要取值器
13
- // (与 turn-failed 报错文案同源)。分类判据、分类链与重试执行体见
12
+ // 执行失败分类:`Agent.classifyTurnError` 认的输入/输出形状 + 摘要取值器(与 turn-failed
13
+ // 报错文案同源)。两轴词表(FailureClass / FailureScope)与包根导出的是同一个形状——adapter
14
+ // 作者与 eval 作者各自的入口拿到同一份类型。判据、分类链与重试执行体见
14
15
  // docs/feature/error-classification/architecture.md。
15
16
  export { turnErrorText } from "../context/turn-errors.ts";
16
- export type { TurnErrorClass, TurnErrorClassifier, TurnFailure } from "../context/turn-errors.ts";
17
+ export type { TurnErrorClassifier, TurnFailure } from "../context/turn-errors.ts";
18
+ export type { FailureClass, FailureScope } from "../shared/failure-class.ts";
17
19
 
18
20
  // span → canonical GenAI 归一(只服务瀑布图,不喂断言)。私有埋点写自己的 spanMapper 时用:
19
21
  // tagSpan 把判定写回 span(原属性只增不改),heuristicTag 是通用兜底判定;mapCodexSpans 是
@@ -0,0 +1,30 @@
1
+ // cases: docs/engineering/testing/unit/results.md
2
+ // 「Usage、facts 与失败命令证据落盘」桶恒互斥归一:LangChain usage_metadata 的 input_tokens
3
+ // 是含缓存读写的输入总量,落桶前扣掉 input_token_details 的 cache_read / cache_creation。
4
+ // bug: memory/estimatecost-openai-inclusive-cache-double-billed.md
5
+
6
+ import { describe, expect, it } from "vitest";
7
+
8
+ import { fromLangGraphEvents } from "./langgraph.ts";
9
+
10
+ describe("fromLangGraphEvents usage 归一(含明细口径)", () => {
11
+ it("cache_read 与 cache_creation 都从 input_tokens 里扣出", () => {
12
+ const stream = fromLangGraphEvents();
13
+ stream.add({
14
+ channel: "messages",
15
+ event: "finish",
16
+ data: {
17
+ message: {
18
+ role: "assistant",
19
+ content: "ok",
20
+ usage_metadata: {
21
+ input_tokens: 1000,
22
+ output_tokens: 20,
23
+ input_token_details: { cache_read: 800, cache_creation: 100 },
24
+ },
25
+ },
26
+ },
27
+ });
28
+ expect(stream.usage).toMatchObject({ inputTokens: 100, cacheReadTokens: 800, cacheCreationTokens: 100, outputTokens: 20 });
29
+ });
30
+ });
@@ -152,12 +152,15 @@ export function fromLangGraphEvents(): LangGraphStream {
152
152
  }
153
153
  return 0;
154
154
  };
155
- const input = num("input_tokens", "inputTokens");
155
+ const rawInput = num("input_tokens", "inputTokens");
156
156
  const output = num("output_tokens", "outputTokens");
157
- if (input === 0 && output === 0) return;
157
+ if (rawInput === 0 && output === 0) return;
158
158
  const details = isRecord(raw.input_token_details) ? raw.input_token_details : undefined;
159
159
  const cacheRead = typeof details?.cache_read === "number" ? details.cache_read : 0;
160
160
  const cacheCreation = typeof details?.cache_creation === "number" ? details.cache_creation : 0;
161
+ // LangChain usage_metadata 的 input_tokens 是含缓存读写的输入总量,落互斥桶前扣掉明细
162
+ // (docs/feature/adapters/sdk/langgraph/cost.md)
163
+ const input = Math.max(0, rawInput - cacheRead - cacheCreation);
161
164
  // LangChain UsageMetadata.output_token_details.reasoning:推理模型经 LangGraph 透传时带回。
162
165
  const outputDetails = isRecord(raw.output_token_details) ? raw.output_token_details : undefined;
163
166
  const reasoning = typeof outputDetails?.reasoning === "number" ? outputDetails.reasoning : 0;
@@ -0,0 +1,35 @@
1
+ // cases: docs/engineering/testing/unit/results.md
2
+ // 「Usage、facts 与失败命令证据落盘」桶恒互斥归一:Chat Completions / Responses 两种形状的
3
+ // cached_tokens 都是输入总量的子集,落 inputTokens 前扣掉;缺 cached 明细时总量原样保留。
4
+ // bug: memory/estimatecost-openai-inclusive-cache-double-billed.md
5
+
6
+ import { describe, expect, it } from "vitest";
7
+
8
+ import { fromChatCompletion, fromResponses } from "./openai-compat.ts";
9
+
10
+ describe("openai-compat usage 归一(OpenAI 口径)", () => {
11
+ it("Chat Completions:prompt_tokens 扣掉 prompt_tokens_details.cached_tokens", () => {
12
+ const turn = fromChatCompletion({
13
+ choices: [{ message: { role: "assistant", content: "ok" } }],
14
+ usage: { prompt_tokens: 1000, completion_tokens: 20, prompt_tokens_details: { cached_tokens: 900 } },
15
+ });
16
+ expect(turn.usage).toMatchObject({ inputTokens: 100, cacheReadTokens: 900, outputTokens: 20 });
17
+ });
18
+
19
+ it("Responses:input_tokens 扣掉 input_tokens_details.cached_tokens", () => {
20
+ const turn = fromResponses({
21
+ output: [{ type: "message", role: "assistant", content: [{ type: "output_text", text: "ok" }] }],
22
+ usage: { input_tokens: 500, output_tokens: 10, input_tokens_details: { cached_tokens: 200 } },
23
+ });
24
+ expect(turn.usage).toMatchObject({ inputTokens: 300, cacheReadTokens: 200, outputTokens: 10 });
25
+ });
26
+
27
+ it("缺 cached 明细时输入总量原样保留,不虚构扣减,cache 桶省略", () => {
28
+ const turn = fromChatCompletion({
29
+ choices: [{ message: { role: "assistant", content: "ok" } }],
30
+ usage: { prompt_tokens: 1000, completion_tokens: 20 },
31
+ });
32
+ expect(turn.usage?.inputTokens).toBe(1000);
33
+ expect(turn.usage?.cacheReadTokens).toBeUndefined();
34
+ });
35
+ });
@@ -50,8 +50,14 @@ export interface ChatCompletionLike {
50
50
 
51
51
  function chatCompletionUsage(usage: ChatCompletionUsageLike | undefined): Usage | undefined {
52
52
  if (!usage) return undefined;
53
- const u: Usage = { inputTokens: usage.prompt_tokens ?? 0, outputTokens: usage.completion_tokens ?? 0 };
54
- if (usage.prompt_tokens_details?.cached_tokens) u.cacheReadTokens = usage.prompt_tokens_details.cached_tokens;
53
+ // prompt_tokens 含缓存命中,cached_tokens 是其子集;落互斥桶前扣掉
54
+ // (docs/feature/adapters/sdk/openai-compat/cost.md)
55
+ const cached = usage.prompt_tokens_details?.cached_tokens ?? 0;
56
+ const u: Usage = {
57
+ inputTokens: Math.max(0, (usage.prompt_tokens ?? 0) - cached),
58
+ outputTokens: usage.completion_tokens ?? 0,
59
+ };
60
+ if (cached) u.cacheReadTokens = cached;
55
61
  if (usage.completion_tokens_details?.reasoning_tokens) u.reasoningTokens = usage.completion_tokens_details.reasoning_tokens;
56
62
  return u;
57
63
  }
@@ -125,8 +131,14 @@ export interface ResponseLike {
125
131
 
126
132
  function responsesUsage(usage: ResponseUsageLike | undefined): Usage | undefined {
127
133
  if (!usage) return undefined;
128
- const u: Usage = { inputTokens: usage.input_tokens ?? 0, outputTokens: usage.output_tokens ?? 0 };
129
- if (usage.input_tokens_details?.cached_tokens) u.cacheReadTokens = usage.input_tokens_details.cached_tokens;
134
+ // input_tokens 含缓存命中,cached_tokens 是其子集;落互斥桶前扣掉
135
+ // (docs/feature/adapters/sdk/openai-compat/cost.md)
136
+ const cached = usage.input_tokens_details?.cached_tokens ?? 0;
137
+ const u: Usage = {
138
+ inputTokens: Math.max(0, (usage.input_tokens ?? 0) - cached),
139
+ outputTokens: usage.output_tokens ?? 0,
140
+ };
141
+ if (cached) u.cacheReadTokens = cached;
130
142
  if (usage.output_tokens_details?.reasoning_tokens) u.reasoningTokens = usage.output_tokens_details.reasoning_tokens;
131
143
  return u;
132
144
  }
@@ -0,0 +1,66 @@
1
+ // cases: docs/engineering/testing/unit/results.md
2
+ // 「Usage、facts 与失败命令证据落盘」桶恒互斥归一:codex(OpenAI 口径,cached ⊂ input)扣减、
3
+ // Anthropic / pi(互斥口径)如实转发。fixture 数值刻意让「扣与不扣」结果可区分。
4
+ // bug: memory/estimatecost-openai-inclusive-cache-double-billed.md
5
+
6
+ import { describe, expect, it } from "vitest";
7
+
8
+ import { fromClaudeSdkMessages, fromCodexThreadEvents, fromPiAgentEvents } from "./sdk-streams.ts";
9
+
10
+ describe("fromCodexThreadEvents usage 归一(OpenAI 口径)", () => {
11
+ it("cached_input_tokens 是 input_tokens 子集:落 inputTokens 前扣掉,cache 单独成桶", () => {
12
+ const stream = fromCodexThreadEvents();
13
+ stream.add({ type: "turn.completed", usage: { input_tokens: 1000, cached_input_tokens: 900, output_tokens: 50 } });
14
+ expect(stream.usage).toMatchObject({ inputTokens: 100, cacheReadTokens: 900, outputTokens: 50, requests: 1 });
15
+ });
16
+
17
+ it("逐轮累加在扣减之后进行,总量仍互斥", () => {
18
+ const stream = fromCodexThreadEvents();
19
+ stream.add({ type: "turn.completed", usage: { input_tokens: 1000, cached_input_tokens: 900, output_tokens: 50 } });
20
+ stream.add({ type: "turn.completed", usage: { input_tokens: 2000, cached_input_tokens: 1700, output_tokens: 30 } });
21
+ expect(stream.usage).toMatchObject({ inputTokens: 400, cacheReadTokens: 2600, outputTokens: 80, requests: 2 });
22
+ });
23
+
24
+ it("协议报出 cached > input 的病态数据时扣减夹底到 0,不产生负 token", () => {
25
+ const stream = fromCodexThreadEvents();
26
+ stream.add({ type: "turn.completed", usage: { input_tokens: 100, cached_input_tokens: 200, output_tokens: 1 } });
27
+ expect(stream.usage?.inputTokens).toBe(0);
28
+ expect(stream.usage?.cacheReadTokens).toBe(200);
29
+ });
30
+ });
31
+
32
+ describe("fromClaudeSdkMessages usage 转发(Anthropic 互斥口径)", () => {
33
+ it("input_tokens 原生不含 cache read:如实转发,不做扣减", () => {
34
+ const stream = fromClaudeSdkMessages();
35
+ stream.add({
36
+ type: "result",
37
+ usage: { input_tokens: 100, output_tokens: 5, cache_read_input_tokens: 900, cache_creation_input_tokens: 50 },
38
+ });
39
+ expect(stream.usage).toMatchObject({
40
+ inputTokens: 100,
41
+ outputTokens: 5,
42
+ cacheReadTokens: 900,
43
+ cacheCreationTokens: 50,
44
+ });
45
+ });
46
+ });
47
+
48
+ describe("fromPiAgentEvents usage 转发(pi 互斥口径)", () => {
49
+ it("input 原生不含 cacheRead/cacheWrite:如实转发,cost.total 累进实测 costUSD", () => {
50
+ const stream = fromPiAgentEvents();
51
+ stream.add({
52
+ type: "message_end",
53
+ message: {
54
+ role: "assistant",
55
+ content: [],
56
+ usage: { input: 100, output: 5, cacheRead: 900, cacheWrite: 50, cost: { total: 0.42 } },
57
+ },
58
+ });
59
+ expect(stream.usage).toMatchObject({
60
+ inputTokens: 100,
61
+ cacheReadTokens: 900,
62
+ cacheCreationTokens: 50,
63
+ costUSD: 0.42,
64
+ });
65
+ });
66
+ });
@@ -484,7 +484,9 @@ export function fromCodexThreadEvents(): CodexThreadStream {
484
484
  if (isRecord(u)) {
485
485
  const num = (v: unknown): number => (typeof v === "number" ? v : 0);
486
486
  usage = {
487
- inputTokens: (usage?.inputTokens ?? 0) + num(u.input_tokens),
487
+ // codex-rs TokenUsage 的 cached_input_tokens 是 input_tokens 的子集,
488
+ // 落互斥桶前扣掉(docs/feature/adapters/sdk/codex-sdk/cost.md)
489
+ inputTokens: (usage?.inputTokens ?? 0) + Math.max(0, num(u.input_tokens) - num(u.cached_input_tokens)),
488
490
  outputTokens: (usage?.outputTokens ?? 0) + num(u.output_tokens),
489
491
  cacheReadTokens: (usage?.cacheReadTokens ?? 0) + num(u.cached_input_tokens),
490
492
  // codex-rs TokenUsage 结构体同一份字段(与 o11y/parsers/codex.ts 的
@@ -369,10 +369,11 @@ export interface Agent {
369
369
  spanMapper?: SpanMapper;
370
370
  send(input: TurnInput, ctx: AgentContext): Promise<Turn>;
371
371
  /**
372
- * 可选 turn 失败分类器:按重试安全性归类一次 send 失败(抛出或返回 `status: "failed"` 的
373
- * Turn),返回 `undefined` 回落保守兜底。分类器只声明决策与诊断词,不影响重试策略(次数、
374
- * 退避对所有 agent 一致);抛错按不可重试处理并被吞掉。形状与分类链、执行体时序见
375
- * docs/feature/error-classification/architecture.md。
372
+ * 可选 turn 失败分类器:归类一次 send 失败(抛出或返回 `status: "failed"` 的 Turn),
373
+ * 返回 `undefined` 表示不认识、回落保守兜底。链上排在实验的 `classifyFailure` 之后,
374
+ * 实验作者认领过的失败问不到这里。分类器只声明决策轴与诊断词,不影响重试策略(次数、
375
+ * 退避对所有 agent 一致);抛错按 `undefined` 回落并被吞掉,不掩盖原始失败。形状与分类链、
376
+ * 执行体时序见 docs/feature/error-classification/architecture.md。
376
377
  */
377
378
  classifyTurnError?: TurnErrorClassifier;
378
379
  teardown?: AgentTeardown;