niceeval 0.10.3-canary.6 → 0.11.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/bin/niceeval.js +9 -4
- package/dist/agents/types.d.ts +5 -4
- package/dist/context/turn-errors.d.ts +27 -23
- package/dist/i18n/en.d.ts +15 -0
- package/dist/i18n/en.js +16 -1
- package/dist/i18n/zh-CN.d.ts +16 -1
- package/dist/i18n/zh-CN.js +16 -1
- package/dist/o11y/derive.js +28 -24
- package/dist/o11y/types.d.ts +8 -5
- package/dist/report/components/attempt-detail/UsageTable.js +4 -7
- package/dist/report/components/attempt-detail/compute.d.ts +2 -2
- package/dist/report/components/attempt-detail/compute.js +2 -6
- package/dist/report/components/attempt-detail/faces.js +7 -7
- package/dist/report/components/attempt-detail/index.js +0 -3
- package/dist/report/components/entity-lists/EvalList.js +0 -0
- package/dist/report/components/metric-views/compute.js +1 -1
- package/dist/report/model/types.d.ts +3 -4
- package/dist/results/locator.js +0 -0
- package/dist/results/select.d.ts +6 -0
- package/dist/results/select.js +8 -0
- package/dist/runner/feedback/sink.d.ts +26 -1
- package/dist/runner/fingerprint.d.ts +23 -0
- package/dist/runner/types.d.ts +105 -7
- package/dist/sandbox/errors.d.ts +29 -0
- package/dist/sandbox/resolve.d.ts +9 -0
- package/dist/shared/failure-class.d.ts +91 -0
- package/dist/types.d.ts +1 -0
- package/dist/util.d.ts +3 -2
- package/dist/util.js +31 -5
- package/docs-site/zh/explanation/runner.mdx +35 -0
- package/docs-site/zh/reference/cli.mdx +10 -1
- package/docs-site/zh/reference/events.mdx +4 -4
- package/docs-site/zh/troubleshooting/debugging.mdx +11 -0
- package/docs-site/zh/tutorials/viewing-results.mdx +35 -3
- package/package.json +28 -24
- package/src/agents/ai-sdk.test.ts +26 -0
- package/src/agents/ai-sdk.ts +7 -4
- package/src/agents/index.ts +5 -3
- package/src/agents/langgraph.test.ts +30 -0
- package/src/agents/langgraph.ts +5 -2
- package/src/agents/openai-compat.test.ts +35 -0
- package/src/agents/openai-compat.ts +16 -4
- package/src/agents/sdk-streams.test.ts +66 -0
- package/src/agents/sdk-streams.ts +3 -1
- package/src/agents/types.ts +5 -4
- package/src/cli.ts +81 -16
- package/src/context/context.test.ts +34 -0
- package/src/context/context.ts +14 -5
- package/src/context/send-retry.test.ts +86 -0
- package/src/context/send-retry.ts +37 -12
- package/src/context/session.ts +24 -0
- package/src/context/turn-errors.test.ts +124 -17
- package/src/context/turn-errors.ts +60 -50
- package/src/define.ts +5 -0
- package/src/i18n/en.ts +22 -1
- package/src/i18n/zh-CN.ts +22 -1
- package/src/index.ts +8 -0
- package/src/o11y/cost.ts +4 -2
- package/src/o11y/derive.test.ts +40 -0
- package/src/o11y/derive.ts +28 -22
- package/src/o11y/otlp/sandbox-receiver.test.ts +201 -0
- package/src/o11y/otlp/sandbox-receiver.ts +73 -27
- package/src/o11y/parsers/bub.test.ts +30 -0
- package/src/o11y/parsers/bub.ts +5 -2
- package/src/o11y/parsers/codex.test.ts +19 -0
- package/src/o11y/parsers/codex.ts +5 -2
- package/src/o11y/types.ts +8 -5
- package/src/report/components/attempt-detail/UsageTable.tsx +4 -6
- package/src/report/components/attempt-detail/attempt-components.test.tsx +5 -8
- package/src/report/components/attempt-detail/compute.ts +2 -7
- package/src/report/components/attempt-detail/faces.ts +7 -7
- package/src/report/components/attempt-detail/index.tsx +0 -3
- package/src/report/components/entity-lists/EvalList.tsx +0 -0
- package/src/report/components/metric-views/compute.ts +1 -1
- package/src/report/model/types.ts +3 -4
- package/src/results/format.ts +9 -2
- package/src/results/index.ts +2 -0
- package/src/results/locator.ts +0 -0
- package/src/results/open.ts +132 -21
- package/src/results/select.ts +12 -0
- package/src/results/skipped-notice.ts +0 -0
- package/src/runner/attempt.test.ts +116 -0
- package/src/runner/attempt.ts +89 -8
- package/src/runner/discover.ts +21 -3
- package/src/runner/feedback/coordinator.ts +24 -2
- package/src/runner/feedback/eval-conclusions.ts +6 -3
- package/src/runner/feedback/human.test.ts +300 -6
- package/src/runner/feedback/human.ts +132 -30
- package/src/runner/feedback/json.test.ts +127 -2
- package/src/runner/feedback/json.ts +51 -3
- package/src/runner/feedback/reducer.test.ts +328 -29
- package/src/runner/feedback/reducer.ts +84 -9
- package/src/runner/feedback/sink.ts +39 -1
- package/src/runner/fingerprint.ts +49 -19
- package/src/runner/gate-lease.test.ts +510 -0
- package/src/runner/gate-lease.ts +350 -0
- package/src/runner/lock.test.ts +454 -0
- package/src/runner/lock.ts +288 -0
- package/src/runner/report.test.ts +1 -0
- package/src/runner/run.test.ts +2044 -9
- package/src/runner/run.ts +825 -61
- package/src/runner/teardown-registry.ts +20 -78
- package/src/runner/types.ts +103 -7
- package/src/sandbox/errors.test.ts +72 -0
- package/src/sandbox/errors.ts +83 -0
- package/src/sandbox/keep-registry.ts +22 -46
- package/src/sandbox/resolve.test.ts +100 -0
- package/src/sandbox/resolve.ts +84 -27
- package/src/shared/entry-file-store.test.ts +149 -0
- package/src/shared/entry-file-store.ts +117 -0
- package/src/shared/failure-class.test.ts +137 -0
- package/src/shared/failure-class.ts +175 -0
- package/src/show/index.ts +5 -6
- package/src/show/render.test.ts +116 -15
- package/src/show/render.ts +124 -35
- package/src/types.ts +9 -0
- package/src/util.ts +31 -4
- package/src/view/app/App.tsx +5 -1
- package/src/view/client-dist/app.js +1 -1
- package/src/view/data.ts +5 -9
- package/src/view/shared/types.ts +7 -1
- package/src/view/view-report.test.ts +54 -0
|
@@ -93,6 +93,16 @@ $ niceeval show @1qrdcfq8 --execution | grep "TOOL ·" | sort | uniq -c
|
|
|
93
93
|
niceeval show @1qrdcfq8 --execution | grep proposals
|
|
94
94
|
```
|
|
95
95
|
|
|
96
|
+
超长的工具结果只显示前 3 行预览,但截断尾巴自带这张卡片的展开句柄——整行复制就是看全量的下一条命令,不用去翻 `events.json`:
|
|
97
|
+
|
|
98
|
+
```text
|
|
99
|
+
result · completed · exit 0
|
|
100
|
+
Proposal 1: …
|
|
101
|
+
(+412 lines · 18553 chars · niceeval show @1qrdcfq8 --execution --expand t1.c17)
|
|
102
|
+
|
|
103
|
+
$ niceeval show @1qrdcfq8 --execution --expand t1.c17 # 只输出这张卡片,完整不截断
|
|
104
|
+
```
|
|
105
|
+
|
|
96
106
|
**3. 看它到底改了什么。** `--diff` 只显示 **agent 自己改动的文件**——你上传的起始文件、跑完后写入的验证材料不会混在里面,所以列表里的每一行都真的是 agent 干的:
|
|
97
107
|
|
|
98
108
|
```text
|
|
@@ -225,6 +235,7 @@ niceeval view --results site-data/run
|
|
|
225
235
|
| 断言挂了,不知道为什么 | `show @loc` → `show @loc --source` |
|
|
226
236
|
| 想知道 agent 当时做了什么 | `show @loc --execution` |
|
|
227
237
|
| 想确认有没有调用过某个工具 / 搜对话关键词 | `show @loc --execution \| grep "TOOL ·" \| sort \| uniq -c` / `\| grep <关键词>` |
|
|
238
|
+
| 工具结果被截断,想看完整内容 | 复制截断尾巴里的命令:`show @loc --execution --expand t2.c3` |
|
|
228
239
|
| 想确认 agent 改了哪些文件 | `show @loc --diff` → `--diff=<path>` |
|
|
229
240
|
| 哪一步慢 / 超时死在哪 | `show @loc`(看 `timing:` 行)→ `show @loc --timing` |
|
|
230
241
|
| Sandbox 创建就失败(配额 / 凭据 / 镜像) | `show @loc` 看 error 的 code 与 cause → 查账号配额、核对凭据、降 `--max-concurrency`(没有现场可留) |
|
|
@@ -57,6 +57,7 @@ niceeval show weather/brooklyn # 收窄到一个 eval:同一份榜单,
|
|
|
57
57
|
niceeval show @1k2m9qrs # 精确到一次 Attempt:断言、执行、diff 与可用证据摘要
|
|
58
58
|
niceeval show @1k2m9qrs --source # 该 Attempt 运行时保存的 Eval 源码,断言标回源码行
|
|
59
59
|
niceeval show @1k2m9qrs --execution # 该 Attempt 的消息、thinking、Skill 加载、工具调用,有 OTel 时补时间
|
|
60
|
+
niceeval show @1k2m9qrs --execution --expand t2.c3 # 展开一张被截断卡片的完整落盘内容,句柄从截断尾巴里抄
|
|
60
61
|
niceeval show @1c3h6twx --timing # 有界诊断时间树:phase、hook、operation、shell、turn、OTel 与收尾
|
|
61
62
|
niceeval show @1c3h6twx --timing=full # 同一棵时间树逐节点完整展开
|
|
62
63
|
niceeval show @1c3h6twx --diff # sandbox 里的文件改动
|
|
@@ -251,6 +252,38 @@ full events: fixtures/button/a1/events.json
|
|
|
251
252
|
full OTel trace: fixtures/button/a1/trace.json
|
|
252
253
|
```
|
|
253
254
|
|
|
255
|
+
超长的工具结果既不整段打印、也不被一刀切丢掉。卡片正文是**有界预览**:每个内容段(消息正文;工具卡的 input 与 result 各算一段)最多显示前 3 行,保留原始换行,另有每段 1 KiB 兜底防单行超长。有折叠的卡片在卡尾报被折的行数与字符数,并自带这张卡片的**展开句柄**——整行就是下一条可直接复制的命令。预览负责回答「这一步做了什么、结果开头长什么样」,让整个 attempt 的树落在一两屏内;要看全量再花第二条命令:
|
|
256
|
+
|
|
257
|
+
```text
|
|
258
|
+
$ niceeval show @1c3h6twx --execution
|
|
259
|
+
…
|
|
260
|
+
TOOL · command_execution 5.2s · 1.9s
|
|
261
|
+
input
|
|
262
|
+
/bin/bash -lc 'rg --files node_modules/niceeval | sort'
|
|
263
|
+
result · completed · exit 0
|
|
264
|
+
node_modules/niceeval/AGENTS.md
|
|
265
|
+
node_modules/niceeval/INDEX.md
|
|
266
|
+
node_modules/niceeval/INIT.md
|
|
267
|
+
(+3055 lines · 263161 chars · niceeval show @1c3h6twx --execution --expand t2.c3)
|
|
268
|
+
```
|
|
269
|
+
|
|
270
|
+
`--expand` 接截断尾巴里的句柄,单独输出这一张卡片的完整落盘内容(原始换行,不再截断),其余卡片不打印;范围必须恰好命中一个 attempt。句柄 `t<轮次>.c<轮内卡序>`(失败 Sandbox 命令卡是 `cmd<n>`)由 `events.json` 的事件序确定性派生,同一 attempt 反复解析恒定——昨天终端记录里抄来的句柄今天照样有效:
|
|
271
|
+
|
|
272
|
+
```text
|
|
273
|
+
$ niceeval show @1c3h6twx --execution --expand t2.c3
|
|
274
|
+
@1c3h6twx · fixtures/button · compare/bub-gpt-5.4 · errored
|
|
275
|
+
|
|
276
|
+
TOOL · command_execution 5.2s · 1.9s
|
|
277
|
+
input
|
|
278
|
+
/bin/bash -lc 'rg --files node_modules/niceeval | sort'
|
|
279
|
+
result · completed · exit 0
|
|
280
|
+
node_modules/niceeval/AGENTS.md
|
|
281
|
+
node_modules/niceeval/INDEX.md
|
|
282
|
+
…完整落盘内容逐行继续,不再截断…
|
|
283
|
+
```
|
|
284
|
+
|
|
285
|
+
两点边界:展开还原的是**落盘证据**——单个值落盘时有 256 KiB 上限,超限的值展开后如实标注 `truncated` 与原始字节数,不冒充运行时全量;`--json` 面恒输出完整值、从不截断,因此它与 `--expand` 互斥——没有可展开的东西。
|
|
286
|
+
|
|
254
287
|
没有 OTel 接入时,同一棵树去掉时间列,节点内容原样保留:
|
|
255
288
|
|
|
256
289
|
```text
|
|
@@ -491,7 +524,7 @@ locator eval result turns tools un
|
|
|
491
524
|
total 6/8 passed 16 16 880.6k 3.5M* 54.8k 131 $4.05
|
|
492
525
|
```
|
|
493
526
|
|
|
494
|
-
`uncached in
|
|
527
|
+
`uncached in`(`inputTokens`,未命中缓存的输入)与 `cache read`(`cacheReadTokens`,缓存命中)是两个互斥的计价桶,相加才是每轮真正送进模型的上下文量:协议逐轮重发上下文时,膨胀多在 `cache read`,这层拆分才能回答「贵在哪」,只看单个数字看不出来。缺字段的格显示 `—`,不按 0 计入合计;合计列有缺失参与时标 `*`。
|
|
495
528
|
|
|
496
529
|
### agent 到底 search 了没:`--execution --grep`
|
|
497
530
|
|
|
@@ -537,8 +570,7 @@ niceeval show --exp compare/claude-mempal --usage --json
|
|
|
537
570
|
"locator": "@1xc8bnj5",
|
|
538
571
|
"evalId": "memory/agent-037-updatetag-cache",
|
|
539
572
|
"verdict": "passed",
|
|
540
|
-
"usage": { "inputTokens":
|
|
541
|
-
"uncachedInputTokens": 48200,
|
|
573
|
+
"usage": { "inputTokens": 48200, "outputTokens": 5100, "cacheReadTokens": 201500, "requests": 13 },
|
|
542
574
|
"estimatedCostUSD": 0.44
|
|
543
575
|
}
|
|
544
576
|
]
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "niceeval",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.11.0",
|
|
4
4
|
"description": "Agent-native eval tool — eval agents, services, functions, and coding-agent fixtures",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"license": "MIT",
|
|
@@ -18,51 +18,63 @@
|
|
|
18
18
|
"exports": {
|
|
19
19
|
".": {
|
|
20
20
|
"types": "./src/index.ts",
|
|
21
|
-
"import": "./src/index.ts"
|
|
21
|
+
"import": "./src/index.ts",
|
|
22
|
+
"require": "./src/index.ts"
|
|
22
23
|
},
|
|
23
24
|
"./sandbox": {
|
|
24
25
|
"types": "./src/sandbox/index.ts",
|
|
25
|
-
"import": "./src/sandbox/index.ts"
|
|
26
|
+
"import": "./src/sandbox/index.ts",
|
|
27
|
+
"require": "./src/sandbox/index.ts"
|
|
26
28
|
},
|
|
27
29
|
"./sandbox/e2b-template": {
|
|
28
30
|
"types": "./src/sandbox/e2b-agent-template.ts",
|
|
29
|
-
"import": "./src/sandbox/e2b-agent-template.ts"
|
|
31
|
+
"import": "./src/sandbox/e2b-agent-template.ts",
|
|
32
|
+
"require": "./src/sandbox/e2b-agent-template.ts"
|
|
30
33
|
},
|
|
31
34
|
"./adapter": {
|
|
32
35
|
"types": "./src/agents/index.ts",
|
|
33
|
-
"import": "./src/agents/index.ts"
|
|
36
|
+
"import": "./src/agents/index.ts",
|
|
37
|
+
"require": "./src/agents/index.ts"
|
|
34
38
|
},
|
|
35
39
|
"./adapter/otel": {
|
|
36
40
|
"types": "./src/agents/ai-sdk-otel.ts",
|
|
37
|
-
"import": "./src/agents/ai-sdk-otel.ts"
|
|
41
|
+
"import": "./src/agents/ai-sdk-otel.ts",
|
|
42
|
+
"require": "./src/agents/ai-sdk-otel.ts"
|
|
38
43
|
},
|
|
39
44
|
"./expect": {
|
|
40
45
|
"types": "./src/expect/index.ts",
|
|
41
|
-
"import": "./src/expect/index.ts"
|
|
46
|
+
"import": "./src/expect/index.ts",
|
|
47
|
+
"require": "./src/expect/index.ts"
|
|
42
48
|
},
|
|
43
49
|
"./reporters": {
|
|
44
50
|
"types": "./src/runner/reporters/index.ts",
|
|
45
|
-
"import": "./src/runner/reporters/index.ts"
|
|
51
|
+
"import": "./src/runner/reporters/index.ts",
|
|
52
|
+
"require": "./src/runner/reporters/index.ts"
|
|
46
53
|
},
|
|
47
54
|
"./loaders": {
|
|
48
55
|
"types": "./src/loaders/index.ts",
|
|
49
|
-
"import": "./src/loaders/index.ts"
|
|
56
|
+
"import": "./src/loaders/index.ts",
|
|
57
|
+
"require": "./src/loaders/index.ts"
|
|
50
58
|
},
|
|
51
59
|
"./results": {
|
|
52
60
|
"types": "./src/results/index.ts",
|
|
53
|
-
"import": "./src/results/index.ts"
|
|
61
|
+
"import": "./src/results/index.ts",
|
|
62
|
+
"require": "./src/results/index.ts"
|
|
54
63
|
},
|
|
55
64
|
"./report": {
|
|
56
65
|
"types": "./dist/report/index.d.ts",
|
|
57
|
-
"import": "./dist/report/index.js"
|
|
66
|
+
"import": "./dist/report/index.js",
|
|
67
|
+
"require": "./dist/report/index.js"
|
|
58
68
|
},
|
|
59
69
|
"./report/react": {
|
|
60
70
|
"types": "./dist/report/react/index.d.ts",
|
|
61
|
-
"import": "./dist/report/react/index.js"
|
|
71
|
+
"import": "./dist/report/react/index.js",
|
|
72
|
+
"require": "./dist/report/react/index.js"
|
|
62
73
|
},
|
|
63
74
|
"./report/built-in": {
|
|
64
75
|
"types": "./dist/report/built-in/index.d.ts",
|
|
65
|
-
"import": "./dist/report/built-in/index.js"
|
|
76
|
+
"import": "./dist/report/built-in/index.js",
|
|
77
|
+
"require": "./dist/report/built-in/index.js"
|
|
66
78
|
},
|
|
67
79
|
"./report/react/styles.css": "./src/report/assets/styles.css",
|
|
68
80
|
"./report/react/enhance.js": "./src/report/assets/enhance.js"
|
|
@@ -81,6 +93,8 @@
|
|
|
81
93
|
"autoevals": "0.0.132",
|
|
82
94
|
"effect": "^3.21.4",
|
|
83
95
|
"mermaid": "^11.16.0",
|
|
96
|
+
"react": "^19.0.0",
|
|
97
|
+
"react-dom": "^19.0.0",
|
|
84
98
|
"tar-stream": "^3.1.7",
|
|
85
99
|
"tsx": "^4.19.2"
|
|
86
100
|
},
|
|
@@ -109,8 +123,6 @@
|
|
|
109
123
|
"mixpanel-browser": "^2.80.0",
|
|
110
124
|
"next": "16.2.10",
|
|
111
125
|
"prism-react-renderer": "^2.4.1",
|
|
112
|
-
"react": "^19.2.7",
|
|
113
|
-
"react-dom": "^19.2.7",
|
|
114
126
|
"react-grab": "^0.1.48",
|
|
115
127
|
"shiki": "^4.3.0",
|
|
116
128
|
"tailwind-merge": "^3.6.0",
|
|
@@ -127,9 +139,7 @@
|
|
|
127
139
|
"ai": ">=5.0.0",
|
|
128
140
|
"braintrust": ">=0.0.150",
|
|
129
141
|
"dockerode": ">=4.0.0",
|
|
130
|
-
"e2b": ">=2.0.0"
|
|
131
|
-
"react": ">=18",
|
|
132
|
-
"react-dom": ">=18"
|
|
142
|
+
"e2b": ">=2.0.0"
|
|
133
143
|
},
|
|
134
144
|
"peerDependenciesMeta": {
|
|
135
145
|
"@ai-sdk/otel": {
|
|
@@ -155,12 +165,6 @@
|
|
|
155
165
|
},
|
|
156
166
|
"e2b": {
|
|
157
167
|
"optional": true
|
|
158
|
-
},
|
|
159
|
-
"react": {
|
|
160
|
-
"optional": true
|
|
161
|
-
},
|
|
162
|
-
"react-dom": {
|
|
163
|
-
"optional": true
|
|
164
168
|
}
|
|
165
169
|
},
|
|
166
170
|
"scripts": {
|
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
// cases: docs/engineering/testing/unit/results.md
|
|
2
|
+
// 「Usage、facts 与失败命令证据落盘」桶恒互斥归一:AI SDK 的 inputTokens 是含缓存明细的
|
|
3
|
+
// 输入总量,落桶前扣掉在场的 cacheRead / cacheWrite 明细。
|
|
4
|
+
// bug: memory/estimatecost-openai-inclusive-cache-double-billed.md
|
|
5
|
+
|
|
6
|
+
import { describe, expect, it } from "vitest";
|
|
7
|
+
|
|
8
|
+
import { fromAiSdk } from "./ai-sdk.ts";
|
|
9
|
+
|
|
10
|
+
describe("fromAiSdk usage 归一(含明细口径)", () => {
|
|
11
|
+
it("v5 形状:cachedInputTokens 从 inputTokens 里扣出", () => {
|
|
12
|
+
const turn = fromAiSdk({
|
|
13
|
+
text: "ok",
|
|
14
|
+
usage: { inputTokens: 1000, outputTokens: 20, cachedInputTokens: 900 },
|
|
15
|
+
});
|
|
16
|
+
expect(turn.usage).toMatchObject({ inputTokens: 100, cacheReadTokens: 900, outputTokens: 20 });
|
|
17
|
+
});
|
|
18
|
+
|
|
19
|
+
it("v7 形状:inputTokenDetails 的 cacheRead 与 cacheWrite 都从总量里扣出", () => {
|
|
20
|
+
const turn = fromAiSdk({
|
|
21
|
+
text: "ok",
|
|
22
|
+
usage: { inputTokens: 1000, outputTokens: 20, inputTokenDetails: { cacheReadTokens: 800, cacheWriteTokens: 100 } },
|
|
23
|
+
});
|
|
24
|
+
expect(turn.usage).toMatchObject({ inputTokens: 100, cacheReadTokens: 800, cacheCreationTokens: 100 });
|
|
25
|
+
});
|
|
26
|
+
});
|
package/src/agents/ai-sdk.ts
CHANGED
|
@@ -327,13 +327,16 @@ function unwrapToolOutput(output: unknown): { output?: JsonValue; status: "compl
|
|
|
327
327
|
function readUsage(result: AiSdkResultLike, stepCount: number): Usage | undefined {
|
|
328
328
|
const u = result.totalUsage ?? result.usage;
|
|
329
329
|
if (!u) return undefined;
|
|
330
|
-
const
|
|
330
|
+
const rawInput = num(u.inputTokens) ?? num(u.promptTokens) ?? 0;
|
|
331
331
|
const outputTokens = num(u.outputTokens) ?? num(u.completionTokens) ?? 0;
|
|
332
|
-
if (
|
|
332
|
+
if (rawInput === 0 && outputTokens === 0) return undefined;
|
|
333
|
+
const cacheRead = num(u.cachedInputTokens) ?? num(u.inputTokenDetails?.cacheReadTokens) ?? 0;
|
|
334
|
+
const cacheCreation = num(u.inputTokenDetails?.cacheWriteTokens) ?? 0;
|
|
335
|
+
// AI SDK 的 inputTokens 是含缓存明细的输入总量,落互斥桶前扣掉在场的明细
|
|
336
|
+
// (docs/feature/adapters/sdk/ai-sdk/cost.md)
|
|
337
|
+
const inputTokens = Math.max(0, rawInput - cacheRead - cacheCreation);
|
|
333
338
|
const usage: Usage = { inputTokens, outputTokens, requests: Math.max(stepCount, 1) };
|
|
334
|
-
const cacheRead = num(u.cachedInputTokens) ?? num(u.inputTokenDetails?.cacheReadTokens);
|
|
335
339
|
if (cacheRead) usage.cacheReadTokens = cacheRead;
|
|
336
|
-
const cacheCreation = num(u.inputTokenDetails?.cacheWriteTokens);
|
|
337
340
|
if (cacheCreation) usage.cacheCreationTokens = cacheCreation;
|
|
338
341
|
const reasoning = num(u.reasoningTokens);
|
|
339
342
|
if (reasoning) usage.reasoningTokens = reasoning;
|
package/src/agents/index.ts
CHANGED
|
@@ -9,11 +9,13 @@ export type { Shared } from "./shared.ts";
|
|
|
9
9
|
export { completeCoverage } from "../scoring/coverage.ts";
|
|
10
10
|
export type { CoverageStatus, CoverageDeclaration, EvidenceCoverage } from "../types.ts";
|
|
11
11
|
|
|
12
|
-
//
|
|
13
|
-
// (
|
|
12
|
+
// 执行失败分类:`Agent.classifyTurnError` 认的输入/输出形状 + 摘要取值器(与 turn-failed
|
|
13
|
+
// 报错文案同源)。两轴词表(FailureClass / FailureScope)与包根导出的是同一个形状——adapter
|
|
14
|
+
// 作者与 eval 作者各自的入口拿到同一份类型。判据、分类链与重试执行体见
|
|
14
15
|
// docs/feature/error-classification/architecture.md。
|
|
15
16
|
export { turnErrorText } from "../context/turn-errors.ts";
|
|
16
|
-
export type {
|
|
17
|
+
export type { TurnErrorClassifier, TurnFailure } from "../context/turn-errors.ts";
|
|
18
|
+
export type { FailureClass, FailureScope } from "../shared/failure-class.ts";
|
|
17
19
|
|
|
18
20
|
// span → canonical GenAI 归一(只服务瀑布图,不喂断言)。私有埋点写自己的 spanMapper 时用:
|
|
19
21
|
// tagSpan 把判定写回 span(原属性只增不改),heuristicTag 是通用兜底判定;mapCodexSpans 是
|
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
// cases: docs/engineering/testing/unit/results.md
|
|
2
|
+
// 「Usage、facts 与失败命令证据落盘」桶恒互斥归一:LangChain usage_metadata 的 input_tokens
|
|
3
|
+
// 是含缓存读写的输入总量,落桶前扣掉 input_token_details 的 cache_read / cache_creation。
|
|
4
|
+
// bug: memory/estimatecost-openai-inclusive-cache-double-billed.md
|
|
5
|
+
|
|
6
|
+
import { describe, expect, it } from "vitest";
|
|
7
|
+
|
|
8
|
+
import { fromLangGraphEvents } from "./langgraph.ts";
|
|
9
|
+
|
|
10
|
+
describe("fromLangGraphEvents usage 归一(含明细口径)", () => {
|
|
11
|
+
it("cache_read 与 cache_creation 都从 input_tokens 里扣出", () => {
|
|
12
|
+
const stream = fromLangGraphEvents();
|
|
13
|
+
stream.add({
|
|
14
|
+
channel: "messages",
|
|
15
|
+
event: "finish",
|
|
16
|
+
data: {
|
|
17
|
+
message: {
|
|
18
|
+
role: "assistant",
|
|
19
|
+
content: "ok",
|
|
20
|
+
usage_metadata: {
|
|
21
|
+
input_tokens: 1000,
|
|
22
|
+
output_tokens: 20,
|
|
23
|
+
input_token_details: { cache_read: 800, cache_creation: 100 },
|
|
24
|
+
},
|
|
25
|
+
},
|
|
26
|
+
},
|
|
27
|
+
});
|
|
28
|
+
expect(stream.usage).toMatchObject({ inputTokens: 100, cacheReadTokens: 800, cacheCreationTokens: 100, outputTokens: 20 });
|
|
29
|
+
});
|
|
30
|
+
});
|
package/src/agents/langgraph.ts
CHANGED
|
@@ -152,12 +152,15 @@ export function fromLangGraphEvents(): LangGraphStream {
|
|
|
152
152
|
}
|
|
153
153
|
return 0;
|
|
154
154
|
};
|
|
155
|
-
const
|
|
155
|
+
const rawInput = num("input_tokens", "inputTokens");
|
|
156
156
|
const output = num("output_tokens", "outputTokens");
|
|
157
|
-
if (
|
|
157
|
+
if (rawInput === 0 && output === 0) return;
|
|
158
158
|
const details = isRecord(raw.input_token_details) ? raw.input_token_details : undefined;
|
|
159
159
|
const cacheRead = typeof details?.cache_read === "number" ? details.cache_read : 0;
|
|
160
160
|
const cacheCreation = typeof details?.cache_creation === "number" ? details.cache_creation : 0;
|
|
161
|
+
// LangChain usage_metadata 的 input_tokens 是含缓存读写的输入总量,落互斥桶前扣掉明细
|
|
162
|
+
// (docs/feature/adapters/sdk/langgraph/cost.md)
|
|
163
|
+
const input = Math.max(0, rawInput - cacheRead - cacheCreation);
|
|
161
164
|
// LangChain UsageMetadata.output_token_details.reasoning:推理模型经 LangGraph 透传时带回。
|
|
162
165
|
const outputDetails = isRecord(raw.output_token_details) ? raw.output_token_details : undefined;
|
|
163
166
|
const reasoning = typeof outputDetails?.reasoning === "number" ? outputDetails.reasoning : 0;
|
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
// cases: docs/engineering/testing/unit/results.md
|
|
2
|
+
// 「Usage、facts 与失败命令证据落盘」桶恒互斥归一:Chat Completions / Responses 两种形状的
|
|
3
|
+
// cached_tokens 都是输入总量的子集,落 inputTokens 前扣掉;缺 cached 明细时总量原样保留。
|
|
4
|
+
// bug: memory/estimatecost-openai-inclusive-cache-double-billed.md
|
|
5
|
+
|
|
6
|
+
import { describe, expect, it } from "vitest";
|
|
7
|
+
|
|
8
|
+
import { fromChatCompletion, fromResponses } from "./openai-compat.ts";
|
|
9
|
+
|
|
10
|
+
describe("openai-compat usage 归一(OpenAI 口径)", () => {
|
|
11
|
+
it("Chat Completions:prompt_tokens 扣掉 prompt_tokens_details.cached_tokens", () => {
|
|
12
|
+
const turn = fromChatCompletion({
|
|
13
|
+
choices: [{ message: { role: "assistant", content: "ok" } }],
|
|
14
|
+
usage: { prompt_tokens: 1000, completion_tokens: 20, prompt_tokens_details: { cached_tokens: 900 } },
|
|
15
|
+
});
|
|
16
|
+
expect(turn.usage).toMatchObject({ inputTokens: 100, cacheReadTokens: 900, outputTokens: 20 });
|
|
17
|
+
});
|
|
18
|
+
|
|
19
|
+
it("Responses:input_tokens 扣掉 input_tokens_details.cached_tokens", () => {
|
|
20
|
+
const turn = fromResponses({
|
|
21
|
+
output: [{ type: "message", role: "assistant", content: [{ type: "output_text", text: "ok" }] }],
|
|
22
|
+
usage: { input_tokens: 500, output_tokens: 10, input_tokens_details: { cached_tokens: 200 } },
|
|
23
|
+
});
|
|
24
|
+
expect(turn.usage).toMatchObject({ inputTokens: 300, cacheReadTokens: 200, outputTokens: 10 });
|
|
25
|
+
});
|
|
26
|
+
|
|
27
|
+
it("缺 cached 明细时输入总量原样保留,不虚构扣减,cache 桶省略", () => {
|
|
28
|
+
const turn = fromChatCompletion({
|
|
29
|
+
choices: [{ message: { role: "assistant", content: "ok" } }],
|
|
30
|
+
usage: { prompt_tokens: 1000, completion_tokens: 20 },
|
|
31
|
+
});
|
|
32
|
+
expect(turn.usage?.inputTokens).toBe(1000);
|
|
33
|
+
expect(turn.usage?.cacheReadTokens).toBeUndefined();
|
|
34
|
+
});
|
|
35
|
+
});
|
|
@@ -50,8 +50,14 @@ export interface ChatCompletionLike {
|
|
|
50
50
|
|
|
51
51
|
function chatCompletionUsage(usage: ChatCompletionUsageLike | undefined): Usage | undefined {
|
|
52
52
|
if (!usage) return undefined;
|
|
53
|
-
|
|
54
|
-
|
|
53
|
+
// prompt_tokens 含缓存命中,cached_tokens 是其子集;落互斥桶前扣掉
|
|
54
|
+
// (docs/feature/adapters/sdk/openai-compat/cost.md)
|
|
55
|
+
const cached = usage.prompt_tokens_details?.cached_tokens ?? 0;
|
|
56
|
+
const u: Usage = {
|
|
57
|
+
inputTokens: Math.max(0, (usage.prompt_tokens ?? 0) - cached),
|
|
58
|
+
outputTokens: usage.completion_tokens ?? 0,
|
|
59
|
+
};
|
|
60
|
+
if (cached) u.cacheReadTokens = cached;
|
|
55
61
|
if (usage.completion_tokens_details?.reasoning_tokens) u.reasoningTokens = usage.completion_tokens_details.reasoning_tokens;
|
|
56
62
|
return u;
|
|
57
63
|
}
|
|
@@ -125,8 +131,14 @@ export interface ResponseLike {
|
|
|
125
131
|
|
|
126
132
|
function responsesUsage(usage: ResponseUsageLike | undefined): Usage | undefined {
|
|
127
133
|
if (!usage) return undefined;
|
|
128
|
-
|
|
129
|
-
|
|
134
|
+
// input_tokens 含缓存命中,cached_tokens 是其子集;落互斥桶前扣掉
|
|
135
|
+
// (docs/feature/adapters/sdk/openai-compat/cost.md)
|
|
136
|
+
const cached = usage.input_tokens_details?.cached_tokens ?? 0;
|
|
137
|
+
const u: Usage = {
|
|
138
|
+
inputTokens: Math.max(0, (usage.input_tokens ?? 0) - cached),
|
|
139
|
+
outputTokens: usage.output_tokens ?? 0,
|
|
140
|
+
};
|
|
141
|
+
if (cached) u.cacheReadTokens = cached;
|
|
130
142
|
if (usage.output_tokens_details?.reasoning_tokens) u.reasoningTokens = usage.output_tokens_details.reasoning_tokens;
|
|
131
143
|
return u;
|
|
132
144
|
}
|
|
@@ -0,0 +1,66 @@
|
|
|
1
|
+
// cases: docs/engineering/testing/unit/results.md
|
|
2
|
+
// 「Usage、facts 与失败命令证据落盘」桶恒互斥归一:codex(OpenAI 口径,cached ⊂ input)扣减、
|
|
3
|
+
// Anthropic / pi(互斥口径)如实转发。fixture 数值刻意让「扣与不扣」结果可区分。
|
|
4
|
+
// bug: memory/estimatecost-openai-inclusive-cache-double-billed.md
|
|
5
|
+
|
|
6
|
+
import { describe, expect, it } from "vitest";
|
|
7
|
+
|
|
8
|
+
import { fromClaudeSdkMessages, fromCodexThreadEvents, fromPiAgentEvents } from "./sdk-streams.ts";
|
|
9
|
+
|
|
10
|
+
describe("fromCodexThreadEvents usage 归一(OpenAI 口径)", () => {
|
|
11
|
+
it("cached_input_tokens 是 input_tokens 子集:落 inputTokens 前扣掉,cache 单独成桶", () => {
|
|
12
|
+
const stream = fromCodexThreadEvents();
|
|
13
|
+
stream.add({ type: "turn.completed", usage: { input_tokens: 1000, cached_input_tokens: 900, output_tokens: 50 } });
|
|
14
|
+
expect(stream.usage).toMatchObject({ inputTokens: 100, cacheReadTokens: 900, outputTokens: 50, requests: 1 });
|
|
15
|
+
});
|
|
16
|
+
|
|
17
|
+
it("逐轮累加在扣减之后进行,总量仍互斥", () => {
|
|
18
|
+
const stream = fromCodexThreadEvents();
|
|
19
|
+
stream.add({ type: "turn.completed", usage: { input_tokens: 1000, cached_input_tokens: 900, output_tokens: 50 } });
|
|
20
|
+
stream.add({ type: "turn.completed", usage: { input_tokens: 2000, cached_input_tokens: 1700, output_tokens: 30 } });
|
|
21
|
+
expect(stream.usage).toMatchObject({ inputTokens: 400, cacheReadTokens: 2600, outputTokens: 80, requests: 2 });
|
|
22
|
+
});
|
|
23
|
+
|
|
24
|
+
it("协议报出 cached > input 的病态数据时扣减夹底到 0,不产生负 token", () => {
|
|
25
|
+
const stream = fromCodexThreadEvents();
|
|
26
|
+
stream.add({ type: "turn.completed", usage: { input_tokens: 100, cached_input_tokens: 200, output_tokens: 1 } });
|
|
27
|
+
expect(stream.usage?.inputTokens).toBe(0);
|
|
28
|
+
expect(stream.usage?.cacheReadTokens).toBe(200);
|
|
29
|
+
});
|
|
30
|
+
});
|
|
31
|
+
|
|
32
|
+
describe("fromClaudeSdkMessages usage 转发(Anthropic 互斥口径)", () => {
|
|
33
|
+
it("input_tokens 原生不含 cache read:如实转发,不做扣减", () => {
|
|
34
|
+
const stream = fromClaudeSdkMessages();
|
|
35
|
+
stream.add({
|
|
36
|
+
type: "result",
|
|
37
|
+
usage: { input_tokens: 100, output_tokens: 5, cache_read_input_tokens: 900, cache_creation_input_tokens: 50 },
|
|
38
|
+
});
|
|
39
|
+
expect(stream.usage).toMatchObject({
|
|
40
|
+
inputTokens: 100,
|
|
41
|
+
outputTokens: 5,
|
|
42
|
+
cacheReadTokens: 900,
|
|
43
|
+
cacheCreationTokens: 50,
|
|
44
|
+
});
|
|
45
|
+
});
|
|
46
|
+
});
|
|
47
|
+
|
|
48
|
+
describe("fromPiAgentEvents usage 转发(pi 互斥口径)", () => {
|
|
49
|
+
it("input 原生不含 cacheRead/cacheWrite:如实转发,cost.total 累进实测 costUSD", () => {
|
|
50
|
+
const stream = fromPiAgentEvents();
|
|
51
|
+
stream.add({
|
|
52
|
+
type: "message_end",
|
|
53
|
+
message: {
|
|
54
|
+
role: "assistant",
|
|
55
|
+
content: [],
|
|
56
|
+
usage: { input: 100, output: 5, cacheRead: 900, cacheWrite: 50, cost: { total: 0.42 } },
|
|
57
|
+
},
|
|
58
|
+
});
|
|
59
|
+
expect(stream.usage).toMatchObject({
|
|
60
|
+
inputTokens: 100,
|
|
61
|
+
cacheReadTokens: 900,
|
|
62
|
+
cacheCreationTokens: 50,
|
|
63
|
+
costUSD: 0.42,
|
|
64
|
+
});
|
|
65
|
+
});
|
|
66
|
+
});
|
|
@@ -484,7 +484,9 @@ export function fromCodexThreadEvents(): CodexThreadStream {
|
|
|
484
484
|
if (isRecord(u)) {
|
|
485
485
|
const num = (v: unknown): number => (typeof v === "number" ? v : 0);
|
|
486
486
|
usage = {
|
|
487
|
-
|
|
487
|
+
// codex-rs TokenUsage 的 cached_input_tokens 是 input_tokens 的子集,
|
|
488
|
+
// 落互斥桶前扣掉(docs/feature/adapters/sdk/codex-sdk/cost.md)
|
|
489
|
+
inputTokens: (usage?.inputTokens ?? 0) + Math.max(0, num(u.input_tokens) - num(u.cached_input_tokens)),
|
|
488
490
|
outputTokens: (usage?.outputTokens ?? 0) + num(u.output_tokens),
|
|
489
491
|
cacheReadTokens: (usage?.cacheReadTokens ?? 0) + num(u.cached_input_tokens),
|
|
490
492
|
// codex-rs TokenUsage 结构体同一份字段(与 o11y/parsers/codex.ts 的
|
package/src/agents/types.ts
CHANGED
|
@@ -369,10 +369,11 @@ export interface Agent {
|
|
|
369
369
|
spanMapper?: SpanMapper;
|
|
370
370
|
send(input: TurnInput, ctx: AgentContext): Promise<Turn>;
|
|
371
371
|
/**
|
|
372
|
-
* 可选 turn
|
|
373
|
-
*
|
|
374
|
-
*
|
|
375
|
-
*
|
|
372
|
+
* 可选 turn 失败分类器:归类一次 send 失败(抛出或返回 `status: "failed"` 的 Turn),
|
|
373
|
+
* 返回 `undefined` 表示不认识、回落保守兜底。链上排在实验的 `classifyFailure` 之后,
|
|
374
|
+
* 实验作者认领过的失败问不到这里。分类器只声明决策轴与诊断词,不影响重试策略(次数、
|
|
375
|
+
* 退避对所有 agent 一致);抛错按 `undefined` 回落并被吞掉,不掩盖原始失败。形状与分类链、
|
|
376
|
+
* 执行体时序见 docs/feature/error-classification/architecture.md。
|
|
376
377
|
*/
|
|
377
378
|
classifyTurnError?: TurnErrorClassifier;
|
|
378
379
|
teardown?: AgentTeardown;
|