niceeval 0.10.3-canary.7 → 0.11.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/agents/types.d.ts +5 -4
- package/dist/context/turn-errors.d.ts +27 -23
- package/dist/i18n/en.d.ts +14 -0
- package/dist/i18n/en.js +15 -1
- package/dist/i18n/zh-CN.d.ts +15 -1
- package/dist/i18n/zh-CN.js +15 -1
- package/dist/o11y/derive.js +28 -24
- package/dist/o11y/types.d.ts +8 -5
- package/dist/report/components/attempt-detail/UsageTable.js +4 -7
- package/dist/report/components/attempt-detail/compute.d.ts +2 -2
- package/dist/report/components/attempt-detail/compute.js +2 -6
- package/dist/report/components/attempt-detail/faces.js +7 -7
- package/dist/report/components/attempt-detail/index.js +0 -3
- package/dist/report/components/entity-lists/EvalList.js +0 -0
- package/dist/report/components/metric-views/compute.js +1 -1
- package/dist/report/model/types.d.ts +3 -4
- package/dist/results/locator.js +0 -0
- package/dist/results/select.d.ts +6 -0
- package/dist/results/select.js +8 -0
- package/dist/runner/feedback/sink.d.ts +26 -1
- package/dist/runner/fingerprint.d.ts +23 -0
- package/dist/runner/types.d.ts +105 -7
- package/dist/sandbox/errors.d.ts +29 -0
- package/dist/sandbox/resolve.d.ts +9 -0
- package/dist/shared/failure-class.d.ts +91 -0
- package/dist/types.d.ts +1 -0
- package/dist/util.d.ts +3 -2
- package/dist/util.js +31 -5
- package/docs-site/zh/explanation/runner.mdx +35 -0
- package/docs-site/zh/reference/cli.mdx +3 -3
- package/docs-site/zh/reference/events.mdx +4 -4
- package/docs-site/zh/troubleshooting/debugging.mdx +2 -2
- package/docs-site/zh/tutorials/viewing-results.mdx +4 -5
- package/package.json +4 -12
- package/src/agents/ai-sdk.test.ts +26 -0
- package/src/agents/ai-sdk.ts +7 -4
- package/src/agents/index.ts +5 -3
- package/src/agents/langgraph.test.ts +30 -0
- package/src/agents/langgraph.ts +5 -2
- package/src/agents/openai-compat.test.ts +35 -0
- package/src/agents/openai-compat.ts +16 -4
- package/src/agents/sdk-streams.test.ts +66 -0
- package/src/agents/sdk-streams.ts +3 -1
- package/src/agents/types.ts +5 -4
- package/src/cli.ts +55 -14
- package/src/context/context.test.ts +34 -0
- package/src/context/context.ts +14 -5
- package/src/context/send-retry.test.ts +86 -0
- package/src/context/send-retry.ts +37 -12
- package/src/context/session.ts +24 -0
- package/src/context/turn-errors.test.ts +124 -17
- package/src/context/turn-errors.ts +60 -50
- package/src/define.ts +5 -0
- package/src/i18n/en.ts +20 -1
- package/src/i18n/zh-CN.ts +20 -1
- package/src/index.ts +8 -0
- package/src/o11y/cost.ts +4 -2
- package/src/o11y/derive.test.ts +40 -0
- package/src/o11y/derive.ts +28 -22
- package/src/o11y/otlp/sandbox-receiver.test.ts +201 -0
- package/src/o11y/otlp/sandbox-receiver.ts +73 -27
- package/src/o11y/parsers/bub.test.ts +30 -0
- package/src/o11y/parsers/bub.ts +5 -2
- package/src/o11y/parsers/codex.test.ts +19 -0
- package/src/o11y/parsers/codex.ts +5 -2
- package/src/o11y/types.ts +8 -5
- package/src/report/components/attempt-detail/UsageTable.tsx +4 -6
- package/src/report/components/attempt-detail/attempt-components.test.tsx +5 -8
- package/src/report/components/attempt-detail/compute.ts +2 -7
- package/src/report/components/attempt-detail/faces.ts +7 -7
- package/src/report/components/attempt-detail/index.tsx +0 -3
- package/src/report/components/entity-lists/EvalList.tsx +0 -0
- package/src/report/components/metric-views/compute.ts +1 -1
- package/src/report/model/types.ts +3 -4
- package/src/results/format.ts +9 -2
- package/src/results/index.ts +2 -0
- package/src/results/locator.ts +0 -0
- package/src/results/open.ts +132 -21
- package/src/results/select.ts +12 -0
- package/src/results/skipped-notice.ts +0 -0
- package/src/runner/attempt.test.ts +116 -0
- package/src/runner/attempt.ts +89 -8
- package/src/runner/discover.ts +21 -3
- package/src/runner/feedback/coordinator.ts +24 -2
- package/src/runner/feedback/eval-conclusions.ts +6 -3
- package/src/runner/feedback/human.test.ts +300 -6
- package/src/runner/feedback/human.ts +132 -30
- package/src/runner/feedback/json.test.ts +127 -2
- package/src/runner/feedback/json.ts +51 -3
- package/src/runner/feedback/reducer.test.ts +328 -29
- package/src/runner/feedback/reducer.ts +84 -9
- package/src/runner/feedback/sink.ts +39 -1
- package/src/runner/fingerprint.ts +49 -19
- package/src/runner/gate-lease.test.ts +510 -0
- package/src/runner/gate-lease.ts +350 -0
- package/src/runner/lock.test.ts +454 -0
- package/src/runner/lock.ts +288 -0
- package/src/runner/report.test.ts +1 -0
- package/src/runner/run.test.ts +2044 -9
- package/src/runner/run.ts +825 -61
- package/src/runner/teardown-registry.ts +20 -78
- package/src/runner/types.ts +103 -7
- package/src/sandbox/errors.test.ts +72 -0
- package/src/sandbox/errors.ts +83 -0
- package/src/sandbox/keep-registry.ts +22 -46
- package/src/sandbox/resolve.test.ts +100 -0
- package/src/sandbox/resolve.ts +84 -27
- package/src/shared/entry-file-store.test.ts +149 -0
- package/src/shared/entry-file-store.ts +117 -0
- package/src/shared/failure-class.test.ts +137 -0
- package/src/shared/failure-class.ts +175 -0
- package/src/show/index.ts +5 -6
- package/src/show/render.test.ts +116 -15
- package/src/show/render.ts +124 -35
- package/src/types.ts +9 -0
- package/src/util.ts +31 -4
- package/src/view/app/App.tsx +5 -1
- package/src/view/client-dist/app.js +1 -1
- package/src/view/data.ts +5 -9
- package/src/view/shared/types.ts +7 -1
- package/src/view/view-report.test.ts +54 -0
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "niceeval",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.11.0",
|
|
4
4
|
"description": "Agent-native eval tool — eval agents, services, functions, and coding-agent fixtures",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"license": "MIT",
|
|
@@ -93,6 +93,8 @@
|
|
|
93
93
|
"autoevals": "0.0.132",
|
|
94
94
|
"effect": "^3.21.4",
|
|
95
95
|
"mermaid": "^11.16.0",
|
|
96
|
+
"react": "^19.0.0",
|
|
97
|
+
"react-dom": "^19.0.0",
|
|
96
98
|
"tar-stream": "^3.1.7",
|
|
97
99
|
"tsx": "^4.19.2"
|
|
98
100
|
},
|
|
@@ -121,8 +123,6 @@
|
|
|
121
123
|
"mixpanel-browser": "^2.80.0",
|
|
122
124
|
"next": "16.2.10",
|
|
123
125
|
"prism-react-renderer": "^2.4.1",
|
|
124
|
-
"react": "^19.2.7",
|
|
125
|
-
"react-dom": "^19.2.7",
|
|
126
126
|
"react-grab": "^0.1.48",
|
|
127
127
|
"shiki": "^4.3.0",
|
|
128
128
|
"tailwind-merge": "^3.6.0",
|
|
@@ -139,9 +139,7 @@
|
|
|
139
139
|
"ai": ">=5.0.0",
|
|
140
140
|
"braintrust": ">=0.0.150",
|
|
141
141
|
"dockerode": ">=4.0.0",
|
|
142
|
-
"e2b": ">=2.0.0"
|
|
143
|
-
"react": ">=18",
|
|
144
|
-
"react-dom": ">=18"
|
|
142
|
+
"e2b": ">=2.0.0"
|
|
145
143
|
},
|
|
146
144
|
"peerDependenciesMeta": {
|
|
147
145
|
"@ai-sdk/otel": {
|
|
@@ -167,12 +165,6 @@
|
|
|
167
165
|
},
|
|
168
166
|
"e2b": {
|
|
169
167
|
"optional": true
|
|
170
|
-
},
|
|
171
|
-
"react": {
|
|
172
|
-
"optional": true
|
|
173
|
-
},
|
|
174
|
-
"react-dom": {
|
|
175
|
-
"optional": true
|
|
176
168
|
}
|
|
177
169
|
},
|
|
178
170
|
"scripts": {
|
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
// cases: docs/engineering/testing/unit/results.md
|
|
2
|
+
// 「Usage、facts 与失败命令证据落盘」桶恒互斥归一:AI SDK 的 inputTokens 是含缓存明细的
|
|
3
|
+
// 输入总量,落桶前扣掉在场的 cacheRead / cacheWrite 明细。
|
|
4
|
+
// bug: memory/estimatecost-openai-inclusive-cache-double-billed.md
|
|
5
|
+
|
|
6
|
+
import { describe, expect, it } from "vitest";
|
|
7
|
+
|
|
8
|
+
import { fromAiSdk } from "./ai-sdk.ts";
|
|
9
|
+
|
|
10
|
+
describe("fromAiSdk usage 归一(含明细口径)", () => {
|
|
11
|
+
it("v5 形状:cachedInputTokens 从 inputTokens 里扣出", () => {
|
|
12
|
+
const turn = fromAiSdk({
|
|
13
|
+
text: "ok",
|
|
14
|
+
usage: { inputTokens: 1000, outputTokens: 20, cachedInputTokens: 900 },
|
|
15
|
+
});
|
|
16
|
+
expect(turn.usage).toMatchObject({ inputTokens: 100, cacheReadTokens: 900, outputTokens: 20 });
|
|
17
|
+
});
|
|
18
|
+
|
|
19
|
+
it("v7 形状:inputTokenDetails 的 cacheRead 与 cacheWrite 都从总量里扣出", () => {
|
|
20
|
+
const turn = fromAiSdk({
|
|
21
|
+
text: "ok",
|
|
22
|
+
usage: { inputTokens: 1000, outputTokens: 20, inputTokenDetails: { cacheReadTokens: 800, cacheWriteTokens: 100 } },
|
|
23
|
+
});
|
|
24
|
+
expect(turn.usage).toMatchObject({ inputTokens: 100, cacheReadTokens: 800, cacheCreationTokens: 100 });
|
|
25
|
+
});
|
|
26
|
+
});
|
package/src/agents/ai-sdk.ts
CHANGED
|
@@ -327,13 +327,16 @@ function unwrapToolOutput(output: unknown): { output?: JsonValue; status: "compl
|
|
|
327
327
|
function readUsage(result: AiSdkResultLike, stepCount: number): Usage | undefined {
|
|
328
328
|
const u = result.totalUsage ?? result.usage;
|
|
329
329
|
if (!u) return undefined;
|
|
330
|
-
const
|
|
330
|
+
const rawInput = num(u.inputTokens) ?? num(u.promptTokens) ?? 0;
|
|
331
331
|
const outputTokens = num(u.outputTokens) ?? num(u.completionTokens) ?? 0;
|
|
332
|
-
if (
|
|
332
|
+
if (rawInput === 0 && outputTokens === 0) return undefined;
|
|
333
|
+
const cacheRead = num(u.cachedInputTokens) ?? num(u.inputTokenDetails?.cacheReadTokens) ?? 0;
|
|
334
|
+
const cacheCreation = num(u.inputTokenDetails?.cacheWriteTokens) ?? 0;
|
|
335
|
+
// AI SDK 的 inputTokens 是含缓存明细的输入总量,落互斥桶前扣掉在场的明细
|
|
336
|
+
// (docs/feature/adapters/sdk/ai-sdk/cost.md)
|
|
337
|
+
const inputTokens = Math.max(0, rawInput - cacheRead - cacheCreation);
|
|
333
338
|
const usage: Usage = { inputTokens, outputTokens, requests: Math.max(stepCount, 1) };
|
|
334
|
-
const cacheRead = num(u.cachedInputTokens) ?? num(u.inputTokenDetails?.cacheReadTokens);
|
|
335
339
|
if (cacheRead) usage.cacheReadTokens = cacheRead;
|
|
336
|
-
const cacheCreation = num(u.inputTokenDetails?.cacheWriteTokens);
|
|
337
340
|
if (cacheCreation) usage.cacheCreationTokens = cacheCreation;
|
|
338
341
|
const reasoning = num(u.reasoningTokens);
|
|
339
342
|
if (reasoning) usage.reasoningTokens = reasoning;
|
package/src/agents/index.ts
CHANGED
|
@@ -9,11 +9,13 @@ export type { Shared } from "./shared.ts";
|
|
|
9
9
|
export { completeCoverage } from "../scoring/coverage.ts";
|
|
10
10
|
export type { CoverageStatus, CoverageDeclaration, EvidenceCoverage } from "../types.ts";
|
|
11
11
|
|
|
12
|
-
//
|
|
13
|
-
// (
|
|
12
|
+
// 执行失败分类:`Agent.classifyTurnError` 认的输入/输出形状 + 摘要取值器(与 turn-failed
|
|
13
|
+
// 报错文案同源)。两轴词表(FailureClass / FailureScope)与包根导出的是同一个形状——adapter
|
|
14
|
+
// 作者与 eval 作者各自的入口拿到同一份类型。判据、分类链与重试执行体见
|
|
14
15
|
// docs/feature/error-classification/architecture.md。
|
|
15
16
|
export { turnErrorText } from "../context/turn-errors.ts";
|
|
16
|
-
export type {
|
|
17
|
+
export type { TurnErrorClassifier, TurnFailure } from "../context/turn-errors.ts";
|
|
18
|
+
export type { FailureClass, FailureScope } from "../shared/failure-class.ts";
|
|
17
19
|
|
|
18
20
|
// span → canonical GenAI 归一(只服务瀑布图,不喂断言)。私有埋点写自己的 spanMapper 时用:
|
|
19
21
|
// tagSpan 把判定写回 span(原属性只增不改),heuristicTag 是通用兜底判定;mapCodexSpans 是
|
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
// cases: docs/engineering/testing/unit/results.md
|
|
2
|
+
// 「Usage、facts 与失败命令证据落盘」桶恒互斥归一:LangChain usage_metadata 的 input_tokens
|
|
3
|
+
// 是含缓存读写的输入总量,落桶前扣掉 input_token_details 的 cache_read / cache_creation。
|
|
4
|
+
// bug: memory/estimatecost-openai-inclusive-cache-double-billed.md
|
|
5
|
+
|
|
6
|
+
import { describe, expect, it } from "vitest";
|
|
7
|
+
|
|
8
|
+
import { fromLangGraphEvents } from "./langgraph.ts";
|
|
9
|
+
|
|
10
|
+
describe("fromLangGraphEvents usage 归一(含明细口径)", () => {
|
|
11
|
+
it("cache_read 与 cache_creation 都从 input_tokens 里扣出", () => {
|
|
12
|
+
const stream = fromLangGraphEvents();
|
|
13
|
+
stream.add({
|
|
14
|
+
channel: "messages",
|
|
15
|
+
event: "finish",
|
|
16
|
+
data: {
|
|
17
|
+
message: {
|
|
18
|
+
role: "assistant",
|
|
19
|
+
content: "ok",
|
|
20
|
+
usage_metadata: {
|
|
21
|
+
input_tokens: 1000,
|
|
22
|
+
output_tokens: 20,
|
|
23
|
+
input_token_details: { cache_read: 800, cache_creation: 100 },
|
|
24
|
+
},
|
|
25
|
+
},
|
|
26
|
+
},
|
|
27
|
+
});
|
|
28
|
+
expect(stream.usage).toMatchObject({ inputTokens: 100, cacheReadTokens: 800, cacheCreationTokens: 100, outputTokens: 20 });
|
|
29
|
+
});
|
|
30
|
+
});
|
package/src/agents/langgraph.ts
CHANGED
|
@@ -152,12 +152,15 @@ export function fromLangGraphEvents(): LangGraphStream {
|
|
|
152
152
|
}
|
|
153
153
|
return 0;
|
|
154
154
|
};
|
|
155
|
-
const
|
|
155
|
+
const rawInput = num("input_tokens", "inputTokens");
|
|
156
156
|
const output = num("output_tokens", "outputTokens");
|
|
157
|
-
if (
|
|
157
|
+
if (rawInput === 0 && output === 0) return;
|
|
158
158
|
const details = isRecord(raw.input_token_details) ? raw.input_token_details : undefined;
|
|
159
159
|
const cacheRead = typeof details?.cache_read === "number" ? details.cache_read : 0;
|
|
160
160
|
const cacheCreation = typeof details?.cache_creation === "number" ? details.cache_creation : 0;
|
|
161
|
+
// LangChain usage_metadata 的 input_tokens 是含缓存读写的输入总量,落互斥桶前扣掉明细
|
|
162
|
+
// (docs/feature/adapters/sdk/langgraph/cost.md)
|
|
163
|
+
const input = Math.max(0, rawInput - cacheRead - cacheCreation);
|
|
161
164
|
// LangChain UsageMetadata.output_token_details.reasoning:推理模型经 LangGraph 透传时带回。
|
|
162
165
|
const outputDetails = isRecord(raw.output_token_details) ? raw.output_token_details : undefined;
|
|
163
166
|
const reasoning = typeof outputDetails?.reasoning === "number" ? outputDetails.reasoning : 0;
|
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
// cases: docs/engineering/testing/unit/results.md
|
|
2
|
+
// 「Usage、facts 与失败命令证据落盘」桶恒互斥归一:Chat Completions / Responses 两种形状的
|
|
3
|
+
// cached_tokens 都是输入总量的子集,落 inputTokens 前扣掉;缺 cached 明细时总量原样保留。
|
|
4
|
+
// bug: memory/estimatecost-openai-inclusive-cache-double-billed.md
|
|
5
|
+
|
|
6
|
+
import { describe, expect, it } from "vitest";
|
|
7
|
+
|
|
8
|
+
import { fromChatCompletion, fromResponses } from "./openai-compat.ts";
|
|
9
|
+
|
|
10
|
+
describe("openai-compat usage 归一(OpenAI 口径)", () => {
|
|
11
|
+
it("Chat Completions:prompt_tokens 扣掉 prompt_tokens_details.cached_tokens", () => {
|
|
12
|
+
const turn = fromChatCompletion({
|
|
13
|
+
choices: [{ message: { role: "assistant", content: "ok" } }],
|
|
14
|
+
usage: { prompt_tokens: 1000, completion_tokens: 20, prompt_tokens_details: { cached_tokens: 900 } },
|
|
15
|
+
});
|
|
16
|
+
expect(turn.usage).toMatchObject({ inputTokens: 100, cacheReadTokens: 900, outputTokens: 20 });
|
|
17
|
+
});
|
|
18
|
+
|
|
19
|
+
it("Responses:input_tokens 扣掉 input_tokens_details.cached_tokens", () => {
|
|
20
|
+
const turn = fromResponses({
|
|
21
|
+
output: [{ type: "message", role: "assistant", content: [{ type: "output_text", text: "ok" }] }],
|
|
22
|
+
usage: { input_tokens: 500, output_tokens: 10, input_tokens_details: { cached_tokens: 200 } },
|
|
23
|
+
});
|
|
24
|
+
expect(turn.usage).toMatchObject({ inputTokens: 300, cacheReadTokens: 200, outputTokens: 10 });
|
|
25
|
+
});
|
|
26
|
+
|
|
27
|
+
it("缺 cached 明细时输入总量原样保留,不虚构扣减,cache 桶省略", () => {
|
|
28
|
+
const turn = fromChatCompletion({
|
|
29
|
+
choices: [{ message: { role: "assistant", content: "ok" } }],
|
|
30
|
+
usage: { prompt_tokens: 1000, completion_tokens: 20 },
|
|
31
|
+
});
|
|
32
|
+
expect(turn.usage?.inputTokens).toBe(1000);
|
|
33
|
+
expect(turn.usage?.cacheReadTokens).toBeUndefined();
|
|
34
|
+
});
|
|
35
|
+
});
|
|
@@ -50,8 +50,14 @@ export interface ChatCompletionLike {
|
|
|
50
50
|
|
|
51
51
|
function chatCompletionUsage(usage: ChatCompletionUsageLike | undefined): Usage | undefined {
|
|
52
52
|
if (!usage) return undefined;
|
|
53
|
-
|
|
54
|
-
|
|
53
|
+
// prompt_tokens 含缓存命中,cached_tokens 是其子集;落互斥桶前扣掉
|
|
54
|
+
// (docs/feature/adapters/sdk/openai-compat/cost.md)
|
|
55
|
+
const cached = usage.prompt_tokens_details?.cached_tokens ?? 0;
|
|
56
|
+
const u: Usage = {
|
|
57
|
+
inputTokens: Math.max(0, (usage.prompt_tokens ?? 0) - cached),
|
|
58
|
+
outputTokens: usage.completion_tokens ?? 0,
|
|
59
|
+
};
|
|
60
|
+
if (cached) u.cacheReadTokens = cached;
|
|
55
61
|
if (usage.completion_tokens_details?.reasoning_tokens) u.reasoningTokens = usage.completion_tokens_details.reasoning_tokens;
|
|
56
62
|
return u;
|
|
57
63
|
}
|
|
@@ -125,8 +131,14 @@ export interface ResponseLike {
|
|
|
125
131
|
|
|
126
132
|
function responsesUsage(usage: ResponseUsageLike | undefined): Usage | undefined {
|
|
127
133
|
if (!usage) return undefined;
|
|
128
|
-
|
|
129
|
-
|
|
134
|
+
// input_tokens 含缓存命中,cached_tokens 是其子集;落互斥桶前扣掉
|
|
135
|
+
// (docs/feature/adapters/sdk/openai-compat/cost.md)
|
|
136
|
+
const cached = usage.input_tokens_details?.cached_tokens ?? 0;
|
|
137
|
+
const u: Usage = {
|
|
138
|
+
inputTokens: Math.max(0, (usage.input_tokens ?? 0) - cached),
|
|
139
|
+
outputTokens: usage.output_tokens ?? 0,
|
|
140
|
+
};
|
|
141
|
+
if (cached) u.cacheReadTokens = cached;
|
|
130
142
|
if (usage.output_tokens_details?.reasoning_tokens) u.reasoningTokens = usage.output_tokens_details.reasoning_tokens;
|
|
131
143
|
return u;
|
|
132
144
|
}
|
|
@@ -0,0 +1,66 @@
|
|
|
1
|
+
// cases: docs/engineering/testing/unit/results.md
|
|
2
|
+
// 「Usage、facts 与失败命令证据落盘」桶恒互斥归一:codex(OpenAI 口径,cached ⊂ input)扣减、
|
|
3
|
+
// Anthropic / pi(互斥口径)如实转发。fixture 数值刻意让「扣与不扣」结果可区分。
|
|
4
|
+
// bug: memory/estimatecost-openai-inclusive-cache-double-billed.md
|
|
5
|
+
|
|
6
|
+
import { describe, expect, it } from "vitest";
|
|
7
|
+
|
|
8
|
+
import { fromClaudeSdkMessages, fromCodexThreadEvents, fromPiAgentEvents } from "./sdk-streams.ts";
|
|
9
|
+
|
|
10
|
+
describe("fromCodexThreadEvents usage 归一(OpenAI 口径)", () => {
|
|
11
|
+
it("cached_input_tokens 是 input_tokens 子集:落 inputTokens 前扣掉,cache 单独成桶", () => {
|
|
12
|
+
const stream = fromCodexThreadEvents();
|
|
13
|
+
stream.add({ type: "turn.completed", usage: { input_tokens: 1000, cached_input_tokens: 900, output_tokens: 50 } });
|
|
14
|
+
expect(stream.usage).toMatchObject({ inputTokens: 100, cacheReadTokens: 900, outputTokens: 50, requests: 1 });
|
|
15
|
+
});
|
|
16
|
+
|
|
17
|
+
it("逐轮累加在扣减之后进行,总量仍互斥", () => {
|
|
18
|
+
const stream = fromCodexThreadEvents();
|
|
19
|
+
stream.add({ type: "turn.completed", usage: { input_tokens: 1000, cached_input_tokens: 900, output_tokens: 50 } });
|
|
20
|
+
stream.add({ type: "turn.completed", usage: { input_tokens: 2000, cached_input_tokens: 1700, output_tokens: 30 } });
|
|
21
|
+
expect(stream.usage).toMatchObject({ inputTokens: 400, cacheReadTokens: 2600, outputTokens: 80, requests: 2 });
|
|
22
|
+
});
|
|
23
|
+
|
|
24
|
+
it("协议报出 cached > input 的病态数据时扣减夹底到 0,不产生负 token", () => {
|
|
25
|
+
const stream = fromCodexThreadEvents();
|
|
26
|
+
stream.add({ type: "turn.completed", usage: { input_tokens: 100, cached_input_tokens: 200, output_tokens: 1 } });
|
|
27
|
+
expect(stream.usage?.inputTokens).toBe(0);
|
|
28
|
+
expect(stream.usage?.cacheReadTokens).toBe(200);
|
|
29
|
+
});
|
|
30
|
+
});
|
|
31
|
+
|
|
32
|
+
describe("fromClaudeSdkMessages usage 转发(Anthropic 互斥口径)", () => {
|
|
33
|
+
it("input_tokens 原生不含 cache read:如实转发,不做扣减", () => {
|
|
34
|
+
const stream = fromClaudeSdkMessages();
|
|
35
|
+
stream.add({
|
|
36
|
+
type: "result",
|
|
37
|
+
usage: { input_tokens: 100, output_tokens: 5, cache_read_input_tokens: 900, cache_creation_input_tokens: 50 },
|
|
38
|
+
});
|
|
39
|
+
expect(stream.usage).toMatchObject({
|
|
40
|
+
inputTokens: 100,
|
|
41
|
+
outputTokens: 5,
|
|
42
|
+
cacheReadTokens: 900,
|
|
43
|
+
cacheCreationTokens: 50,
|
|
44
|
+
});
|
|
45
|
+
});
|
|
46
|
+
});
|
|
47
|
+
|
|
48
|
+
describe("fromPiAgentEvents usage 转发(pi 互斥口径)", () => {
|
|
49
|
+
it("input 原生不含 cacheRead/cacheWrite:如实转发,cost.total 累进实测 costUSD", () => {
|
|
50
|
+
const stream = fromPiAgentEvents();
|
|
51
|
+
stream.add({
|
|
52
|
+
type: "message_end",
|
|
53
|
+
message: {
|
|
54
|
+
role: "assistant",
|
|
55
|
+
content: [],
|
|
56
|
+
usage: { input: 100, output: 5, cacheRead: 900, cacheWrite: 50, cost: { total: 0.42 } },
|
|
57
|
+
},
|
|
58
|
+
});
|
|
59
|
+
expect(stream.usage).toMatchObject({
|
|
60
|
+
inputTokens: 100,
|
|
61
|
+
cacheReadTokens: 900,
|
|
62
|
+
cacheCreationTokens: 50,
|
|
63
|
+
costUSD: 0.42,
|
|
64
|
+
});
|
|
65
|
+
});
|
|
66
|
+
});
|
|
@@ -484,7 +484,9 @@ export function fromCodexThreadEvents(): CodexThreadStream {
|
|
|
484
484
|
if (isRecord(u)) {
|
|
485
485
|
const num = (v: unknown): number => (typeof v === "number" ? v : 0);
|
|
486
486
|
usage = {
|
|
487
|
-
|
|
487
|
+
// codex-rs TokenUsage 的 cached_input_tokens 是 input_tokens 的子集,
|
|
488
|
+
// 落互斥桶前扣掉(docs/feature/adapters/sdk/codex-sdk/cost.md)
|
|
489
|
+
inputTokens: (usage?.inputTokens ?? 0) + Math.max(0, num(u.input_tokens) - num(u.cached_input_tokens)),
|
|
488
490
|
outputTokens: (usage?.outputTokens ?? 0) + num(u.output_tokens),
|
|
489
491
|
cacheReadTokens: (usage?.cacheReadTokens ?? 0) + num(u.cached_input_tokens),
|
|
490
492
|
// codex-rs TokenUsage 结构体同一份字段(与 o11y/parsers/codex.ts 的
|
package/src/agents/types.ts
CHANGED
|
@@ -369,10 +369,11 @@ export interface Agent {
|
|
|
369
369
|
spanMapper?: SpanMapper;
|
|
370
370
|
send(input: TurnInput, ctx: AgentContext): Promise<Turn>;
|
|
371
371
|
/**
|
|
372
|
-
* 可选 turn
|
|
373
|
-
*
|
|
374
|
-
*
|
|
375
|
-
*
|
|
372
|
+
* 可选 turn 失败分类器:归类一次 send 失败(抛出或返回 `status: "failed"` 的 Turn),
|
|
373
|
+
* 返回 `undefined` 表示不认识、回落保守兜底。链上排在实验的 `classifyFailure` 之后,
|
|
374
|
+
* 实验作者认领过的失败问不到这里。分类器只声明决策轴与诊断词,不影响重试策略(次数、
|
|
375
|
+
* 退避对所有 agent 一致);抛错按 `undefined` 回落并被吞掉,不掩盖原始失败。形状与分类链、
|
|
376
|
+
* 执行体时序见 docs/feature/error-classification/architecture.md。
|
|
376
377
|
*/
|
|
377
378
|
classifyTurnError?: TurnErrorClassifier;
|
|
378
379
|
teardown?: AgentTeardown;
|
package/src/cli.ts
CHANGED
|
@@ -19,6 +19,8 @@ import { fingerprintEvalsFilter, resolveExperimentEvals, selectedEvalsForRun, sp
|
|
|
19
19
|
import { failureDetailFromResult } from "./runner/feedback/failure.ts";
|
|
20
20
|
import { stopAllSandboxes, liveSandboxCount } from "./sandbox/registry.ts";
|
|
21
21
|
import { drainExperimentTeardowns } from "./runner/experiment-cleanup-registry.ts";
|
|
22
|
+
import { drainHeldCaseLocks, isCaseLockStale, readCaseLock } from "./runner/lock.ts";
|
|
23
|
+
import { drainHeldGateLeases } from "./runner/gate-lease.ts";
|
|
22
24
|
import { CLEANUP_TIMEOUT_MS, withCleanupTimeout } from "./runner/cleanup-timeout.ts";
|
|
23
25
|
import type { ExperimentHookContext } from "./runner/types.ts";
|
|
24
26
|
import { evalLevelStats } from "./shared/verdict.ts";
|
|
@@ -175,7 +177,7 @@ const FLAG_OPTIONS = {
|
|
|
175
177
|
// 数字 `--attempt`,选哪个 attempt 由 locator 精确指名,不是「先选 eval 再挑第几次」。
|
|
176
178
|
/** `show` 命令专用:该 attempt 运行时保存的 Eval 源码,gate/soft 断言标回源码行(证据切面)。 */
|
|
177
179
|
source: { type: "boolean" },
|
|
178
|
-
/** `show` 命令专用:该 attempt 的标准执行事件流(消息、thinking、Skill load、工具调用/结果);有 OTel 时同一节点补时间(证据切面)
|
|
180
|
+
/** `show` 命令专用:该 attempt 的标准执行事件流(消息、thinking、Skill load、工具调用/结果);有 OTel 时同一节点补时间(证据切面)。每个内容段最多预览前 3 行,截断尾巴自带 `--expand` 展开句柄。 */
|
|
179
181
|
execution: { type: "boolean" },
|
|
180
182
|
/** `show` 命令专用:整个 Attempt 的统一时间树;裸 `--timing` 给有界诊断投影,`--timing=full` 逐节点展开全部 runner/已关联 OTel 节点。 */
|
|
181
183
|
timing: { type: "boolean" },
|
|
@@ -545,19 +547,32 @@ async function openBrowser(url: string): Promise<boolean> {
|
|
|
545
547
|
function assembleInvocationCompletion(state: RunFeedbackState): InvocationCompletion {
|
|
546
548
|
let unstarted = 0;
|
|
547
549
|
let failFastSkipped = 0;
|
|
550
|
+
let haltedSkipped = 0;
|
|
548
551
|
let interrupted = false;
|
|
549
552
|
const reporterErrors: ReporterError[] = [];
|
|
550
553
|
for (const d of state.diagnostics) {
|
|
551
|
-
|
|
554
|
+
// 归类按**稳定词法** `code`,不按 `key`:`key` 里编着折叠身份(experimentId / evalId /
|
|
555
|
+
// reporter 名),拿它做前缀匹配会在「把身份从 key 里摘出去」时静默失配——记账悄悄归零,
|
|
556
|
+
// 没有任何测试或类型会报警。`code` 省略时回落到 key 的首段:缺省 key 恒是
|
|
557
|
+
// `${code}:${identity}`(见 sink.ts 的 DiagnosticInput),首段即 code。
|
|
558
|
+
const code = d.code ?? d.key.split(":", 1)[0];
|
|
559
|
+
if (code === "interrupted") {
|
|
552
560
|
interrupted = true;
|
|
553
|
-
} else if (
|
|
561
|
+
} else if (code === "budget-exhausted") {
|
|
554
562
|
unstarted += d.count;
|
|
555
|
-
} else if (
|
|
563
|
+
} else if (code === "fail-fast") {
|
|
556
564
|
// run 级 fail-fast 造成的未派发同样计入 unstarted(结论落 incomplete,见
|
|
557
565
|
// docs/feature/experiments/architecture.md「Completion 与退出」)。
|
|
558
566
|
unstarted += d.count;
|
|
559
567
|
failFastSkipped += d.count;
|
|
560
|
-
} else if (
|
|
568
|
+
} else if (code === "dispatch-halted") {
|
|
569
|
+
// 止损闸停派发造成的未派发(见 docs/feature/error-classification/architecture.md
|
|
570
|
+
// 「记账」)。这条诊断的 count 是「同一死因被声明了几次」(重复声明折叠),不是未派发数——
|
|
571
|
+
// 未派发数由 emitter 累计后写在 data.unstarted 里(与 budget-exhausted 同一口径)。
|
|
572
|
+
const halted = typeof d.data?.unstarted === "number" ? d.data.unstarted : 0;
|
|
573
|
+
unstarted += halted;
|
|
574
|
+
haltedSkipped += halted;
|
|
575
|
+
} else if (code === "reporter-error") {
|
|
561
576
|
// required 决定这条错误是否写进 InvocationCompletion.reporterErrors 并让 completion 非 complete
|
|
562
577
|
// (见 docs/cli.md「required reporter」);best-effort reporter 的失败只保留为 diagnostic。
|
|
563
578
|
if (d.data?.required !== true) continue;
|
|
@@ -571,9 +586,9 @@ function assembleInvocationCompletion(state: RunFeedbackState): InvocationComple
|
|
|
571
586
|
// 中断造成的未派发(仍在 queued 的 attempt)同样计入 unstarted(见 docs/feature/experiments/
|
|
572
587
|
// architecture.md「Completion 与退出」:budget 耗尽、fail-fast 或中断造成的未派发都不伪装成全绿)。
|
|
573
588
|
if (interrupted) unstarted += state.queued;
|
|
574
|
-
// attempt:early-exit 计数含 fail-fast
|
|
575
|
-
// 「省下的重复验证」=
|
|
576
|
-
const earlyExitUnstarted = Math.max(0, state.earlyExitSkipped - failFastSkipped);
|
|
589
|
+
// attempt:early-exit 计数含 fail-fast 与止损闸的未派发(反馈层同一事件驱动计数守恒);
|
|
590
|
+
// 「省下的重复验证」= 总数减去那两部分。
|
|
591
|
+
const earlyExitUnstarted = Math.max(0, state.earlyExitSkipped - failFastSkipped - haltedSkipped);
|
|
577
592
|
const status: CompletionStatus = interrupted
|
|
578
593
|
? "interrupted"
|
|
579
594
|
: unstarted > 0 || reporterErrors.length > 0
|
|
@@ -875,6 +890,9 @@ async function main(): Promise<void> {
|
|
|
875
890
|
maxConcurrency: exp.maxConcurrency,
|
|
876
891
|
setup: exp.setup,
|
|
877
892
|
teardown: exp.teardown,
|
|
893
|
+
// 实验级失败分类器:随 AgentRun 进 attempt(turn 链与生命周期链共用同一份),
|
|
894
|
+
// 产出的 scope 由止损闸消费(见 docs/feature/error-classification/architecture.md)。
|
|
895
|
+
classifyFailure: exp.classifyFailure,
|
|
878
896
|
});
|
|
879
897
|
}
|
|
880
898
|
} else {
|
|
@@ -926,21 +944,36 @@ async function main(): Promise<void> {
|
|
|
926
944
|
// 人读 `--dry` 首行的携入摘要(见 docs/feature/experiments/cli.md 开头示例与「事件与计划
|
|
927
945
|
// 文档的 TypeScript 形状」),口径必须与真正开跑时一致。
|
|
928
946
|
const priorResults = flags.force ? undefined : await loadLatestResultsPerEval(join(cwd, ".niceeval"));
|
|
929
|
-
const carryPlan = priorResults?.length
|
|
947
|
+
const carryPlan = priorResults?.length
|
|
948
|
+
? await planCarry(evals, agentRuns, priorResults, config.sandbox, config.timeoutMs)
|
|
949
|
+
: undefined;
|
|
930
950
|
|
|
931
951
|
if (flags.dry) {
|
|
932
952
|
// --dry 只按所选形态打印计划,不运行、不落盘——一次完成的读取,不是事件流
|
|
933
953
|
// (见 docs/feature/experiments/cli.md「机器怎么读:--json」)。两种形态共用同一份摊平
|
|
934
954
|
// 矩阵——(experimentId, evalId) 逐行,携带同一口径的 reused 预测——不是各自重算一遍。
|
|
935
955
|
const dryRuns = Math.max(1, ...agentRuns.map((r) => r.runs));
|
|
936
|
-
const
|
|
956
|
+
const rowInputs: { experimentId: string; evalId: string; reused: boolean }[] = [];
|
|
937
957
|
for (let i = 0; i < agentRuns.length; i++) {
|
|
938
958
|
const run = agentRuns[i]!;
|
|
939
959
|
for (const e of matchedByRun[i]!) {
|
|
940
960
|
const carriedCount = carryPlan?.carriedAttemptsByKey.get(cacheKey(run, e.id))?.size ?? 0;
|
|
941
|
-
|
|
961
|
+
rowInputs.push({ experimentId: run.experimentId ?? "", evalId: e.id, reused: carriedCount >= run.runs });
|
|
942
962
|
}
|
|
943
963
|
}
|
|
964
|
+
// 只读锁目录,不取锁、不等待(见 docs/feature/experiments/architecture.md「并发
|
|
965
|
+
// Invocation:用例锁」);过期(无人续心跳)的锁不算"正被持锁运行",不标注。裸 run(没有
|
|
966
|
+
// experimentId)不参与锁,恒不标注。并行读——矩阵行数可能不小,不逐行串行等磁盘。
|
|
967
|
+
const niceevalRootForDry = resolvePath(cwd, ".niceeval");
|
|
968
|
+
const now = Date.now();
|
|
969
|
+
const lockedFlags = await Promise.all(
|
|
970
|
+
rowInputs.map(async (row) => {
|
|
971
|
+
if (!row.experimentId) return false;
|
|
972
|
+
const lock = await readCaseLock(niceevalRootForDry, row.experimentId, row.evalId).catch(() => undefined);
|
|
973
|
+
return lock !== undefined && !isCaseLockStale(lock, now);
|
|
974
|
+
}),
|
|
975
|
+
);
|
|
976
|
+
const matrix: JsonPlanRow[] = rowInputs.map((row, i) => ({ ...row, ...(lockedFlags[i] ? { locked: true } : {}) }));
|
|
944
977
|
if (outputForm === "json") {
|
|
945
978
|
process.stdout.write(
|
|
946
979
|
renderJsonPlanDocument({
|
|
@@ -959,7 +992,7 @@ async function main(): Promise<void> {
|
|
|
959
992
|
configs: agentRuns.length,
|
|
960
993
|
runs: dryRuns,
|
|
961
994
|
reused: carryPlan?.carriedResults.length ?? 0,
|
|
962
|
-
rows: matrix.map((row) => ({ experimentId: row.experimentId, evalId: row.evalId })),
|
|
995
|
+
rows: matrix.map((row) => ({ experimentId: row.experimentId, evalId: row.evalId, locked: row.locked })),
|
|
963
996
|
}),
|
|
964
997
|
);
|
|
965
998
|
}
|
|
@@ -1025,6 +1058,8 @@ async function main(): Promise<void> {
|
|
|
1025
1058
|
const settled = Promise.allSettled([
|
|
1026
1059
|
...(runInFlight ? [runInFlight] : []),
|
|
1027
1060
|
drainExperimentTeardowns(),
|
|
1061
|
+
drainHeldCaseLocks(),
|
|
1062
|
+
drainHeldGateLeases(),
|
|
1028
1063
|
]);
|
|
1029
1064
|
await Promise.race([
|
|
1030
1065
|
settled.then(() => {}),
|
|
@@ -1101,9 +1136,12 @@ async function main(): Promise<void> {
|
|
|
1101
1136
|
}
|
|
1102
1137
|
|
|
1103
1138
|
// 正常返回(含被中断后走部分汇总)后再兜一刀:Scope finalizer 没停掉的残留沙箱、没被运行
|
|
1104
|
-
// 路径消费的实验级 cleanup
|
|
1139
|
+
// 路径消费的实验级 cleanup、没被 per-attempt Effect.ensuring 释放的用例锁与实验闸租约在这里
|
|
1140
|
+
// 强清。跑顺利时四份登记表都已空,是 no-op。
|
|
1105
1141
|
await stopAllSandboxes();
|
|
1106
1142
|
await drainExperimentTeardowns();
|
|
1143
|
+
await drainHeldCaseLocks();
|
|
1144
|
+
await drainHeldGateLeases();
|
|
1107
1145
|
|
|
1108
1146
|
// completion 要先算好,--junit 是否"这次真的写出"才有依据(见下)。
|
|
1109
1147
|
const completion = assembleInvocationCompletion(coordinator.state);
|
|
@@ -1140,8 +1178,11 @@ async function main(): Promise<void> {
|
|
|
1140
1178
|
|
|
1141
1179
|
main().catch(async (e) => {
|
|
1142
1180
|
process.stderr.write(t("cli.error", { error: formatThrown(e) }));
|
|
1143
|
-
// 真·崩溃路径也别留孤儿:强清还活着的沙箱(带超时)、排空实验级 cleanup
|
|
1181
|
+
// 真·崩溃路径也别留孤儿:强清还活着的沙箱(带超时)、排空实验级 cleanup 注册表、用例锁与
|
|
1182
|
+
// 实验闸租约,再退。
|
|
1144
1183
|
await stopAllSandboxes();
|
|
1145
1184
|
await drainExperimentTeardowns();
|
|
1185
|
+
await drainHeldCaseLocks();
|
|
1186
|
+
await drainHeldGateLeases();
|
|
1146
1187
|
process.exit(2);
|
|
1147
1188
|
});
|
|
@@ -355,6 +355,40 @@ function baseScoringContext(state: ContextState) {
|
|
|
355
355
|
};
|
|
356
356
|
}
|
|
357
357
|
|
|
358
|
+
describe("t.* 作用域断言聚合全部轮次(callId 跨轮复用)", () => {
|
|
359
|
+
// 回归:续轮场景下 adapter 常按轮各自编号(复用 c1)。第一轮读了 INDEX,第二轮才给答复;
|
|
360
|
+
// t.calledTool 聚合全部轮次,应命中第一轮的 read——旧折叠按 callId 覆盖会让它「只扫最后一轮」而 miss。
|
|
361
|
+
it("t.calledTool 命中发生在第一轮、callId 被第二轮复用的工具调用", async () => {
|
|
362
|
+
const agent = scriptedAgent([
|
|
363
|
+
{
|
|
364
|
+
status: "completed",
|
|
365
|
+
events: [
|
|
366
|
+
{ type: "action.called", callId: "c1", name: "read", input: { path: "INDEX.md" } },
|
|
367
|
+
{ type: "action.result", callId: "c1", output: "index contents", status: "completed" },
|
|
368
|
+
{ type: "message", role: "assistant", text: "读完了 INDEX,继续" },
|
|
369
|
+
],
|
|
370
|
+
},
|
|
371
|
+
{
|
|
372
|
+
status: "completed",
|
|
373
|
+
events: [
|
|
374
|
+
{ type: "action.called", callId: "c1", name: "write", input: { path: "note.md" } },
|
|
375
|
+
{ type: "action.result", callId: "c1", output: "ok", status: "completed" },
|
|
376
|
+
{ type: "message", role: "assistant", text: "答复" },
|
|
377
|
+
],
|
|
378
|
+
},
|
|
379
|
+
]);
|
|
380
|
+
const { context, state } = makeContext(agent);
|
|
381
|
+
await context.send("第一轮"); // 读 INDEX
|
|
382
|
+
await context.send("第二轮"); // 续轮,复用 callId c1
|
|
383
|
+
|
|
384
|
+
context.calledTool("read", { input: { path: "INDEX.md" } });
|
|
385
|
+
|
|
386
|
+
const [result] = await state.collector.finalize(baseScoringContext(state));
|
|
387
|
+
expect(result.name).toBe("calledTool(read)");
|
|
388
|
+
expect(result.outcome).toBe("passed");
|
|
389
|
+
});
|
|
390
|
+
});
|
|
391
|
+
|
|
358
392
|
describe("TurnHandle scoped assertions (parked/loadedSkill/noFailedActions/maxTokens/maxCost)", () => {
|
|
359
393
|
it("mirror t/session scope: turn.parked() reflects this turn's own waiting status", async () => {
|
|
360
394
|
const agent = scriptedAgent([
|
package/src/context/context.ts
CHANGED
|
@@ -12,6 +12,7 @@ import * as Scoped from "../scoring/scoped.ts";
|
|
|
12
12
|
import { buildJudge } from "../scoring/judge.ts";
|
|
13
13
|
import { EvalSkipped, EvalRequirementFailed, TurnFailed } from "./control-flow.ts";
|
|
14
14
|
import { turnErrorText } from "./turn-errors.ts";
|
|
15
|
+
import { attachFailureClass, type FailureClass } from "../shared/failure-class.ts";
|
|
15
16
|
import type { ConcurrencySlot } from "./send-retry.ts";
|
|
16
17
|
import { deriveRunFacts } from "../o11y/derive.ts";
|
|
17
18
|
import { diffIsEmpty, diffMatches, emptyDiffData } from "../scoring/diff.ts";
|
|
@@ -91,6 +92,8 @@ export interface ContextDeps {
|
|
|
91
92
|
onTurn?: import("./session.ts").SessionDeps["onTurn"];
|
|
92
93
|
/** turn 级重试退避期间释放/收回的全局并发槽位;透传给 SessionManager。 */
|
|
93
94
|
concurrencySlot?: ConcurrencySlot;
|
|
95
|
+
/** 实验声明的失败分类器(`ExperimentDef.classifyFailure`);透传给 SessionManager。 */
|
|
96
|
+
experimentClassifier?: import("./session.ts").SessionDeps["experimentClassifier"];
|
|
94
97
|
/** 仅供确定性单测注入:透传给 SessionManager 的 turn 重试随机数/睡眠(生产路径省略)。 */
|
|
95
98
|
retryRandom?: import("./session.ts").SessionDeps["retryRandom"];
|
|
96
99
|
retrySleep?: import("./session.ts").SessionDeps["retrySleep"];
|
|
@@ -131,6 +134,7 @@ export function createEvalContext(deps: ContextDeps): { context: TestContext; st
|
|
|
131
134
|
onTurn: deps.onTurn,
|
|
132
135
|
ledgerHooks: deps.ledgerHooks,
|
|
133
136
|
concurrencySlot: deps.concurrencySlot,
|
|
137
|
+
experimentClassifier: deps.experimentClassifier,
|
|
134
138
|
retryRandom: deps.retryRandom,
|
|
135
139
|
retrySleep: deps.retrySleep,
|
|
136
140
|
});
|
|
@@ -332,11 +336,11 @@ export function createEvalContext(deps: ContextDeps): { context: TestContext; st
|
|
|
332
336
|
const text = typeof input === "string" ? input : input.text;
|
|
333
337
|
const files = typeof input === "string" ? undefined : input.files;
|
|
334
338
|
const turn = await send(session, text, files);
|
|
335
|
-
return makeTurnHandle(turn, collector, deps, text, manager.resolveTurnCoverage(turn));
|
|
339
|
+
return makeTurnHandle(turn, collector, deps, text, manager.resolveTurnCoverage(turn), manager.resolveTurnFailureClass(turn));
|
|
336
340
|
},
|
|
337
341
|
sendFile: async (path, text) => {
|
|
338
342
|
const turn = await send(session, text ?? "", [await readInputFile(path)]);
|
|
339
|
-
return makeTurnHandle(turn, collector, deps, text ?? "", manager.resolveTurnCoverage(turn));
|
|
343
|
+
return makeTurnHandle(turn, collector, deps, text ?? "", manager.resolveTurnCoverage(turn), manager.resolveTurnFailureClass(turn));
|
|
340
344
|
},
|
|
341
345
|
requireInputRequest: (filter) => requireInputRequest(session, filter),
|
|
342
346
|
respond: async (...answers) => {
|
|
@@ -344,7 +348,7 @@ export function createEvalContext(deps: ContextDeps): { context: TestContext; st
|
|
|
344
348
|
const built = buildRespondInput(session, answers);
|
|
345
349
|
session.pendingInputRequests.length = 0;
|
|
346
350
|
const turn = await send(session, built.text, undefined, built.responses);
|
|
347
|
-
return makeTurnHandle(turn, collector, deps, built.text, manager.resolveTurnCoverage(turn));
|
|
351
|
+
return makeTurnHandle(turn, collector, deps, built.text, manager.resolveTurnCoverage(turn), manager.resolveTurnFailureClass(turn));
|
|
348
352
|
},
|
|
349
353
|
respondAll: async (optionId) => {
|
|
350
354
|
if (session.pendingInputRequests.length === 0) {
|
|
@@ -359,7 +363,7 @@ export function createEvalContext(deps: ContextDeps): { context: TestContext; st
|
|
|
359
363
|
session.pendingInputRequests.length = 0;
|
|
360
364
|
const input = requests.map(() => optionId).join("\n");
|
|
361
365
|
const turn = await send(session, input, undefined, responses);
|
|
362
|
-
return makeTurnHandle(turn, collector, deps, input, manager.resolveTurnCoverage(turn));
|
|
366
|
+
return makeTurnHandle(turn, collector, deps, input, manager.resolveTurnCoverage(turn), manager.resolveTurnFailureClass(turn));
|
|
363
367
|
},
|
|
364
368
|
get reply() {
|
|
365
369
|
return session.lastMessage;
|
|
@@ -565,6 +569,7 @@ function makeTurnHandle(
|
|
|
565
569
|
deps: ContextDeps,
|
|
566
570
|
input: string,
|
|
567
571
|
coverage: ResolvedCoverage,
|
|
572
|
+
failureClass?: FailureClass,
|
|
568
573
|
): TurnHandle {
|
|
569
574
|
const message = lastAssistantText(turn.events) ?? "";
|
|
570
575
|
const facts = deriveRunFacts(turn.events);
|
|
@@ -598,7 +603,11 @@ function makeTurnHandle(
|
|
|
598
603
|
// 与保守兜底分类器、turn 级重试摘要读的同一段文本(见 turn-errors.ts 的 turnErrorText)——
|
|
599
604
|
// 不出现「报错说 A、分类看 B」。
|
|
600
605
|
const message = turnErrorText(turn);
|
|
601
|
-
|
|
606
|
+
const error = new TurnFailed(message !== undefined ? t("context.turnFailed", { message }) : undefined);
|
|
607
|
+
// 终局失败的分类随错误浮出:attempt 封口据此落止损闸(见
|
|
608
|
+
// docs/feature/error-classification/architecture.md「止损执行体」)。没有分类
|
|
609
|
+
// (未经重试执行体的手工构造 Turn)时不标记,落缺省 attempt 档。
|
|
610
|
+
throw failureClass ? attachFailureClass(error, failureClass) : error;
|
|
602
611
|
}
|
|
603
612
|
return handle;
|
|
604
613
|
},
|