niceeval 0.10.3-canary.7 → 0.11.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/agents/types.d.ts +5 -4
- package/dist/context/turn-errors.d.ts +27 -23
- package/dist/i18n/en.d.ts +14 -0
- package/dist/i18n/en.js +15 -1
- package/dist/i18n/zh-CN.d.ts +15 -1
- package/dist/i18n/zh-CN.js +15 -1
- package/dist/o11y/derive.js +28 -24
- package/dist/o11y/types.d.ts +8 -5
- package/dist/report/components/attempt-detail/UsageTable.js +4 -7
- package/dist/report/components/attempt-detail/compute.d.ts +2 -2
- package/dist/report/components/attempt-detail/compute.js +2 -6
- package/dist/report/components/attempt-detail/faces.js +7 -7
- package/dist/report/components/attempt-detail/index.js +0 -3
- package/dist/report/components/entity-lists/EvalList.js +0 -0
- package/dist/report/components/metric-views/compute.js +1 -1
- package/dist/report/model/types.d.ts +3 -4
- package/dist/results/locator.js +0 -0
- package/dist/results/select.d.ts +6 -0
- package/dist/results/select.js +8 -0
- package/dist/runner/feedback/sink.d.ts +26 -1
- package/dist/runner/fingerprint.d.ts +23 -0
- package/dist/runner/types.d.ts +105 -7
- package/dist/sandbox/errors.d.ts +29 -0
- package/dist/sandbox/resolve.d.ts +9 -0
- package/dist/shared/failure-class.d.ts +91 -0
- package/dist/types.d.ts +1 -0
- package/dist/util.d.ts +3 -2
- package/dist/util.js +31 -5
- package/docs-site/zh/explanation/runner.mdx +35 -0
- package/docs-site/zh/reference/cli.mdx +3 -3
- package/docs-site/zh/reference/events.mdx +4 -4
- package/docs-site/zh/troubleshooting/debugging.mdx +2 -2
- package/docs-site/zh/tutorials/viewing-results.mdx +4 -5
- package/package.json +4 -12
- package/src/agents/ai-sdk.test.ts +26 -0
- package/src/agents/ai-sdk.ts +7 -4
- package/src/agents/index.ts +5 -3
- package/src/agents/langgraph.test.ts +30 -0
- package/src/agents/langgraph.ts +5 -2
- package/src/agents/openai-compat.test.ts +35 -0
- package/src/agents/openai-compat.ts +16 -4
- package/src/agents/sdk-streams.test.ts +66 -0
- package/src/agents/sdk-streams.ts +3 -1
- package/src/agents/types.ts +5 -4
- package/src/cli.ts +55 -14
- package/src/context/context.test.ts +34 -0
- package/src/context/context.ts +14 -5
- package/src/context/send-retry.test.ts +86 -0
- package/src/context/send-retry.ts +37 -12
- package/src/context/session.ts +24 -0
- package/src/context/turn-errors.test.ts +124 -17
- package/src/context/turn-errors.ts +60 -50
- package/src/define.ts +5 -0
- package/src/i18n/en.ts +20 -1
- package/src/i18n/zh-CN.ts +20 -1
- package/src/index.ts +8 -0
- package/src/o11y/cost.ts +4 -2
- package/src/o11y/derive.test.ts +40 -0
- package/src/o11y/derive.ts +28 -22
- package/src/o11y/otlp/sandbox-receiver.test.ts +201 -0
- package/src/o11y/otlp/sandbox-receiver.ts +73 -27
- package/src/o11y/parsers/bub.test.ts +30 -0
- package/src/o11y/parsers/bub.ts +5 -2
- package/src/o11y/parsers/codex.test.ts +19 -0
- package/src/o11y/parsers/codex.ts +5 -2
- package/src/o11y/types.ts +8 -5
- package/src/report/components/attempt-detail/UsageTable.tsx +4 -6
- package/src/report/components/attempt-detail/attempt-components.test.tsx +5 -8
- package/src/report/components/attempt-detail/compute.ts +2 -7
- package/src/report/components/attempt-detail/faces.ts +7 -7
- package/src/report/components/attempt-detail/index.tsx +0 -3
- package/src/report/components/entity-lists/EvalList.tsx +0 -0
- package/src/report/components/metric-views/compute.ts +1 -1
- package/src/report/model/types.ts +3 -4
- package/src/results/format.ts +9 -2
- package/src/results/index.ts +2 -0
- package/src/results/locator.ts +0 -0
- package/src/results/open.ts +132 -21
- package/src/results/select.ts +12 -0
- package/src/results/skipped-notice.ts +0 -0
- package/src/runner/attempt.test.ts +116 -0
- package/src/runner/attempt.ts +89 -8
- package/src/runner/discover.ts +21 -3
- package/src/runner/feedback/coordinator.ts +24 -2
- package/src/runner/feedback/eval-conclusions.ts +6 -3
- package/src/runner/feedback/human.test.ts +300 -6
- package/src/runner/feedback/human.ts +132 -30
- package/src/runner/feedback/json.test.ts +127 -2
- package/src/runner/feedback/json.ts +51 -3
- package/src/runner/feedback/reducer.test.ts +328 -29
- package/src/runner/feedback/reducer.ts +84 -9
- package/src/runner/feedback/sink.ts +39 -1
- package/src/runner/fingerprint.ts +49 -19
- package/src/runner/gate-lease.test.ts +510 -0
- package/src/runner/gate-lease.ts +350 -0
- package/src/runner/lock.test.ts +454 -0
- package/src/runner/lock.ts +288 -0
- package/src/runner/report.test.ts +1 -0
- package/src/runner/run.test.ts +2044 -9
- package/src/runner/run.ts +825 -61
- package/src/runner/teardown-registry.ts +20 -78
- package/src/runner/types.ts +103 -7
- package/src/sandbox/errors.test.ts +72 -0
- package/src/sandbox/errors.ts +83 -0
- package/src/sandbox/keep-registry.ts +22 -46
- package/src/sandbox/resolve.test.ts +100 -0
- package/src/sandbox/resolve.ts +84 -27
- package/src/shared/entry-file-store.test.ts +149 -0
- package/src/shared/entry-file-store.ts +117 -0
- package/src/shared/failure-class.test.ts +137 -0
- package/src/shared/failure-class.ts +175 -0
- package/src/show/index.ts +5 -6
- package/src/show/render.test.ts +116 -15
- package/src/show/render.ts +124 -35
- package/src/types.ts +9 -0
- package/src/util.ts +31 -4
- package/src/view/app/App.tsx +5 -1
- package/src/view/client-dist/app.js +1 -1
- package/src/view/data.ts +5 -9
- package/src/view/shared/types.ts +7 -1
- package/src/view/view-report.test.ts +54 -0
package/src/i18n/en.ts
CHANGED
|
@@ -74,6 +74,14 @@ export const en = {
|
|
|
74
74
|
"experiment {{experimentId}}'s teardown was not triggered by the normal countdown path; it has been executed by the end-of-run sweep instead. Results are unaffected; seeing this line means an unlocated intermittent scheduling issue fired — please record this run in the memory ledger.\n",
|
|
75
75
|
"runner.teardownRegistrationWriteFailed":
|
|
76
76
|
"writing the crash-recovery teardown registration for experiment {{experimentId}} failed: {{message}}. The run continues normally, but a SIGKILL during this run cannot be recovered via `niceeval exp --teardown` or the startup self-heal — check disk space/permissions under .niceeval/teardowns/.\n",
|
|
77
|
+
"runner.lockTakenOver":
|
|
78
|
+
"took over an expired case lock for {{experimentId}}/{{evalId}} (previously held by pid {{pid}} on {{host}}; its heartbeat went stale) — that run likely died without releasing it; this run now owns dispatching this case.\n",
|
|
79
|
+
"runner.gateLeaseTakenOver":
|
|
80
|
+
"took over an expired concurrency-slot lease for experiment {{experimentId}} (slot {{slot}}, previously held by pid {{pid}} on {{host}}; its heartbeat went stale) — that run likely died without releasing it; this run now owns the slot.\n",
|
|
81
|
+
"runner.gateLeaseWaiting":
|
|
82
|
+
"waiting on another run for experiment {{experimentId}}'s concurrency slots: all {{effectiveN}} in use ({{holders}}). Concurrent runs share this experiment's slots, and the smallest maxConcurrency in play wins — this run declared {{declaredN}}. Nothing dispatches until a slot frees up; the other run's slots release when its attempts finish, or 30s after it dies.\n",
|
|
83
|
+
"runner.dispatchHaltedExperiment": "experiment halted (dispatch-halted): {{message}}\n",
|
|
84
|
+
"runner.dispatchHaltedEval": "eval halted: {{message}}\n",
|
|
77
85
|
"judge.modelMissing":
|
|
78
86
|
"No judge model configured. Set it in defineConfig({ judge: { model: \"...\" } }), the eval's judge config, or the NICEEVAL_JUDGE_MODEL environment variable (there is no built-in default model).\n" +
|
|
79
87
|
" Docs: node_modules/niceeval/docs-site/zh/tutorials/scoring-guide.mdx",
|
|
@@ -236,6 +244,7 @@ export const en = {
|
|
|
236
244
|
"define.experimentFlagNotJson": "experiment.flags.{{key}} is not JSON-serializable (functions / undefined / cycles / bigint are not allowed); flags are persisted verbatim into result snapshots and must be plain JSON.",
|
|
237
245
|
"define.experimentLabelInvalid": "experiment.labels.{{key}} must be a string or a finite number; labels are report-side grouping coordinates persisted verbatim into result snapshots.",
|
|
238
246
|
"define.experimentSetupNotFunction": "experiment.setup must be a function ((ctx) => void); use experiment.teardown for cleanup; to prepare the in-sandbox environment per experiment, chain .setup() hooks on the sandbox spec instead.",
|
|
247
|
+
"define.experimentClassifyFailureNotFunction": "experiment.classifyFailure must be a function ((failure) => FailureClass | undefined); it classifies failures that surface as third-party errors and must return undefined for anything it does not recognize.",
|
|
239
248
|
"define.experimentIdRejected": "defineExperiment does not accept id; ids are derived from file paths.",
|
|
240
249
|
"define.sandboxAgentNameRequired": "defineSandboxAgent requires name.",
|
|
241
250
|
"define.sandboxCreateRequired": "defineSandbox requires a create() function.",
|
|
@@ -249,7 +258,8 @@ export const en = {
|
|
|
249
258
|
"feedback.human.active": "ACTIVE",
|
|
250
259
|
"feedback.human.budgetExhausted": "budget exhausted for {{experimentId}} (spent {{spent}}, unstarted {{unstarted}})",
|
|
251
260
|
"feedback.human.compare": "Compare: niceeval view",
|
|
252
|
-
"feedback.human.counts":
|
|
261
|
+
"feedback.human.counts":
|
|
262
|
+
"{{total}} total · {{reused}} reused · {{running}} running · {{queued}} queued · {{passed}} passed · {{failed}} failed · {{errored}} errored · {{skipped}} skipped",
|
|
253
263
|
"feedback.human.diffHint": "Diff: niceeval show {{locator}} --diff",
|
|
254
264
|
"feedback.human.evalHint": "Eval: niceeval show {{locator}} --source",
|
|
255
265
|
"feedback.human.failuresHeader": "FAILURES",
|
|
@@ -282,6 +292,15 @@ export const en = {
|
|
|
282
292
|
"feedback.human.hookFailed": "failed",
|
|
283
293
|
"feedback.human.precheckJudge": "prechecking judge config",
|
|
284
294
|
"feedback.human.precheckJudgeDone": "judge config ok",
|
|
295
|
+
"feedback.human.countsWithElsewhere":
|
|
296
|
+
"{{total}} total · {{reused}} reused · {{running}} running · {{elsewhere}} elsewhere · {{queued}} queued · {{passed}} passed · {{failed}} failed · {{errored}} errored · {{skipped}} skipped",
|
|
297
|
+
"feedback.human.waitingOnAnotherRun": "waiting on another run",
|
|
298
|
+
"feedback.human.lockWaitDetail": "{{count}} evals · pid {{pid}}",
|
|
299
|
+
"feedback.human.lockWaitStarted": "waiting on another run · {{experimentId}} ({{count}} evals, pid {{pid}})",
|
|
300
|
+
"feedback.human.lockWaitResolved": "lock wait resolved · {{experimentId}} ({{summary}}, {{elapsed}})",
|
|
301
|
+
"feedback.human.lockWaitCarried": "{{count}} carried",
|
|
302
|
+
"feedback.human.lockWaitDispatched": "{{count}} to run",
|
|
303
|
+
"feedback.human.lockedRowSuffix": "locked",
|
|
285
304
|
"feedback.phase.sandboxSetup": "sandbox setup",
|
|
286
305
|
"feedback.phase.scoring": "scoring",
|
|
287
306
|
"feedback.phase.teardown": "cleaning up",
|
package/src/i18n/zh-CN.ts
CHANGED
|
@@ -71,6 +71,14 @@ export const zhCN = {
|
|
|
71
71
|
"实验 {{experimentId}} 的 teardown 未被正常计数路径触发,已在运行收尾兜底执行。结果不受影响;这行出现说明命中了一个未定位的调度间歇问题,请把本次运行信息记入 memory 台账。\n",
|
|
72
72
|
"runner.teardownRegistrationWriteFailed":
|
|
73
73
|
"实验 {{experimentId}} 的强杀恢复收尾登记写入失败:{{message}}。本次运行照常继续,但这次运行期间若被 SIGKILL,`niceeval exp --teardown` 与启动自愈都无法找到它——检查 .niceeval/teardowns/ 下的磁盘空间或权限。\n",
|
|
74
|
+
"runner.lockTakenOver":
|
|
75
|
+
"接管了 {{experimentId}}/{{evalId}} 的过期用例锁(原持有者 pid {{pid}}@{{host}},心跳已过期)——那次运行大概率没能正常释放它;本次运行现在接手派发这条用例。\n",
|
|
76
|
+
"runner.gateLeaseTakenOver":
|
|
77
|
+
"接管了实验 {{experimentId}} 的过期并发名额租约(槽位 {{slot}},原持有者 pid {{pid}}@{{host}},心跳已过期)——那次运行大概率没能正常释放它;本次运行现在接手这个名额。\n",
|
|
78
|
+
"runner.gateLeaseWaiting":
|
|
79
|
+
"在等别的运行让出实验 {{experimentId}} 的并发名额:生效的 {{effectiveN}} 个位子全被占着({{holders}})。并行运行共用同一实验的名额,生效值取在场声明里最小的那个——本次运行声明的是 {{declaredN}}。名额腾出来之前不会派发任何 attempt;对方的名额会在它的 attempt 跑完时释放,它若已死则 30s 后过期被接管。\n",
|
|
80
|
+
"runner.dispatchHaltedExperiment": "实验已止损(dispatch-halted):{{message}}\n",
|
|
81
|
+
"runner.dispatchHaltedEval": "eval 已止损:{{message}}\n",
|
|
74
82
|
"judge.modelMissing":
|
|
75
83
|
"judge 未配置模型:在 defineConfig({ judge: { model: \"...\" } })、eval 的 judge 配置或环境变量 NICEEVAL_JUDGE_MODEL 里指定裁判模型(没有内置默认模型)。\n" +
|
|
76
84
|
" 文档:node_modules/niceeval/docs-site/zh/tutorials/scoring-guide.mdx",
|
|
@@ -229,6 +237,7 @@ export const zhCN = {
|
|
|
229
237
|
"define.experimentFlagNotJson": "experiment.flags.{{key}} 不是可 JSON 序列化的值(函数 / undefined / 循环引用 / bigint 不允许);flags 会原样进入结果快照,必须是纯 JSON。",
|
|
230
238
|
"define.experimentLabelInvalid": "experiment.labels.{{key}} 必须是字符串或有限数字;labels 是报告侧的归类坐标,会原样进入结果快照。",
|
|
231
239
|
"define.experimentSetupNotFunction": "experiment.setup 必须是函数((ctx) => void);要清理请挂 experiment.teardown;要按实验准备沙箱内环境请挂 sandbox spec 的 .setup() 钩子链。",
|
|
240
|
+
"define.experimentClassifyFailureNotFunction": "experiment.classifyFailure 必须是函数((failure) => FailureClass | undefined):它识别以第三方错误形态浮出的失败,认不出的一律返回 undefined 交给后续链路。",
|
|
232
241
|
"define.experimentIdRejected": "defineExperiment 不接受 id —— id 由文件路径推导。",
|
|
233
242
|
"define.sandboxAgentNameRequired": "defineSandboxAgent 需要 name。",
|
|
234
243
|
"define.sandboxCreateRequired": "defineSandbox 需要一个 create() 函数。",
|
|
@@ -242,7 +251,8 @@ export const zhCN = {
|
|
|
242
251
|
"feedback.human.active": "ACTIVE",
|
|
243
252
|
"feedback.human.budgetExhausted": "{{experimentId}} 预算已耗尽(已花 {{spent}},未跑 {{unstarted}})",
|
|
244
253
|
"feedback.human.compare": "Compare: niceeval view",
|
|
245
|
-
"feedback.human.counts":
|
|
254
|
+
"feedback.human.counts":
|
|
255
|
+
"共 {{total}} · 复用 {{reused}} · 运行中 {{running}} · 排队 {{queued}} · 通过 {{passed}} · 失败 {{failed}} · 出错 {{errored}} · 跳过 {{skipped}}",
|
|
246
256
|
"feedback.human.diffHint": "Diff: niceeval show {{locator}} --diff",
|
|
247
257
|
"feedback.human.evalHint": "Eval: niceeval show {{locator}} --source",
|
|
248
258
|
"feedback.human.failuresHeader": "FAILURES",
|
|
@@ -275,6 +285,15 @@ export const zhCN = {
|
|
|
275
285
|
"feedback.human.hookFailed": "失败",
|
|
276
286
|
"feedback.human.precheckJudge": "预检 judge 配置",
|
|
277
287
|
"feedback.human.precheckJudgeDone": "judge 配置就绪",
|
|
288
|
+
"feedback.human.countsWithElsewhere":
|
|
289
|
+
"共 {{total}} · 复用 {{reused}} · 运行中 {{running}} · 等待中 {{elsewhere}} · 排队 {{queued}} · 通过 {{passed}} · 失败 {{failed}} · 出错 {{errored}} · 跳过 {{skipped}}",
|
|
290
|
+
"feedback.human.waitingOnAnotherRun": "等待另一个并行 run",
|
|
291
|
+
"feedback.human.lockWaitDetail": "{{count}} 条用例 · pid {{pid}}",
|
|
292
|
+
"feedback.human.lockWaitStarted": "等待另一个并行 run · {{experimentId}}({{count}} 条用例,pid {{pid}})",
|
|
293
|
+
"feedback.human.lockWaitResolved": "等待结束 · {{experimentId}}({{summary}},{{elapsed}})",
|
|
294
|
+
"feedback.human.lockWaitCarried": "{{count}} 条携入",
|
|
295
|
+
"feedback.human.lockWaitDispatched": "{{count}} 条自跑",
|
|
296
|
+
"feedback.human.lockedRowSuffix": "locked",
|
|
278
297
|
"feedback.phase.sandboxSetup": "沙箱预置",
|
|
279
298
|
"feedback.phase.scoring": "评分",
|
|
280
299
|
"feedback.phase.teardown": "清理中",
|
package/src/index.ts
CHANGED
|
@@ -5,6 +5,10 @@ export { defineEval, defineScoreEval, defineConfig, defineExperiment } from "./d
|
|
|
5
5
|
|
|
6
6
|
export { requireEnv, getEnv, stripComments } from "./util.ts";
|
|
7
7
|
|
|
8
|
+
// 执行失败分类:抛出点糖衣类(声明死因波及多远)+ 结构守卫。判据、分类链与止损语义见
|
|
9
|
+
// docs/feature/error-classification/README.md;`niceeval/adapter` 复导出同一份词表类型。
|
|
10
|
+
export { ExperimentFatalError, EvalFatalError, failureClassOf } from "./shared/failure-class.ts";
|
|
11
|
+
|
|
8
12
|
// 类型(eval 作者会用到;跑哪个 agent / 用哪个 sandbox 见对应子路径)
|
|
9
13
|
export type {
|
|
10
14
|
StreamEvent,
|
|
@@ -59,6 +63,10 @@ export type {
|
|
|
59
63
|
TraceSpan,
|
|
60
64
|
SpanKind,
|
|
61
65
|
DerivedFacts,
|
|
66
|
+
FailureClass,
|
|
67
|
+
FailureScope,
|
|
68
|
+
AttemptFailureInfo,
|
|
69
|
+
AttemptFailureClassifier,
|
|
62
70
|
} from "./types.ts";
|
|
63
71
|
|
|
64
72
|
export type { ParsedTranscript } from "./o11y/parsers/index.ts";
|
package/src/o11y/cost.ts
CHANGED
|
@@ -56,8 +56,10 @@ function lookupBuiltin(model: string): Price | undefined {
|
|
|
56
56
|
}
|
|
57
57
|
|
|
58
58
|
/**
|
|
59
|
-
* 按 token 桶 ×
|
|
60
|
-
* (
|
|
59
|
+
* 按 token 桶 × 单价估算一次运行的美元成本。逐桶相加成立的前提是 Usage 桶恒互斥
|
|
60
|
+
* (inputTokens 不含 cache 命中,归一义务在 adapter,见 docs/feature/results/architecture.md#usage);
|
|
61
|
+
* cache 桶缺专门单价时退回 input 价——宁可高估,不静默低估。
|
|
62
|
+
* 无 model / 查不到价(用户表 + 内置快照都没有)→ undefined。
|
|
61
63
|
*/
|
|
62
64
|
export function estimateCost(
|
|
63
65
|
model: string | undefined,
|
package/src/o11y/derive.test.ts
CHANGED
|
@@ -58,6 +58,46 @@ describe("deriveRunFacts:pending 折叠", () => {
|
|
|
58
58
|
});
|
|
59
59
|
});
|
|
60
60
|
|
|
61
|
+
describe("deriveRunFacts:callId 跨轮复用不覆盖", () => {
|
|
62
|
+
// 回归:adapter 常按轮各自编号(OpenAI 兼容 / transcript 归一复用 c1/c2…)。同一 callId 在其
|
|
63
|
+
// result 之后再次出现是新的一次调用——旧折叠按 callId 存 Map 会被后一轮覆盖,跨轮聚合断言
|
|
64
|
+
// (t.calledTool)于是「只扫最后一轮」,前几轮的工具调用被抹掉(定稿见 events.md 不变量 2)。
|
|
65
|
+
it("同一 callId 完成后在后一轮再次 called,折叠成两条独立调用而非覆盖成一条", () => {
|
|
66
|
+
const events: StreamEvent[] = [
|
|
67
|
+
// 第一轮:读 init、读 INDEX(都用 c1/c2 编号)
|
|
68
|
+
{ type: "action.called", callId: "c1", name: "read", input: { path: "init.md" } },
|
|
69
|
+
{ type: "action.result", callId: "c1", output: "init contents", status: "completed" },
|
|
70
|
+
{ type: "action.called", callId: "c2", name: "read", input: { path: "INDEX.md" } },
|
|
71
|
+
{ type: "action.result", callId: "c2", output: "index contents", status: "completed" },
|
|
72
|
+
{ type: "message", role: "assistant", text: "第一轮做完" },
|
|
73
|
+
// 第二轮:续轮,adapter 从 c1 重新编号
|
|
74
|
+
{ type: "action.called", callId: "c1", name: "write", input: { path: "note.md" } },
|
|
75
|
+
{ type: "action.result", callId: "c1", output: "ok", status: "completed" },
|
|
76
|
+
{ type: "message", role: "assistant", text: "答复" },
|
|
77
|
+
];
|
|
78
|
+
const facts = deriveRunFacts(events);
|
|
79
|
+
expect(facts.toolCalls).toHaveLength(3); // 两轮共 3 次调用,后一轮不覆盖前一轮
|
|
80
|
+
expect(facts.toolCalls.map((tc) => tc.originalName)).toEqual(["read", "read", "write"]);
|
|
81
|
+
// 第一轮的两次 read 都在,且各配对到自己的 result(不是都取最后一条)
|
|
82
|
+
expect(facts.toolCalls[0]!.input).toEqual({ path: "init.md" });
|
|
83
|
+
expect(facts.toolCalls[0]!.output).toBe("init contents");
|
|
84
|
+
expect(facts.toolCalls[1]!.input).toEqual({ path: "INDEX.md" });
|
|
85
|
+
expect(facts.toolCalls[1]!.output).toBe("index contents");
|
|
86
|
+
});
|
|
87
|
+
|
|
88
|
+
it("subagent 折叠同理:同一 callId 完成后再次 called 起新记录", () => {
|
|
89
|
+
const events: StreamEvent[] = [
|
|
90
|
+
{ type: "subagent.called", callId: "s1", name: "researcher" },
|
|
91
|
+
{ type: "subagent.completed", callId: "s1", output: "round1", status: "completed" },
|
|
92
|
+
{ type: "subagent.called", callId: "s1", name: "writer" },
|
|
93
|
+
{ type: "subagent.completed", callId: "s1", output: "round2", status: "completed" },
|
|
94
|
+
];
|
|
95
|
+
const facts = deriveRunFacts(events);
|
|
96
|
+
expect(facts.subagentCalls).toHaveLength(2);
|
|
97
|
+
expect(facts.subagentCalls.map((sc) => sc.name)).toEqual(["researcher", "writer"]);
|
|
98
|
+
});
|
|
99
|
+
});
|
|
100
|
+
|
|
61
101
|
describe("deriveRunFacts:contextInjections 计数", () => {
|
|
62
102
|
it("统计事件流里 context.injected 事件的次数,不与 messageCount 混计", () => {
|
|
63
103
|
const events: StreamEvent[] = [
|
package/src/o11y/derive.ts
CHANGED
|
@@ -61,10 +61,16 @@ function pickExitCode(output: JsonValue | undefined): number | undefined {
|
|
|
61
61
|
// ───────────────────────── deriveRunFacts ─────────────────────────
|
|
62
62
|
|
|
63
63
|
export function deriveRunFacts(events: readonly StreamEvent[]): DerivedFacts {
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
|
|
64
|
+
// 折叠是逐条按发生顺序进行的:called 追加一条新调用,result 回填「当前还没配上 result 的
|
|
65
|
+
// 同 callId 调用」。callId 只在一个 called→result 配对内保证稳定,不保证跨轮唯一——adapter
|
|
66
|
+
// 常按轮各自编号(OpenAI 兼容协议、transcript 归一都会复用 c1/c2…)。所以一个 callId 在其
|
|
67
|
+
// result 之后再次以 called 出现,是新的一次调用,起一条新记录,不覆盖前一轮那条(否则跨轮
|
|
68
|
+
// 聚合会把前面几轮的工具调用抹成「只剩最后一轮」)。用 open*ByCallId 只跟踪各 callId 当前
|
|
69
|
+
// 敞口的那条,配上 result 即关闭。
|
|
70
|
+
const toolCalls: ToolCall[] = [];
|
|
71
|
+
const openToolByCallId = new Map<string, number>();
|
|
72
|
+
const subagentCalls: SubagentCall[] = [];
|
|
73
|
+
const openSubagentByCallId = new Map<string, number>();
|
|
68
74
|
const inputRequests: InputRequest[] = [];
|
|
69
75
|
let messageCount = 0;
|
|
70
76
|
let compactions = 0;
|
|
@@ -81,8 +87,8 @@ export function deriveRunFacts(events: readonly StreamEvent[]): DerivedFacts {
|
|
|
81
87
|
break;
|
|
82
88
|
|
|
83
89
|
case "action.called": {
|
|
84
|
-
|
|
85
|
-
|
|
90
|
+
openToolByCallId.set(ev.callId, toolCalls.length);
|
|
91
|
+
toolCalls.push({
|
|
86
92
|
callId: ev.callId,
|
|
87
93
|
name: ev.tool ?? "unknown",
|
|
88
94
|
originalName: ev.name,
|
|
@@ -94,14 +100,14 @@ export function deriveRunFacts(events: readonly StreamEvent[]): DerivedFacts {
|
|
|
94
100
|
}
|
|
95
101
|
|
|
96
102
|
case "action.result": {
|
|
97
|
-
const
|
|
98
|
-
if (
|
|
99
|
-
|
|
100
|
-
|
|
103
|
+
const idx = openToolByCallId.get(ev.callId);
|
|
104
|
+
if (idx !== undefined) {
|
|
105
|
+
toolCalls[idx].output = ev.output;
|
|
106
|
+
toolCalls[idx].status = ev.status;
|
|
107
|
+
openToolByCallId.delete(ev.callId);
|
|
101
108
|
} else {
|
|
102
109
|
// 只有结果、没配上调用:补一条占位 ToolCall。
|
|
103
|
-
|
|
104
|
-
toolCallMap.set(ev.callId, {
|
|
110
|
+
toolCalls.push({
|
|
105
111
|
callId: ev.callId,
|
|
106
112
|
name: "unknown",
|
|
107
113
|
input: null,
|
|
@@ -113,8 +119,8 @@ export function deriveRunFacts(events: readonly StreamEvent[]): DerivedFacts {
|
|
|
113
119
|
}
|
|
114
120
|
|
|
115
121
|
case "subagent.called": {
|
|
116
|
-
|
|
117
|
-
|
|
122
|
+
openSubagentByCallId.set(ev.callId, subagentCalls.length);
|
|
123
|
+
subagentCalls.push({
|
|
118
124
|
callId: ev.callId,
|
|
119
125
|
name: ev.name,
|
|
120
126
|
remoteUrl: ev.remoteUrl,
|
|
@@ -125,13 +131,13 @@ export function deriveRunFacts(events: readonly StreamEvent[]): DerivedFacts {
|
|
|
125
131
|
}
|
|
126
132
|
|
|
127
133
|
case "subagent.completed": {
|
|
128
|
-
const
|
|
129
|
-
if (
|
|
130
|
-
|
|
131
|
-
|
|
134
|
+
const idx = openSubagentByCallId.get(ev.callId);
|
|
135
|
+
if (idx !== undefined) {
|
|
136
|
+
subagentCalls[idx].output = ev.output;
|
|
137
|
+
subagentCalls[idx].status = ev.status;
|
|
138
|
+
openSubagentByCallId.delete(ev.callId);
|
|
132
139
|
} else {
|
|
133
|
-
|
|
134
|
-
subagentMap.set(ev.callId, {
|
|
140
|
+
subagentCalls.push({
|
|
135
141
|
callId: ev.callId,
|
|
136
142
|
name: "unknown",
|
|
137
143
|
output: ev.output,
|
|
@@ -164,8 +170,8 @@ export function deriveRunFacts(events: readonly StreamEvent[]): DerivedFacts {
|
|
|
164
170
|
}
|
|
165
171
|
|
|
166
172
|
return {
|
|
167
|
-
toolCalls
|
|
168
|
-
subagentCalls
|
|
173
|
+
toolCalls,
|
|
174
|
+
subagentCalls,
|
|
169
175
|
inputRequests,
|
|
170
176
|
parked,
|
|
171
177
|
messageCount,
|
|
@@ -0,0 +1,201 @@
|
|
|
1
|
+
// cases: docs/engineering/testing/unit/experiments-runner.md
|
|
2
|
+
// bug: memory/insandbox-otlp-port-wait-3s-no-retry.md
|
|
3
|
+
// 沙箱内 OTLP 采集器的启动韧性:这条路径外面没有任何重试兜底(命令执行不进 IO 重试、
|
|
4
|
+
// provision 重试只覆盖 create、runner 没有 attempt 级重试),起不来就是一条 errored attempt。
|
|
5
|
+
// 两层 fixture:
|
|
6
|
+
// · 脚本化 fake sandbox —— 重试轮次、每轮换路径、重试前杀上一轮,确定性断言;
|
|
7
|
+
// · 真实 /bin/sh 执行 —— 生成的 shell 脚本本身能跑(语法、算术、退出边),以及采集器进程
|
|
8
|
+
// 一起来就死时不空等满预算(观察实测耗时,不看脚本字节)。
|
|
9
|
+
|
|
10
|
+
import { mkdtemp, rm, writeFile, chmod, readdir } from "node:fs/promises";
|
|
11
|
+
import { spawn } from "node:child_process";
|
|
12
|
+
import { tmpdir } from "node:os";
|
|
13
|
+
import { join } from "node:path";
|
|
14
|
+
import { Effect, Exit, Cause } from "effect";
|
|
15
|
+
import { afterEach, describe, expect, it } from "vitest";
|
|
16
|
+
import type { CommandResult, Sandbox } from "../../types.ts";
|
|
17
|
+
import { createInSandboxTraceReceiver } from "./sandbox-receiver.ts";
|
|
18
|
+
|
|
19
|
+
const notUsed = () => {
|
|
20
|
+
throw new Error("not used by the in-sandbox receiver");
|
|
21
|
+
};
|
|
22
|
+
|
|
23
|
+
function baseSandbox(overrides: Partial<Sandbox>): Sandbox {
|
|
24
|
+
return {
|
|
25
|
+
workdir: "/work",
|
|
26
|
+
sandboxId: "fake",
|
|
27
|
+
otlpHost: null,
|
|
28
|
+
runCommand: notUsed,
|
|
29
|
+
runShell: notUsed,
|
|
30
|
+
readFile: notUsed,
|
|
31
|
+
fileExists: notUsed,
|
|
32
|
+
writeFiles: async () => {},
|
|
33
|
+
uploadFiles: notUsed,
|
|
34
|
+
uploadDirectory: notUsed,
|
|
35
|
+
downloadDirectory: notUsed,
|
|
36
|
+
downloadFile: notUsed,
|
|
37
|
+
uploadFile: notUsed,
|
|
38
|
+
stop: async () => {},
|
|
39
|
+
...overrides,
|
|
40
|
+
} as Sandbox;
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
const isStart = (script: string) => script.includes("niceeval-otlp-collector") && script.includes("&");
|
|
44
|
+
const isKill = (script: string) => script.trimStart().startsWith("kill ");
|
|
45
|
+
|
|
46
|
+
/**
|
|
47
|
+
* 启动脚本按轮次回放一条预置 stdout(`PID\n端口`,端口空表示这一轮没等到),日志读取回放
|
|
48
|
+
* 固定内容,kill 一律成功。按脚本形态派发而不是按调用序号——被测实现每轮发几条命令是它自己
|
|
49
|
+
* 的事,fixture 不该把这个数字焊死。
|
|
50
|
+
*/
|
|
51
|
+
function scriptedSandbox(startOutputs: string[], log = "") {
|
|
52
|
+
const shells: string[] = [];
|
|
53
|
+
const written: string[] = [];
|
|
54
|
+
const sandbox = baseSandbox({
|
|
55
|
+
writeFiles: async (files: Record<string, string>) => {
|
|
56
|
+
written.push(...Object.keys(files));
|
|
57
|
+
},
|
|
58
|
+
runShell: async (script: string): Promise<CommandResult> => {
|
|
59
|
+
shells.push(script);
|
|
60
|
+
const stdout = isStart(script) ? (startOutputs.shift() ?? "") : script.startsWith("cat ") ? log : "";
|
|
61
|
+
return { stdout, stderr: "", exitCode: 0 };
|
|
62
|
+
},
|
|
63
|
+
});
|
|
64
|
+
return { sandbox, shells, written };
|
|
65
|
+
}
|
|
66
|
+
|
|
67
|
+
/** 在 Scope 内取端点,并把「此刻为止发过的命令」快照出来——release 阶段的收尾 kill 不混进来。 */
|
|
68
|
+
async function receiverExit(sandbox: Sandbox, shells: string[] = []) {
|
|
69
|
+
return Effect.runPromiseExit(
|
|
70
|
+
Effect.scoped(
|
|
71
|
+
Effect.gen(function* () {
|
|
72
|
+
const receiver = yield* createInSandboxTraceReceiver(sandbox);
|
|
73
|
+
return { endpoint: receiver.endpoint(""), shells: [...shells] };
|
|
74
|
+
}),
|
|
75
|
+
),
|
|
76
|
+
);
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
describe("沙箱内 OTLP 采集器启动:脚本化 fixture", () => {
|
|
80
|
+
it("第一轮拿不到端口时重试,换一套路径并先杀掉上一轮的进程", async () => {
|
|
81
|
+
// 第一轮:拿到 PID 但端口行为空(采集器慢/死);第二轮:端口写回来了。
|
|
82
|
+
const { sandbox, shells, written } = scriptedSandbox(["4242\n", "4243\n61000"]);
|
|
83
|
+
|
|
84
|
+
const exit = await receiverExit(sandbox, shells);
|
|
85
|
+
|
|
86
|
+
expect(Exit.isSuccess(exit) && exit.value.endpoint).toBe("http://127.0.0.1:61000/v1/traces");
|
|
87
|
+
const sent = Exit.isSuccess(exit) ? exit.value.shells : [];
|
|
88
|
+
expect(sent.filter(isStart)).toHaveLength(2);
|
|
89
|
+
// 上一轮那个「慢但还活着」的采集器必须被杀掉:留着它会占内存,还会在重试之后才把端口
|
|
90
|
+
// 写进它自己那份文件。
|
|
91
|
+
expect(sent.filter(isKill)).toEqual([expect.stringContaining("kill 4242")]);
|
|
92
|
+
// 每轮换一套带随机后缀的脚本路径:上一轮迟到的采集器写不进新一轮的端口 / spans 文件。
|
|
93
|
+
expect(written).toHaveLength(2);
|
|
94
|
+
expect(written[0]).not.toBe(written[1]);
|
|
95
|
+
});
|
|
96
|
+
|
|
97
|
+
it("首轮就拿到端口时只起一次,不杀任何进程", async () => {
|
|
98
|
+
const { sandbox, shells } = scriptedSandbox(["4242\n61001"]);
|
|
99
|
+
|
|
100
|
+
const exit = await receiverExit(sandbox, shells);
|
|
101
|
+
|
|
102
|
+
expect(Exit.isSuccess(exit) && exit.value.endpoint).toBe("http://127.0.0.1:61001/v1/traces");
|
|
103
|
+
const sent = Exit.isSuccess(exit) ? exit.value.shells : [];
|
|
104
|
+
expect(sent.filter(isStart)).toHaveLength(1);
|
|
105
|
+
expect(sent.filter(isKill)).toEqual([]);
|
|
106
|
+
});
|
|
107
|
+
|
|
108
|
+
it("重试用尽后抛错,带上预算、轮次与采集器自己的日志", async () => {
|
|
109
|
+
const { sandbox, shells } = scriptedSandbox(["4242\n", "4243\n"], "node: not found");
|
|
110
|
+
|
|
111
|
+
const exit = await receiverExit(sandbox, shells);
|
|
112
|
+
|
|
113
|
+
expect(Exit.isFailure(exit)).toBe(true);
|
|
114
|
+
const message = Exit.isFailure(exit) ? Cause.squash(exit.cause) : undefined;
|
|
115
|
+
expect(String(message)).toContain("within 20s");
|
|
116
|
+
expect(String(message)).toContain("2 attempts");
|
|
117
|
+
expect(String(message)).toContain("node: not found");
|
|
118
|
+
expect(shells.filter(isStart)).toHaveLength(2);
|
|
119
|
+
});
|
|
120
|
+
});
|
|
121
|
+
|
|
122
|
+
describe("沙箱内 OTLP 采集器启动:真实 /bin/sh 执行", () => {
|
|
123
|
+
const dirs: string[] = [];
|
|
124
|
+
const pids: number[] = [];
|
|
125
|
+
|
|
126
|
+
afterEach(async () => {
|
|
127
|
+
for (const pid of pids.splice(0)) {
|
|
128
|
+
try {
|
|
129
|
+
process.kill(pid);
|
|
130
|
+
} catch {
|
|
131
|
+
// 已经退出
|
|
132
|
+
}
|
|
133
|
+
}
|
|
134
|
+
for (const dir of dirs.splice(0)) await rm(dir, { recursive: true, force: true });
|
|
135
|
+
// 本用例真的会往 /tmp 写采集器脚本 / 端口文件,跑完清掉自己那批。
|
|
136
|
+
for (const name of await readdir(tmpdir())) {
|
|
137
|
+
if (name.startsWith(".niceeval-otlp-")) await rm(join(tmpdir(), name), { force: true }).catch(() => {});
|
|
138
|
+
}
|
|
139
|
+
});
|
|
140
|
+
|
|
141
|
+
/** runShell 真的交给 /bin/sh 跑,writeFiles 真的落盘——生成的脚本语法错误在这里现形。 */
|
|
142
|
+
function shellSandbox(pathPrefix?: string) {
|
|
143
|
+
return baseSandbox({
|
|
144
|
+
writeFiles: async (files: Record<string, string>) => {
|
|
145
|
+
for (const [path, content] of Object.entries(files)) await writeFile(path, content);
|
|
146
|
+
},
|
|
147
|
+
runShell: (script: string) =>
|
|
148
|
+
new Promise<CommandResult>((resolve) => {
|
|
149
|
+
const child = spawn("/bin/sh", ["-c", script], {
|
|
150
|
+
env: { ...process.env, ...(pathPrefix ? { PATH: `${pathPrefix}:${process.env.PATH}` } : {}) },
|
|
151
|
+
});
|
|
152
|
+
let stdout = "";
|
|
153
|
+
let stderr = "";
|
|
154
|
+
child.stdout.on("data", (c) => (stdout += c));
|
|
155
|
+
child.stderr.on("data", (c) => (stderr += c));
|
|
156
|
+
child.on("close", (code) => resolve({ stdout, stderr, exitCode: code ?? 0 }));
|
|
157
|
+
}),
|
|
158
|
+
});
|
|
159
|
+
}
|
|
160
|
+
|
|
161
|
+
it("真实 shell 下起得来:端口写回宿主,采集器确实在监听", async () => {
|
|
162
|
+
const sandbox = shellSandbox();
|
|
163
|
+
|
|
164
|
+
const result = await Effect.runPromise(
|
|
165
|
+
Effect.scoped(
|
|
166
|
+
Effect.gen(function* () {
|
|
167
|
+
const receiver = yield* createInSandboxTraceReceiver(sandbox);
|
|
168
|
+
const endpoint = receiver.endpoint("");
|
|
169
|
+
const status = yield* Effect.promise(async () => {
|
|
170
|
+
const res = await fetch(endpoint, {
|
|
171
|
+
method: "POST",
|
|
172
|
+
headers: { "content-type": "application/json" },
|
|
173
|
+
body: "{}",
|
|
174
|
+
});
|
|
175
|
+
return res.status;
|
|
176
|
+
});
|
|
177
|
+
return { endpoint, status };
|
|
178
|
+
}),
|
|
179
|
+
),
|
|
180
|
+
);
|
|
181
|
+
|
|
182
|
+
expect(result.endpoint).toMatch(/^http:\/\/127\.0\.0\.1:\d+\/v1\/traces$/);
|
|
183
|
+
expect(result.status).toBe(200);
|
|
184
|
+
});
|
|
185
|
+
|
|
186
|
+
it("采集器一起来就死时不空等满预算,错误里带着它的日志", async () => {
|
|
187
|
+
const shimDir = await mkdtemp(join(tmpdir(), "niceeval-node-shim-"));
|
|
188
|
+
dirs.push(shimDir);
|
|
189
|
+
await writeFile(join(shimDir, "node"), '#!/bin/sh\necho "node: simulated crash" >&2\nexit 1\n');
|
|
190
|
+
await chmod(join(shimDir, "node"), 0o755);
|
|
191
|
+
|
|
192
|
+
const startedAt = Date.now();
|
|
193
|
+
const exit = await receiverExit(shellSandbox(shimDir));
|
|
194
|
+
const elapsed = Date.now() - startedAt;
|
|
195
|
+
|
|
196
|
+
expect(Exit.isFailure(exit)).toBe(true);
|
|
197
|
+
expect(String(Exit.isFailure(exit) ? Cause.squash(exit.cause) : "")).toContain("node: simulated crash");
|
|
198
|
+
// 进程已死时循环立刻 break,两轮加起来也远小于一轮的等待预算(20s)。
|
|
199
|
+
expect(elapsed).toBeLessThan(5_000);
|
|
200
|
+
});
|
|
201
|
+
});
|
|
@@ -11,7 +11,8 @@
|
|
|
11
11
|
// 6. close() 尝试 kill PID(沙箱本身也会停,所以 best-effort)
|
|
12
12
|
//
|
|
13
13
|
// 文件路径带随机后缀:同一沙箱跨 eval 复用时,每个 receiver 实例的脚本 / spans /
|
|
14
|
-
// 端口文件互不串扰,collector 也不会读到上一个 eval 的 span
|
|
14
|
+
// 端口文件互不串扰,collector 也不会读到上一个 eval 的 span。启动重试同样每轮换一套后缀
|
|
15
|
+
// (见 startCollector)。
|
|
15
16
|
|
|
16
17
|
import { randomUUID } from "node:crypto";
|
|
17
18
|
import { Effect } from "effect";
|
|
@@ -56,6 +57,16 @@ server.listen(0, '127.0.0.1', () => {
|
|
|
56
57
|
`;
|
|
57
58
|
}
|
|
58
59
|
|
|
60
|
+
// 端口等待预算。这条路径外面没有任何一层重试兜着——`runShell` 不进 `withSandboxIoRetry`
|
|
61
|
+
// (命令执行有不可重复副作用),`withProvisionRetry` 只覆盖沙箱 create,runner 也没有
|
|
62
|
+
// attempt 级重试:等不到端口就是一条 errored attempt,agent 一次都没跑
|
|
63
|
+
// (memory/insandbox-otlp-port-wait-3s-no-retry.md)。所以预算按冷沙箱首次起 node 的最坏
|
|
64
|
+
// 情况给,而不是按热沙箱的常见值给。
|
|
65
|
+
const PORT_WAIT_MS = 20_000;
|
|
66
|
+
// 启动整体的重试次数。慢不需要重试(一次等满预算就够),这里兜的是 collector 起来就死
|
|
67
|
+
// (镜像里没有 node、脚本被 OOM kill 等)——那种情况下沙箱侧循环会立刻 break,重试很便宜。
|
|
68
|
+
const START_ATTEMPTS = 2;
|
|
69
|
+
|
|
59
70
|
export function createInSandboxTraceReceiver(sandbox: Sandbox) {
|
|
60
71
|
return Effect.acquireRelease(
|
|
61
72
|
Effect.promise(() => makeInSandboxReceiver(sandbox)),
|
|
@@ -63,35 +74,70 @@ export function createInSandboxTraceReceiver(sandbox: Sandbox) {
|
|
|
63
74
|
);
|
|
64
75
|
}
|
|
65
76
|
|
|
66
|
-
|
|
67
|
-
|
|
77
|
+
interface StartedCollector {
|
|
78
|
+
pid: number;
|
|
79
|
+
port: number;
|
|
80
|
+
spansPath: string;
|
|
81
|
+
}
|
|
68
82
|
|
|
69
|
-
|
|
70
|
-
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
|
|
83
|
+
/**
|
|
84
|
+
* 上传脚本 + 后台起 collector + 等端口文件,失败重试。每一轮换一套带随机后缀的路径:
|
|
85
|
+
* 上一轮那个「慢但还活着」的 collector 如果在重试之后才起来,会写回它自己那份端口 / spans
|
|
86
|
+
* 文件,不会污染新一轮。
|
|
87
|
+
*/
|
|
88
|
+
async function startCollector(sandbox: Sandbox): Promise<StartedCollector> {
|
|
89
|
+
let lastLog = "";
|
|
90
|
+
for (let attempt = 1; ; attempt++) {
|
|
91
|
+
const tag = randomUUID().slice(0, 8);
|
|
92
|
+
const collectorPath = `/tmp/.niceeval-otlp-collector-${tag}.cjs`;
|
|
93
|
+
const spansPath = `/tmp/.niceeval-otlp-spans-${tag}.jsonl`;
|
|
94
|
+
const portPath = `/tmp/.niceeval-otlp-port-${tag}`;
|
|
95
|
+
const logPath = `/tmp/.niceeval-otlp-collector-${tag}.log`;
|
|
96
|
+
|
|
97
|
+
await sandbox.writeFiles({ [collectorPath]: collectorScript(spansPath, portPath) });
|
|
98
|
+
|
|
99
|
+
// 后台启动 + 等端口文件,折进一次 shell 往返(远程沙箱一次 exec 要 100-500ms,
|
|
100
|
+
// host 侧逐次轮询会把几秒的启动等待放大成 N 个 API round-trip)。循环两条退出边:
|
|
101
|
+
// · `kill -0` 失败 → collector 已经死了,别再空等满预算,立刻回 host 重试;
|
|
102
|
+
// · 到 deadline → 真的太慢。
|
|
103
|
+
// 预算按 `date +%s` 的墙钟算而不是数 tick:`sleep` 的小数秒支持因镜像而异,数 tick 会让
|
|
104
|
+
// 不支持小数的镜像瞬间跑完循环、伪装成「等了 20s」。
|
|
105
|
+
// 输出两行:PID、端口(等不到则空)。
|
|
106
|
+
const startResult = await sandbox.runShell(
|
|
107
|
+
`node ${collectorPath} >${logPath} 2>&1 & pid=$!; echo $pid; ` +
|
|
108
|
+
`end=$(( $(date +%s) + ${Math.ceil(PORT_WAIT_MS / 1000)} )); ` +
|
|
109
|
+
`while [ ! -s ${portPath} ]; do ` +
|
|
110
|
+
`kill -0 $pid 2>/dev/null || break; ` +
|
|
111
|
+
`[ "$(date +%s)" -lt "$end" ] || break; ` +
|
|
112
|
+
`sleep 0.1 2>/dev/null || sleep 1; ` +
|
|
113
|
+
`done; ` +
|
|
114
|
+
`cat ${portPath} 2>/dev/null || true`,
|
|
93
115
|
);
|
|
116
|
+
const [pidLine, portLine] = startResult.stdout.trim().split("\n");
|
|
117
|
+
const pid = parseInt((pidLine ?? "").trim(), 10);
|
|
118
|
+
const port = parseInt((portLine ?? "").trim(), 10) || 0;
|
|
119
|
+
if (port) return { pid, port, spansPath };
|
|
120
|
+
|
|
121
|
+
const log = await sandbox.runShell(`cat ${logPath} 2>/dev/null || true`).catch(() => undefined);
|
|
122
|
+
lastLog = log?.stdout.trim() ?? "";
|
|
123
|
+
// 这一轮可能只是慢、进程还活着:重试前先杀掉,不留孤儿 collector 占内存
|
|
124
|
+
//(沙箱复用 / --keep-sandbox 下它会一直在)。
|
|
125
|
+
if (Number.isFinite(pid) && pid > 0) {
|
|
126
|
+
await sandbox.runShell(`kill ${pid} 2>/dev/null || true`).catch(() => {});
|
|
127
|
+
}
|
|
128
|
+
if (attempt >= START_ATTEMPTS) {
|
|
129
|
+
throw new Error(
|
|
130
|
+
`in-sandbox OTLP collector failed to report its port within ${Math.round(PORT_WAIT_MS / 1000)}s ` +
|
|
131
|
+
`(${attempt} attempts). Collector log:\n${lastLog || "(empty)"}`,
|
|
132
|
+
);
|
|
133
|
+
}
|
|
94
134
|
}
|
|
135
|
+
}
|
|
136
|
+
|
|
137
|
+
async function makeInSandboxReceiver(sandbox: Sandbox): Promise<TraceReceiver> {
|
|
138
|
+
let cached: TraceSpan[] = [];
|
|
139
|
+
|
|
140
|
+
const { pid, port, spansPath } = await startCollector(sandbox);
|
|
95
141
|
|
|
96
142
|
return {
|
|
97
143
|
endpoint: (_host) => `http://127.0.0.1:${port}/v1/traces`,
|
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
// cases: docs/engineering/testing/unit/results.md
|
|
2
|
+
// 「Usage、facts 与失败命令证据落盘」桶恒互斥归一:bub tape 的 usage 是 Chat Completions
|
|
3
|
+
// 形状,prompt_tokens_details.cached_tokens 是 prompt_tokens 子集,聚合前扣减;
|
|
4
|
+
// 同对象的 cost 是实测计费,照常累进 costUSD。
|
|
5
|
+
// bug: memory/estimatecost-openai-inclusive-cache-double-billed.md
|
|
6
|
+
|
|
7
|
+
import { describe, expect, it } from "vitest";
|
|
8
|
+
|
|
9
|
+
import { parseBubTranscript } from "./bub.ts";
|
|
10
|
+
|
|
11
|
+
describe("parseBubTranscript usage 归一(OpenAI 口径)", () => {
|
|
12
|
+
it("cached_tokens 从 prompt_tokens 里扣出,cost 累进实测 costUSD", () => {
|
|
13
|
+
const line = JSON.stringify({
|
|
14
|
+
kind: "event",
|
|
15
|
+
payload: {
|
|
16
|
+
name: "run",
|
|
17
|
+
data: {
|
|
18
|
+
usage: {
|
|
19
|
+
prompt_tokens: 1000,
|
|
20
|
+
completion_tokens: 20,
|
|
21
|
+
prompt_tokens_details: { cached_tokens: 900 },
|
|
22
|
+
cost: 0.05,
|
|
23
|
+
},
|
|
24
|
+
},
|
|
25
|
+
},
|
|
26
|
+
});
|
|
27
|
+
const parsed = parseBubTranscript(line);
|
|
28
|
+
expect(parsed.usage).toMatchObject({ inputTokens: 100, cacheReadTokens: 900, outputTokens: 20, costUSD: 0.05 });
|
|
29
|
+
});
|
|
30
|
+
});
|